diff --git a/.github/workflows/pr-checks.yml b/.github/workflows/pr-checks.yml index 2a4be67b..662f3720 100644 --- a/.github/workflows/pr-checks.yml +++ b/.github/workflows/pr-checks.yml @@ -58,7 +58,7 @@ jobs: - name: Run tests run: go test -v -race -coverprofile=coverage.out -covermode=atomic ./... - - name: Run installer tests + - name: Run installer and bootstrap script tests run: make test-install - name: Generate coverage report diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 1da86db0..a713a211 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -435,8 +435,8 @@ jobs: **Manual installation:** 1. Download the appropriate binary for your platform 2. Extract the archive: \`tar -xzf aks-flex-node-*.tar.gz\` - 3. Move the binary to your PATH: \`sudo mv aks-flex-node-* /usr/local/bin/aks-flex-node\` - 4. Make it executable: \`sudo chmod +x /usr/local/bin/aks-flex-node\` + 3. Install the binary where the agent looks for it: \`sudo install -D -m 0755 aks-flex-node-linux-amd64 /opt/unbounded/bin/aks-flex-node\`, or \`aks-flex-node-linux-arm64\` on ARM64 + 4. Run it as \`/opt/unbounded/bin/aks-flex-node\`, or add \`/opt/unbounded/bin\` to your PATH ### Supported Platforms diff --git a/Makefile b/Makefile index 86fd5aac..eba02657 100644 --- a/Makefile +++ b/Makefile @@ -60,6 +60,8 @@ test: test-install test-install: @echo "Running installer tests..." @scripts/install_test.sh + @scripts/uninstall_test.sh + @scripts/bootstrap_test.sh .PHONY: test-coverage test-coverage: @@ -195,7 +197,7 @@ help: @echo "" @echo "Test & Quality Targets:" @echo " test Run tests" - @echo " test-install Run installer tests" + @echo " test-install Run installer, uninstaller, and bootstrap script tests" @echo " test-coverage Run tests with coverage report" @echo " test-race Run tests with race detector" @echo " lint Run golangci-lint" diff --git a/cmd/aks-flex-node/main.go b/cmd/aks-flex-node/main.go index a2f26019..056dd6b0 100644 --- a/cmd/aks-flex-node/main.go +++ b/cmd/aks-flex-node/main.go @@ -12,6 +12,8 @@ import ( "github.com/Azure/AKSFlexNode/pkg/cmd/bootstrapdata" "github.com/Azure/AKSFlexNode/pkg/cmd/daemon" + "github.com/Azure/AKSFlexNode/pkg/cmd/hostroot" + "github.com/Azure/AKSFlexNode/pkg/cmd/ignition" "github.com/Azure/AKSFlexNode/pkg/cmd/nspawnlifecycle" "github.com/Azure/AKSFlexNode/pkg/cmd/preflight" "github.com/Azure/AKSFlexNode/pkg/cmd/reset" @@ -47,10 +49,12 @@ func newRootCommand() *cobra.Command { rootCmd.AddCommand(start.NewCommand()) rootCmd.AddCommand(bootstrapdata.NewCommand()) rootCmd.AddCommand(preflight.NewCommand()) + rootCmd.AddCommand(ignition.NewCommand()) rootCmd.AddCommand(daemon.NewCommands()...) rootCmd.AddCommand(nspawnlifecycle.NewCommand()) rootCmd.AddCommand(reset.NewCommand()) rootCmd.AddCommand(version.NewCommand()) + rootCmd.AddCommand(hostroot.NewCommand()) rootCmd.AddCommand(token.Command) return rootCmd diff --git a/cmd/aks-flex-node/main_test.go b/cmd/aks-flex-node/main_test.go index 257e4017..694c914f 100644 --- a/cmd/aks-flex-node/main_test.go +++ b/cmd/aks-flex-node/main_test.go @@ -16,3 +16,25 @@ func TestRootCommandRegistersGeneratedNSpawnLifecycleShape(t *testing.T) { t.Fatalf("Find() remaining args = %v, want [kube1]", remaining) } } + +// TestRootCommandRegistersTopLevelCommands is not parallel. newRootCommand +// attaches the package-level token.Command, so building two roots at once races +// on it. The subtests share one root for the same reason. +// +// host-root is what the install scripts and AgentUpgrade ask a release for, so +// a build that dropped it would be taken for a release before the host root. +func TestRootCommandRegistersTopLevelCommands(t *testing.T) { + root := newRootCommand() + + for _, name := range []string{"ignition", "host-root"} { + t.Run(name, func(t *testing.T) { + cmd, _, err := root.Find([]string{name}) + if err != nil { + t.Fatalf("Find() error = %v", err) + } + if cmd.Name() != name { + t.Fatalf("Find() command = %q, want %s", cmd.Name(), name) + } + }) + } +} diff --git a/docs/design/storage-backed-bootstrap.md b/docs/design/storage-backed-bootstrap.md index a40382a4..013b52bc 100644 --- a/docs/design/storage-backed-bootstrap.md +++ b/docs/design/storage-backed-bootstrap.md @@ -265,6 +265,16 @@ how to retry failed provisioning. A systemd oneshot wrapper can use `ConditionPathExists=!/var/lib/aks-flex-node/first-boot-complete` to prevent a successful node from being bootstrapped again after reboot. +`aks-flex-node ignition` renders such a wrapper for hosts provisioned by +Ignition, `aks-flex-node-bootstrap.service`, but conditions it on the agent +unit, `/etc/systemd/system/aks-flex-node-agent.service`, instead of a marker. +The agent unit exists once bootstrap has installed the agent, and +`aks-flex-node reset` removes it together with the wrapper unit. A marker under +`/var/lib/aks-flex-node` would survive reset and block the host from being +provisioned again. The wrapper removes the script, which carries the base +config, once bootstrap succeeds, and keeps the credential file that the config +references. + ### 7. Verify convergence The provisioning system should not treat script exit alone as complete cluster @@ -408,6 +418,10 @@ AKS_FLEX_NODE_INSTALL_DIR AKS_FLEX_NODE_CONFIG_PATH ``` +`AKS_FLEX_NODE_INSTALL_DIR` and `--install-dir` are deprecated. The binary +directory is the one the agent reports, and a value that names any other +directory is rejected. + The equivalent non-secret values have CLI flags. A service-principal client secret has no CLI value because command arguments are process-visible. Use a protected secret file or, when unavoidable, the dedicated environment variable. @@ -426,20 +440,22 @@ The script processes JSON in this order: 1. Write the embedded base config into a mode `0700` temporary workspace. 2. Validate that it is a JSON object. -3. Apply dedicated cluster resource ID, pool name, and ARM endpoint overrides so +3. Download the agent and install it where it reports, as described in + [Agent download and installation](#agent-download-and-installation). This + happens before rendering because fetching bootstrap data runs the installed + binary. +4. Apply dedicated cluster resource ID, pool name, and ARM endpoint overrides so they are available to the bootstrap-data request. -4. When enabled, acquire an ARM token with MSI or SP, call +5. When enabled, acquire an ARM token with MSI or SP, call `listBootstrapData`, and deep-merge the response. -5. Deep-merge `AKS_FLEX_NODE_CONFIG_OVERRIDES`, when present. -6. Deep-merge each CLI `--config-overrides` object in invocation order. -7. Reapply dedicated cluster/pool/endpoint overrides so they remain +6. Deep-merge `AKS_FLEX_NODE_CONFIG_OVERRIDES`, when present. +7. Deep-merge each CLI `--config-overrides` object in invocation order. +8. Reapply dedicated cluster/pool/endpoint overrides so they remain authoritative. -8. Apply dedicated rootfs and offline-artifact source overrides. -9. Set `agent.nodeName` from the lowercase host name only when absent. -10. Apply the dedicated auth selection. -11. Validate the final JSON with jq. -12. Keep the rendered result in the protected workspace while the agent archive - is downloaded and installed. +9. Apply dedicated rootfs and offline-artifact source overrides. +10. Set `agent.nodeName` from the lowercase host name only when absent. +11. Apply the dedicated auth selection. +12. Validate the final JSON with jq. 13. Atomically install the config at `/etc/aks-flex-node/config.json` with mode `0600`. 14. Clear bootstrap environment variables, including signed artifact URLs and @@ -586,7 +602,16 @@ The script: 3. Optionally validates the archive SHA-256. 4. Rejects absolute and parent-traversal tar paths. 5. Extracts `aks-flex-node-linux-` or `aks-flex-node`. -6. Atomically replaces `/usr/local/bin/aks-flex-node` with mode `0755`. +6. Asks the agent for its host root, and atomically replaces + `/bin/aks-flex-node` with mode `0755`. + +The host root is what the agent's `host-root` command prints: `/opt/unbounded`, +or `/usr/local` on a host an earlier release installed. The command runs from a +copy under `/var/lib/aks-flex-node`, because the temp dir may be on a noexec +`/tmp` and a failure to run there would be taken for an earlier release. A +release without the command predates the host root and is installed in +`/usr/local/bin`; on a host with a read-only `/usr`, such as Azure Container +Linux, such a release cannot be installed. The checksum covers the downloaded archive. Supplying a digest is strongly recommended, especially for signed URLs or mirrors. diff --git a/docs/usage/cli.md b/docs/usage/cli.md index fffc16f2..7e2c57cc 100644 --- a/docs/usage/cli.md +++ b/docs/usage/cli.md @@ -118,6 +118,8 @@ The following table is generated from the Cobra command tree. Run `make docs-cli | `aks-flex-node daemon [flags]` | Run the AKS Flex Node daemon | `agent` | listed | `--config` (required) | | `aks-flex-node fetch-bootstrap-data [flags]` | Fetch current FlexNodes join data from AKS RP | — | listed | `--agent-pool-name` (required); `--api-version` (default: `2026-05-02-preview`); `--auth` (required); `--authority-host` (default: `https://login.microsoftonline.com`); `--cluster-resource-id` (required); `--msi-client-id`; `--output`, `-o` (required); `--resource-manager-endpoint` (default: `https://management.azure.com`); `--sp-client-certificate-file`; `--sp-client-credential-file`; `--sp-client-id`; `--sp-client-secret-file`; `--sp-tenant-id` | | `aks-flex-node help [command]` | Help about any command | — | listed | — | +| `aks-flex-node host-root` | Print the directory that holds the agent's host-side files | — | hidden | — | +| `aks-flex-node ignition [flags] -- BOOTSTRAP_ARGS...` | Render an Ignition config that bootstraps the host on first boot | — | listed | `--base-config`; `--output`, `-o` (default: `-`); `--sp-client-certificate-file`; `--sp-client-secret-file` | | `aks-flex-node nspawn-lifecycle` | Run internal nspawn lifecycle hooks | — | hidden | — | | `aks-flex-node nspawn-lifecycle post-start MACHINE` | Reconcile in-machine state after machine start | — | hidden | — | | `aks-flex-node nspawn-lifecycle pre-start MACHINE` | Refresh host-side nspawn state before machine start | — | hidden | — | diff --git a/docs/usage/getting-started.md b/docs/usage/getting-started.md index ed6a295e..deab4bc1 100644 --- a/docs/usage/getting-started.md +++ b/docs/usage/getting-started.md @@ -9,7 +9,7 @@ The walkthrough uses a public AKS API endpoint and private Layer 3 connectivity Run Azure CLI, `kubectl`, artifact download, and SSH commands in your **Bash environment**. Run host preparation and bootstrap commands on the separate **flex node host** only when a step explicitly directs you to. -The bootstrap script is downloaded and run interactively on the host. This guide doesn't use cloud-init. For the architecture and security rationale, see [Generated bootstrap script](../design/storage-backed-bootstrap.md). +The bootstrap script is downloaded and run interactively on the host. Hosts provisioned by Ignition, such as Azure Container Linux, get it in an Ignition config instead; see [Hosts provisioned by Ignition](#hosts-provisioned-by-ignition). This guide doesn't use cloud-init. For the architecture and security rationale, see [Generated bootstrap script](../design/storage-backed-bootstrap.md). ## Flow @@ -447,12 +447,14 @@ For a host outside Azure, prefer an already-connected Azure Arc managed identity The script performs these operations: 1. Loads the empty JSON base. -2. Applies the cluster and pool overrides. -3. Uses the Azure VM managed identity to request an ARM token. -4. Calls `listBootstrapData` for a fresh bootstrap token, API endpoint, CA, and +2. Downloads and verifies the AKS Flex Node agent archive from GitHub Releases, + and installs the binary at `/opt/unbounded/bin/aks-flex-node`. See + [Where the agent is installed](operations.md#where-the-agent-is-installed). +3. Applies the cluster and pool overrides. +4. Uses the Azure VM managed identity to request an ARM token. +5. Calls `listBootstrapData` for a fresh bootstrap token, API endpoint, CA, and component version. -5. Applies runtime configuration overrides. -6. Downloads and verifies the AKS Flex Node agent archive from GitHub Releases. +6. Applies runtime configuration overrides. 7. Writes `/etc/aks-flex-node/config.json` as `0600 root:root`. 8. Runs non-mutating preflight. 9. Registers the ARM Machine and starts the nspawn worker. @@ -467,6 +469,30 @@ install -d -m 0755 /var/lib/aks-flex-node install -m 0600 /dev/null /var/lib/aks-flex-node/first-boot-complete ``` +### Hosts provisioned by Ignition + +Azure Container Linux and other hosts provisioned by Ignition have a read-only `/usr` and no interactive first boot. For those, render an Ignition config in your Bash environment instead of running the script on the host. `aks-flex-node ignition` embeds `bootstrap.sh` and passes everything after `--` to it: + +```bash +aks-flex-node ignition --output node.ign -- \ + --auth msi \ + --msi-client-id "" \ + --fetch-bootstrap-data \ + --cluster-resource-id "$AKS_RESOURCE_ID" \ + --agent-pool-name "$FLEX_POOL_NAME" \ + --agent-version "$AKS_FLEX_NODE_VERSION" +``` + +Use `node.ign` as the host's Ignition config; on Azure, that's the VM's custom data. On first boot, Ignition writes: + +- `/etc/aks-flex-node/first-boot/bootstrap.sh`, with the `--base-config` file, if one was given, in place of the embedded config. +- For a service principal, the credential under `/etc/aks-flex-node/credentials/` with mode `0600`. Pass `--sp-client-secret-file` or `--sp-client-certificate-file` to `aks-flex-node ignition`, before `--`. +- `aks-flex-node-bootstrap.service`, which runs the script with the arguments after `--` once the network is online. + +The agent is installed under `/opt/unbounded`, which is writable on these hosts. The unit retries failures with a delay that grows to five minutes, until `aks-flex-node-agent.service` is installed, and is skipped on later boots. Once bootstrap succeeds, it removes the script, which carries the base config. `aks-flex-node reset` disables and removes the unit. Follow progress on the host with `journalctl -u aks-flex-node-bootstrap.service`. + +The command checks the arguments against the script's options and refuses the ones it sets itself, so mistakes are reported in your Bash environment rather than at first boot. The output contains the base config and any credential, so keep it as private as they are; `--output` creates the file with mode `0600` and doesn't overwrite an existing one. + ## 6. Verify the joined node diff --git a/docs/usage/operations.md b/docs/usage/operations.md index 3fbdee19..6145db36 100644 --- a/docs/usage/operations.md +++ b/docs/usage/operations.md @@ -17,6 +17,12 @@ This guide covers host inspection, current lifecycle operations, reset, and trou AKS or the operator owns workload disruption decisions. Cordon and drain the Kubernetes Node before a disruptive operation when the surrounding control-plane workflow hasn't already done so. +## Where the agent is installed + +The agent keeps its binaries and helpers under `/opt/unbounded`: the `aks-flex-node` link in `/opt/unbounded/bin`, the blue/green binaries and recovery script in `/opt/unbounded/lib/aks-flex-node`, and the LocalDNS helper in `/opt/unbounded/libexec`. The config, state, logs, and systemd units stay under `/etc`, `/var`, and `/etc/systemd/system`. `/opt/unbounded/bin` is not on the default `PATH`, so run the host commands below as `/opt/unbounded/bin/aks-flex-node`, or add the directory to `PATH`. + +Earlier releases installed these files under `/usr/local`. On a host installed by one of them, the first command of a newer release that changes the host, such as the agent after an AgentUpgrade, links `/opt/unbounded` to `/usr/local`. The files stay where they are and the units are unchanged, so the host can still be returned to the earlier release. `aks-flex-node reset` removes the helpers under both locations and keeps the binaries; the uninstall script removes the binaries from both, and the link. An AgentUpgrade to an earlier release is refused on a host installed under `/opt/unbounded`, because that release would look for its files under `/usr/local`. + ## Preflight Run preflight before mutating the host. The command validates the config, resolves the nspawn goal state, and checks host prerequisites, API server reachability, rootfs image reachability, and bootstrap artifact sources. diff --git a/go.mod b/go.mod index e70e17c0..dacee75b 100644 --- a/go.mod +++ b/go.mod @@ -8,7 +8,7 @@ require ( github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/containerservice/armcontainerservice/v8 v8.3.0-beta.2 github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/containerservice/armcontainerservice/v9 v9.6.0-beta.1 github.com/Azure/kubelogin v0.2.19 - github.com/Azure/unbounded v0.8.0 + github.com/Azure/unbounded v0.8.1-0.20260925111623-d01744cd3f7c github.com/go-logr/logr v1.4.4 github.com/google/renameio/v2 v2.0.2 github.com/spf13/cobra v1.10.2 @@ -17,7 +17,7 @@ require ( k8s.io/apimachinery v0.37.0 k8s.io/client-go v0.37.0 k8s.io/utils v0.0.0-20260626114624-be93311217bd - sigs.k8s.io/controller-runtime v0.24.1 + sigs.k8s.io/controller-runtime v0.25.1 ) require ( @@ -69,7 +69,7 @@ require ( github.com/inconshreveable/mousetrap v1.1.0 // indirect github.com/json-iterator/go v1.1.12 // indirect github.com/keybase/go-keychain v0.0.1 // indirect - github.com/klauspost/compress v1.19.1 // indirect + github.com/klauspost/compress v1.19.2 // indirect github.com/klauspost/pgzip v1.2.6 // indirect github.com/kylelemons/godebug v1.1.0 // indirect github.com/moby/sys/user v0.4.1 // indirect @@ -99,8 +99,8 @@ require ( golang.org/x/crypto v0.56.0 // indirect golang.org/x/net v0.58.0 // indirect golang.org/x/oauth2 v0.36.0 // indirect - golang.org/x/sync v0.22.0 // indirect - golang.org/x/sys v0.47.0 // indirect + golang.org/x/sync v0.23.0 // indirect + golang.org/x/sys v0.48.0 // indirect golang.org/x/term v0.45.0 // indirect golang.org/x/text v0.41.0 // indirect golang.org/x/time v0.15.0 // indirect diff --git a/go.sum b/go.sum index 3d9e34e1..97174098 100644 --- a/go.sum +++ b/go.sum @@ -34,8 +34,8 @@ github.com/Azure/go-autorest/tracing v0.6.0 h1:TYi4+3m5t6K48TGI9AUdb+IzbnSxvnvUM github.com/Azure/go-autorest/tracing v0.6.0/go.mod h1:+vhtPC754Xsa23ID7GlGsrdKBpUA79WCAKPPZVC2DeU= github.com/Azure/kubelogin v0.2.19 h1:8rGVr461qqqkGDO+uUiZw7zKDSfRZtJ8AmhcaiwNuAQ= github.com/Azure/kubelogin v0.2.19/go.mod h1:DKHV3Cs/j7PbTP9fdkOWNxTGDuKV4fcKYkG3d6cD9Go= -github.com/Azure/unbounded v0.8.0 h1:DuEBTYRLyaEZ5kJkWfzqhd+IUPiNcWRr8pjIBDCjotk= -github.com/Azure/unbounded v0.8.0/go.mod h1:We2Vdp4VqiGU5n+mA/HLg5dzMaYhzuSVx+fIDQqBE9g= +github.com/Azure/unbounded v0.8.1-0.20260925111623-d01744cd3f7c h1:Gewg4twxPygVOXGia6fj5rypY8T+cCTSskiuE40X9Ps= +github.com/Azure/unbounded v0.8.1-0.20260925111623-d01744cd3f7c/go.mod h1:ek9riTz/LEKzZEvPPnYViN1C4aVCp2rCGWztYezVmSE= github.com/AzureAD/microsoft-authentication-extensions-for-go/cache v0.1.1 h1:WJTmL004Abzc5wDB5VtZG2PJk5ndYDgVacGqfirKxjM= github.com/AzureAD/microsoft-authentication-extensions-for-go/cache v0.1.1/go.mod h1:tCcJZ0uHAmvjsVYzEFivsRTN00oz5BEsRgQHu5JZ9WE= github.com/AzureAD/microsoft-authentication-library-for-go v1.8.0 h1:Nljr4q1GRA/5vCrMONS+g4u4LRHNgOXVSh3O43J2CnI= @@ -157,8 +157,8 @@ github.com/json-iterator/go v1.1.12 h1:PV8peI4a0ysnczrg+LtxykD8LfKY9ML6u2jnxaEnr github.com/json-iterator/go v1.1.12/go.mod h1:e30LSqwooZae/UwlEbR2852Gd8hjQvJoHmT4TnhNGBo= github.com/keybase/go-keychain v0.0.1 h1:way+bWYa6lDppZoZcgMbYsvC7GxljxrskdNInRtuthU= github.com/keybase/go-keychain v0.0.1/go.mod h1:PdEILRW3i9D8JcdM+FmY6RwkHGnhHxXwkPPMeUgOK1k= -github.com/klauspost/compress v1.19.1 h1:VsB4HPswih7mmZ8WleSFQ75c/Ui1M4trX5oAsJnhSlk= -github.com/klauspost/compress v1.19.1/go.mod h1:cwPg85FWrGar70rWktvGQj8/hthj3wpl0PGDogxkrSQ= +github.com/klauspost/compress v1.19.2 h1:hMRETovs/pu/dVWN7zIT1PGG8t509MwT6bO7XSi26R8= +github.com/klauspost/compress v1.19.2/go.mod h1:cwPg85FWrGar70rWktvGQj8/hthj3wpl0PGDogxkrSQ= github.com/klauspost/pgzip v1.2.6 h1:8RXeL5crjEUFnR2/Sn6GJNWtSQ3Dk8pq4CL3jvdDyjU= github.com/klauspost/pgzip v1.2.6/go.mod h1:Ch1tH69qFZu15pkjo5kYi6mth2Zzwzt50oCQKQE9RUs= github.com/kr/logfmt v0.0.0-20140226030751-b84e30acd515/go.mod h1:+0opPa2QZZtGFBFZlji/RkVcI2GknAs/DXo4wKdlNEc= @@ -289,8 +289,8 @@ golang.org/x/crypto v0.6.0/go.mod h1:OFC/31mSvZgRz0V1QTNCzfAI1aIRzbiufJtkMIlEp58 golang.org/x/crypto v0.56.0 h1:GUh5Ii4J5jtcseSMiRqr1jXCNHoxjeV9Fmekc2oLy6Y= golang.org/x/crypto v0.56.0/go.mod h1:OMW5y6CY9l38uPLmxU6l6pwcXp1obtLo3e6gT7gQR2I= golang.org/x/mod v0.6.0-dev.0.20220419223038-86c51ed26bb4/go.mod h1:jJ57K6gSWd91VN4djpZkiMVwK6gcyfeH4XE8wZrZaV4= -golang.org/x/mod v0.40.0 h1:hUv+3cXcdRHz08UmSiOob7sadHig73uo5bkXxQ/tvUs= -golang.org/x/mod v0.40.0/go.mod h1:0/weTWkPWGBikyTWAX3dkjVztMmBA5hM0DH6BElSupE= +golang.org/x/mod v0.41.0 h1:qJmnOUb4YB+FsEuM3HcWucdZASCPGhsX6uljO6pog0c= +golang.org/x/mod v0.41.0/go.mod h1:Ek9pY8RKWXwsWvd3rQiHYtMqkjSUV+s1Rj7j4H5Ur6o= golang.org/x/net v0.0.0-20180906233101-161cd47e91fd/go.mod h1:mL1N/T3taQHkDXs73rZJwtUhF3w3ftmwwsq0BUmARs4= golang.org/x/net v0.0.0-20190404232315-eb5bcb51f2a3/go.mod h1:t9HGtf8HONx5eT2rtn7q6eTqICYqUVnKs3thJo3Qplg= golang.org/x/net v0.0.0-20190620200207-3b0461eec859/go.mod h1:z5CRVTTTmAJ677TzLLGU+0bjPO0LkuOLi4/5GtJWs/s= @@ -305,8 +305,8 @@ golang.org/x/oauth2 v0.36.0/go.mod h1:YDBUJMTkDnJS+A4BP4eZBjCqtokkg1hODuPjwiGPO7 golang.org/x/sync v0.0.0-20180314180146-1d60e4601c6f/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM= golang.org/x/sync v0.0.0-20190423024810-112230192c58/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM= golang.org/x/sync v0.0.0-20220722155255-886fb9371eb4/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM= -golang.org/x/sync v0.22.0 h1:SZjpbeLmrCk4xhRSZFNZW5gFUeCeFgjekvI/+gfScek= -golang.org/x/sync v0.22.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0= +golang.org/x/sync v0.23.0 h1:KameEIfc1IkluZyXWLn39Wd4tURc6GbCiISGiZm2bQk= +golang.org/x/sync v0.23.0/go.mod h1:sUUOizhqBxiL6pEWpqNLUiaJn1ShEbZ6BBqskPbjZm0= golang.org/x/sys v0.0.0-20180909124046-d0be0721c37e/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY= golang.org/x/sys v0.0.0-20190215142949-d0b11bdaac8a/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY= golang.org/x/sys v0.0.0-20190222072716-a9d3bda3a223/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY= @@ -318,8 +318,8 @@ golang.org/x/sys v0.0.0-20220520151302-bc2c85ada10a/go.mod h1:oPkhp1MJrh7nUepCBc golang.org/x/sys v0.0.0-20220722155257-8c9f86f7a55f/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= golang.org/x/sys v0.1.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= golang.org/x/sys v0.5.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= -golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs= -golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw= +golang.org/x/sys v0.48.0 h1:bbX/i/6MgT9BVLM9RT1thmxL04yeTAhbEz4SyadbXoo= +golang.org/x/sys v0.48.0/go.mod h1:hNLxWAXmnKAxqDtdwIYC4bM9oQPEecfsnNMuSxOs3og= golang.org/x/term v0.0.0-20201126162022-7de9c90e9dd1/go.mod h1:bj7SfCRtBDWHUb9snDiAeCFNEtKQo2Wmx5Cou7ajbmo= golang.org/x/term v0.0.0-20210927222741-03fcf44c2211/go.mod h1:jbD1KX2456YbFQfuXm/mYQcufACuNUgVhRMnK/tPxf8= golang.org/x/term v0.5.0/go.mod h1:jMB1sMXY+tzblOD4FWmEbocvup2/aLOaQEp7JmGp78k= @@ -383,8 +383,8 @@ k8s.io/utils v0.0.0-20260626114624-be93311217bd h1:Ea7fgQ5we8Y9T0OX5o0dAHzQOBRI0 k8s.io/utils v0.0.0-20260626114624-be93311217bd/go.mod h1:xDxuJ0whA3d0I4mf/C4ppKHxXynQ+fxnkmQH0vTHnuk= oras.land/oras-go/v2 v2.6.2 h1:N04RXngAp1LJKTG6ifz3xHPipasEkWr+hFmInja5YKo= oras.land/oras-go/v2 v2.6.2/go.mod h1:PlTtg4JTDJkDe8yVHpM2wz7/YDc00GVas+i4jAW2TZ4= -sigs.k8s.io/controller-runtime v0.24.1 h1:miPEwrmirImAvgME1L9qebGHrOnGJoVmVdtOU9fRfo4= -sigs.k8s.io/controller-runtime v0.24.1/go.mod h1:vFkfY5fGt5xAC/sKb8IBFKgWPNKG9OUG29dR8Y2wImw= +sigs.k8s.io/controller-runtime v0.25.1 h1:BKgU9OeE8xv8EbbM8cY0NVzTQs35rokkdq1jh12fMb4= +sigs.k8s.io/controller-runtime v0.25.1/go.mod h1:4QqLdT6z/L6Olj8JJCtvztid4/fnIiYsfaTFScegctc= sigs.k8s.io/json v0.0.0-20250730193827-2d320260d730 h1:IpInykpT6ceI+QxKBbEflcR5EXP7sU1kvOlxwZh5txg= sigs.k8s.io/json v0.0.0-20250730193827-2d320260d730/go.mod h1:mdzfpAEoE6DHQEN0uh9ZbOCuHbLK5wOm7dK4ctXE9Tg= sigs.k8s.io/randfill v1.0.0 h1:JfjMILfT8A6RbawdsK2JXGBR5AQVfd+9TbzrlneTyrU= diff --git a/hack/acl/README.md b/hack/acl/README.md new file mode 100644 index 00000000..f2b755db --- /dev/null +++ b/hack/acl/README.md @@ -0,0 +1,82 @@ +# Azure Container Linux harness + +`acl.py` boots an Azure Container Linux VM from the Ignition config that +`aks-flex-node ignition` renders, and joins it to a local kind cluster. It +exercises the path an Ignition-provisioned host takes: the first-boot unit runs +`bootstrap.sh`, the agent is installed under `/opt/unbounded`, which is writable +while `/usr` is not, and reset removes what bootstrap installed. + +No Azure resources are used. The node joins with a bootstrap token, and the +agent's machine client talks to an AKS Flex controller deployed in the kind +cluster, reached through the API server's service proxy. + +## Requirements + +- Linux with KVM (`/dev/kvm` readable and writable) +- `docker`, `kind`, `kubectl`, `go` +- `qemu-system-x86_64`, `qemu-img`, `qemu-nbd`, and OVMF firmware (the `ovmf` + or `edk2-ovmf` package) +- `ip`, `iptables`, `nsenter`, `ssh`, `ssh-keygen` +- `sudo` without a password prompt, for the bridge, TAP device, and firewall + rules. Run `sudo -v` first if your sudo caches credentials. +- An Azure Container Linux qcow2 image + +## Usage + +```bash +hack/acl/acl.py up --image /path/to/acl.qcow2 +hack/acl/acl.py test +hack/acl/acl.py down +``` + +`up` builds the agent and controller, creates the kind cluster and a bridge on +`192.168.111.0/24`, deploys the controller, renders the Ignition config, and +boots the VM. It returns once the first-boot unit has finished and the node is +Ready. The Ignition config and agent archive are served only while `up` runs, so +later boots cannot depend on them. + +`test` checks that: + +- the first-boot unit succeeded, removed the script that carries the bootstrap + token, and installed the agent under `/opt/unbounded` and nothing under + `/usr/local` +- an argument containing `$`, `%`, quotes, and backslashes reached `bootstrap.sh` + unchanged +- a pod runs on the node and `kubectl logs` reaches its kubelet +- after a reboot, the first-boot unit is skipped and the node is Ready again +- reset removes the units, config, host helpers, LocalDNS state, and machines, + each of which is first checked to exist + +`--reset agent-reset` (the default) resets through an AgentReset +MachineOperation, `--reset cli` runs `aks-flex-node reset` on the VM, and +`--reset none` skips it. With the reboot, the first-boot unit is already +inactive when reset runs, so `test --no-reboot` is needed to check that reset +stops it. + +`acl.py ssh [COMMAND]` opens a shell on the VM or runs a command there. + +`down` stops the VM and removes the network, cluster, and `.vm/acl`. Run it +after an interrupted `up` as well, since it also removes leftover interfaces +and firewall rules. + +## Settings + +| Variable | Default | | +|---|---|---| +| `ACL_IMAGE` | | Image, instead of `--image` | +| `ACL_KIND_NODE_IMAGE` | kind's default | kind node image, for another Kubernetes version | +| `ACL_KIND_CLUSTER` | `aks-flex-acl` | kind cluster name | +| `ACL_SUBNET` | `192.168.111` | First three octets of the bridge subnet | +| `ACL_SERVE_PORT` | `8299` | Port for the Ignition config and agent archive | +| `ACL_VM_MEMORY`, `ACL_VM_CPUS` | `4096`, `2` | VM size | +| `ACL_STATE_DIR` | `.vm/acl` | Build output, disk overlay, logs, and SSH key | + +## Troubleshooting + +The VM's serial console is logged to `.vm/acl/serial.log`. On the VM, +`journalctl -u aks-flex-node-bootstrap.service` shows each bootstrap attempt. + +The bridge networking, VM launch, and UKI patching are adapted from the +unbounded repository's `hack/agent/e2e-kind` harness. `ukiboot.py` is copied +from it unchanged; it adds the Ignition URL to a UKI command line addon, so that +Ignition runs only on the first boot. diff --git a/hack/acl/acl.py b/hack/acl/acl.py new file mode 100755 index 00000000..37a719f4 --- /dev/null +++ b/hack/acl/acl.py @@ -0,0 +1,1004 @@ +#!/usr/bin/env python3 +"""Boot Azure Container Linux from `aks-flex-node ignition` and join it to a kind cluster. + + hack/acl/acl.py up --image ACL.qcow2 build, create the cluster and network, boot the VM + hack/acl/acl.py test [--reset MODE] check the node, reboot it, reset it, check cleanup + hack/acl/acl.py down remove the VM, the network, and the cluster + +The VM is provisioned only by the Ignition config that `aks-flex-node ignition` renders, plus +harness access (SSH user, hostname, static address). bootstrap.sh then installs the agent under +/opt/unbounded, the host root, and joins the node with a bootstrap token. The agent's machine client talks to +an in-cluster AKS Flex controller through the API server's service proxy, so no Azure resources +are involved. + +The bridge networking, VM launch, and UKI patching are adapted from the unbounded repository's +hack/agent/e2e-kind harness. ukiboot.py is copied from it unchanged. +""" + +from __future__ import annotations + +import argparse +import base64 +import hashlib +import http.server +import json +import os +import secrets +import shutil +import subprocess +import sys +import tarfile +import textwrap +import threading +import time +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent)) +import ukiboot # noqa: E402 + +REPO = Path(__file__).resolve().parents[2] +STATE = Path(os.environ.get("ACL_STATE_DIR", str(REPO / ".vm" / "acl"))) + +CLUSTER = os.environ.get("ACL_KIND_CLUSTER", "aks-flex-acl") +KIND_NODE = f"{CLUSTER}-control-plane" +KUBE_CONTEXT = f"kind-{CLUSTER}" + +# The subnet and interface names differ from the unbounded harness, so both can run on one host. +SUBNET = os.environ.get("ACL_SUBNET", "192.168.111") +GATEWAY = f"{SUBNET}.1" +KIND_BRIDGE_IP = f"{SUBNET}.2" +VM_IP = f"{SUBNET}.10" +BRIDGE = "virbr-acl" +TAP = "tap-acl" +VETH_HOST = "veth-kind-acl" +VETH_KIND = "eth-acl" +SERVE_PORT = int(os.environ.get("ACL_SERVE_PORT", "8299")) +VM_MEMORY = os.environ.get("ACL_VM_MEMORY", "4096") +VM_CPUS = os.environ.get("ACL_VM_CPUS", "2") +VM_MIN_DISK = 20 * 1024**3 + +NODE_NAME = "acl-harness" +HOST_ROOT = "/opt/unbounded" +SSH_USER = "core" +CONTROLLER_IMAGE = "localhost/aks-flex-controller:acl-harness" +PROXY_PATH = "/api/v1/namespaces/kube-system/services/http:aks-flex-controller:80/proxy" +BOOTSTRAP_GROUP = "system:bootstrappers:aks-flex-node" +BOOTSTRAP_UNIT = "aks-flex-node-bootstrap.service" +AGENT_UNIT = "aks-flex-node-agent.service" +FAKE_CLUSTER_ID = ( + "/subscriptions/00000000-0000-0000-0000-000000000000/resourceGroups/acl-harness" + "/providers/Microsoft.ContainerService/managedClusters/acl-harness" +) +# Passed to bootstrap.sh through the first-boot unit's command line. It arrives in the installed +# config unchanged only if the unit quotes $, %, quotes, and backslashes correctly. +QUOTING_PROBE = """$HOME ${X} %h %% "q" \\ it's""" + +AGENT_ARCHIVE = "aks-flex-node-linux-amd64.tar.gz" +IGNITION_NAME = "config.ign" +SSH_KEY = STATE / "ssh" / "id_ed25519" +SSH_OPTS = [ + "-o", "StrictHostKeyChecking=no", + "-o", "UserKnownHostsFile=/dev/null", + "-o", "LogLevel=ERROR", + "-o", "ConnectTimeout=10", + "-i", str(SSH_KEY), +] + +OVMF_CANDIDATES = [ + ("/usr/share/OVMF/OVMF_CODE.fd", "/usr/share/OVMF/OVMF_VARS.fd"), + ("/usr/share/OVMF/OVMF_CODE_4M.fd", "/usr/share/OVMF/OVMF_VARS_4M.fd"), + ("/usr/share/edk2/ovmf/OVMF_CODE.fd", "/usr/share/edk2/ovmf/OVMF_VARS.fd"), + ("/usr/share/qemu/ovmf-x86_64-code.bin", "/usr/share/qemu/ovmf-x86_64-vars.bin"), +] + + +# --------------------------------------------------------------------------------------------- +# Process helpers + + +def log(message: str) -> None: + print(f"[acl] {message}", flush=True) + + +def die(message: str) -> None: + print(f"[acl] error: {message}", file=sys.stderr, flush=True) + sys.exit(1) + + +def run(cmd: list[str], *, check: bool = True, capture: bool = False, input: str | None = None, + env: dict[str, str] | None = None, cwd: Path | None = None) -> subprocess.CompletedProcess[str]: + result = subprocess.run( + cmd, check=False, text=True, input=input, cwd=cwd, + env={**os.environ, **env} if env else None, + stdout=subprocess.PIPE if capture else None, + stderr=subprocess.PIPE if capture else None, + ) + if check and result.returncode != 0: + detail = f": {result.stderr.strip()}" if capture and result.stderr else "" + die(f"{' '.join(cmd)} exited {result.returncode}{detail}") + return result + + +def capture(cmd: list[str], **kwargs) -> str: + return run(cmd, capture=True, **kwargs).stdout.strip() + + +def sudo(*cmd: str, check: bool = True) -> subprocess.CompletedProcess[str]: + return run(["sudo", *cmd], check=check, capture=not check) + + +def kubectl(*args: str, check: bool = True, input: str | None = None, + quiet: bool = False) -> subprocess.CompletedProcess[str]: + return run(["kubectl", "--context", KUBE_CONTEXT, *args], check=check, input=input, + capture=quiet) + + +def kubectl_out(*args: str) -> str: + return capture(["kubectl", "--context", KUBE_CONTEXT, *args]) + + +def wait_until(description: str, timeout: float, probe, interval: float = 5) -> None: + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + if probe(): + return + time.sleep(interval) + die(f"timed out after {int(timeout)}s waiting for {description}") + + +# --------------------------------------------------------------------------------------------- +# Build + + +def build_binaries() -> tuple[Path, Path]: + """Build a static linux/amd64 agent and controller. The agent runs on the ACL host, whose glibc + may be older than the build host's, so cgo is off.""" + STATE.mkdir(parents=True, exist_ok=True) + env = {"GOOS": "linux", "GOARCH": "amd64", "CGO_ENABLED": "0"} + agent = STATE / "aks-flex-node-linux-amd64" + controller = STATE / "controller" / "aks-flex-controller" + controller.parent.mkdir(exist_ok=True) + log("building aks-flex-node and aks-flex-controller") + run(["go", "build", "-o", str(agent), "./cmd/aks-flex-node"], env=env, cwd=REPO) + run(["go", "build", "-o", str(controller), "./cmd/aks-flex-controller"], env=env, cwd=REPO) + + with tarfile.open(STATE / AGENT_ARCHIVE, "w:gz") as archive: + archive.add(agent, arcname=agent.name) + return agent, controller + + +def sha256(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as f: + for chunk in iter(lambda: f.read(1 << 20), b""): + digest.update(chunk) + return digest.hexdigest() + + +# --------------------------------------------------------------------------------------------- +# Host networking + + +def nm_unmanage(iface: str) -> None: + """Keep NetworkManager from detaching the interface from the bridge.""" + if shutil.which("nmcli"): + sudo("nmcli", "device", "set", iface, "managed", "no", check=False) + + +def docker_uses_nft() -> bool: + return bool(shutil.which("nft")) and sudo("nft", "list", "chain", "ip", "filter", "DOCKER-USER", + check=False).returncode == 0 + + +def create_network() -> None: + log(f"creating bridge {BRIDGE} ({GATEWAY}/24) and {TAP}") + delete_network() + sudo("ip", "link", "add", BRIDGE, "type", "bridge") + sudo("ip", "addr", "add", f"{GATEWAY}/24", "dev", BRIDGE) + sudo("ip", "link", "set", BRIDGE, "up") + sudo("ip", "tuntap", "add", "dev", TAP, "mode", "tap") + sudo("ip", "link", "set", TAP, "master", BRIDGE) + sudo("ip", "link", "set", TAP, "up") + nm_unmanage(BRIDGE) + nm_unmanage(TAP) + + # Docker's forwarding policy drops traffic it did not create. + if docker_uses_nft(): + sudo("nft", "insert", "rule", "ip", "filter", "DOCKER-USER", "iifname", BRIDGE, "accept") + sudo("nft", "insert", "rule", "ip", "filter", "DOCKER-USER", "oifname", BRIDGE, "accept") + else: + sudo("iptables", "-I", "FORWARD", "-i", BRIDGE, "-j", "ACCEPT") + sudo("iptables", "-I", "FORWARD", "-o", BRIDGE, "-j", "ACCEPT") + # Docker may drop non-Docker traffic to container addresses in raw PREROUTING. + sudo("iptables", "-t", "raw", "-I", "PREROUTING", "-i", BRIDGE, "-j", "ACCEPT", check=False) + # The VM reaches the internet for the rootfs image and Kubernetes binaries. + sudo("iptables", "-t", "nat", "-A", "POSTROUTING", "-s", f"{SUBNET}.0/24", + "!", "-d", f"{SUBNET}.0/24", "-j", "MASQUERADE") + + +def delete_network() -> None: + for iface in (TAP, VETH_HOST, BRIDGE): + sudo("ip", "link", "delete", iface, check=False) + for chain_opt in ("-i", "-o"): + while sudo("iptables", "-D", "FORWARD", chain_opt, BRIDGE, "-j", "ACCEPT", + check=False).returncode == 0: + pass + while sudo("iptables", "-t", "raw", "-D", "PREROUTING", "-i", BRIDGE, "-j", "ACCEPT", + check=False).returncode == 0: + pass + while sudo("iptables", "-t", "nat", "-D", "POSTROUTING", "-s", f"{SUBNET}.0/24", + "!", "-d", f"{SUBNET}.0/24", "-j", "MASQUERADE", check=False).returncode == 0: + pass + if docker_uses_nft(): + listing = sudo("nft", "-a", "list", "chain", "ip", "filter", "DOCKER-USER", check=False).stdout + for line in listing.splitlines(): + if f'"{BRIDGE}"' in line and "# handle" in line: + handle = line.rsplit("# handle", 1)[1].strip() + sudo("nft", "delete", "rule", "ip", "filter", "DOCKER-USER", "handle", handle, + check=False) + + +def kind_pid() -> str: + return capture(["docker", "inspect", KIND_NODE, "--format", "{{.State.Pid}}"]) + + +def kind_docker_ip() -> str: + ip = capture(["docker", "inspect", KIND_NODE, "--format", + "{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}"]) + if not ip: + die(f"no address for {KIND_NODE}") + return ip + + +def attach_kind_to_bridge() -> None: + """Give the kind node an address on the bridge, so it can reach the VM's kubelet.""" + log(f"attaching {KIND_NODE} to {BRIDGE} as {KIND_BRIDGE_IP}") + pid = kind_pid() + sudo("ip", "link", "delete", VETH_HOST, check=False) + sudo("ip", "link", "add", VETH_HOST, "type", "veth", "peer", "name", VETH_KIND) + sudo("ip", "link", "set", VETH_HOST, "master", BRIDGE) + sudo("ip", "link", "set", VETH_HOST, "up") + sudo("ip", "link", "set", VETH_KIND, "netns", pid) + sudo("nsenter", "-t", pid, "-n", "ip", "addr", "add", f"{KIND_BRIDGE_IP}/24", "dev", VETH_KIND) + sudo("nsenter", "-t", pid, "-n", "ip", "link", "set", VETH_KIND, "up") + nm_unmanage(VETH_HOST) + + +# --------------------------------------------------------------------------------------------- +# Cluster + + +def cluster_exists() -> bool: + return CLUSTER in capture(["kind", "get", "clusters"]).split() + + +def create_cluster(node_image: str) -> None: + if cluster_exists(): + log(f"kind cluster {CLUSTER} exists") + else: + cmd = ["kind", "create", "cluster", "--name", CLUSTER, "--wait", "120s"] + if node_image: + cmd += ["--image", node_image] + run(cmd) + + attach_kind_to_bridge() + docker_ip = kind_docker_ip() + advertise_bridge_address() + + # kindnet and kube-proxy on the VM's node would otherwise use the control plane's container + # hostname, which the VM cannot resolve. + endpoint = f"{docker_ip}:6443" + patch = {"spec": {"template": {"spec": {"containers": [ + {"name": "kindnet-cni", "env": [{"name": "CONTROL_PLANE_ENDPOINT", "value": endpoint}]}]}}}} + kubectl("-n", "kube-system", "patch", "daemonset", "kindnet", "--type=strategic", + "-p", json.dumps(patch)) + kubeconfig = kubectl_out("-n", "kube-system", "get", "configmap", "kube-proxy", + "-o", "jsonpath={.data.kubeconfig\\.conf}") + lines = [f" server: https://{endpoint}" if line.strip().startswith("server:") else line + for line in kubeconfig.splitlines()] + kubectl("-n", "kube-system", "patch", "configmap", "kube-proxy", "--type=merge", + "-p", json.dumps({"data": {"kubeconfig.conf": "\n".join(lines) + "\n"}})) + kubectl("-n", "kube-system", "rollout", "restart", "daemonset/kube-proxy") + # Restarting kindnet after the address change avoids a race that leaves it crash looping. + kubectl("-n", "kube-system", "delete", "pod", "-l", "app=kindnet", "--wait=false") + kubectl("-n", "kube-system", "rollout", "status", "daemonset/kindnet", "--timeout=120s") + kubectl("-n", "kube-system", "rollout", "status", "daemonset/kube-proxy", "--timeout=120s") + + +def advertise_bridge_address() -> None: + """Make the control plane advertise its bridge address. + + kindnet on the VM's node routes the control plane's pod CIDR through the control plane's node + address. The container's Docker address is not on the VM's link, so the route fails and + kindnet never becomes ready. + """ + script = textwrap.dedent(f"""\ + set -eu + . /var/lib/kubelet/kubeadm-flags.env + set -- $KUBELET_KUBEADM_ARGS + new_args="" + for arg do + case "$arg" in + --node-ip=*) ;; + *) new_args="$new_args $arg" ;; + esac + done + printf 'KUBELET_KUBEADM_ARGS="%s --node-ip={KIND_BRIDGE_IP}"\\n' "${{new_args# }}" \\ + >/var/lib/kubelet/kubeadm-flags.env + systemctl restart kubelet + """) + run(["docker", "exec", KIND_NODE, "sh", "-c", script]) + + def advertised() -> bool: + result = kubectl("get", "node", KIND_NODE, "-o", + "jsonpath={.status.addresses[?(@.type=='InternalIP')].address}", + check=False, quiet=True) + return result.returncode == 0 and result.stdout.split() == [KIND_BRIDGE_IP] + + wait_until(f"{KIND_NODE} to advertise {KIND_BRIDGE_IP}", 120, advertised, 3) + + +def install_controller(controller: Path) -> None: + """Deploy the controller from the repository manifests, with a locally built image.""" + log("deploying aks-flex-controller") + context = controller.parent + (context / "Dockerfile").write_text(textwrap.dedent("""\ + FROM scratch + COPY aks-flex-controller /usr/local/bin/aks-flex-controller + USER 65532:65532 + ENTRYPOINT ["/usr/local/bin/aks-flex-controller"] + """)) + run(["docker", "build", "-q", "-t", CONTROLLER_IMAGE, str(context)]) + run(["kind", "load", "docker-image", CONTROLLER_IMAGE, "--name", CLUSTER]) + + # The daemon discovers MachineOperations at startup, so the CRD comes first. + unbounded = capture(["go", "list", "-m", "-f", "{{.Dir}}", "github.com/Azure/unbounded"], cwd=REPO) + kubectl("apply", "-f", f"{unbounded}/deploy/machina/crd/unbounded-cloud.io_machineoperations.yaml") + kubectl("apply", "-f", "-", input=RBAC_MANIFEST) + kubectl("apply", "-k", str(REPO / "hack" / "controller-deployment")) + patch = [ + {"op": "replace", "path": "/spec/template/spec/containers/0/image", "value": CONTROLLER_IMAGE}, + {"op": "replace", "path": "/spec/template/spec/containers/0/imagePullPolicy", "value": "Never"}, + {"op": "replace", "path": "/spec/template/spec/containers/0/args", "value": [ + "--listen-address=:8080", + "--machine-configmap-namespace=kube-system", + "--machine-configmap-name=aks-flex-machines", + # Without the approver the daemon's client certificate is never issued. + "--enable-csr-approver=true", + ]}, + ] + kubectl("-n", "kube-system", "patch", "deployment", "aks-flex-controller", "--type=json", + "-p", json.dumps(patch)) + kubectl("-n", "kube-system", "rollout", "status", "deployment/aks-flex-controller", + "--timeout=180s") + kubectl("get", "--raw", f"{PROXY_PATH}/healthz") + + +# The bindings scripts/aks-flex-config setup-node-rbac applies for bootstrap-token nodes. +RBAC_MANIFEST = f""" +apiVersion: rbac.authorization.k8s.io/v1 +kind: ClusterRoleBinding +metadata: + name: aks-flex-node-bootstrapper +roleRef: + apiGroup: rbac.authorization.k8s.io + kind: ClusterRole + name: system:node-bootstrapper +subjects: +- apiGroup: rbac.authorization.k8s.io + kind: Group + name: {BOOTSTRAP_GROUP} +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: ClusterRoleBinding +metadata: + name: aks-flex-node-auto-approve-csr +roleRef: + apiGroup: rbac.authorization.k8s.io + kind: ClusterRole + name: system:certificates.k8s.io:certificatesigningrequests:nodeclient +subjects: +- apiGroup: rbac.authorization.k8s.io + kind: Group + name: {BOOTSTRAP_GROUP} +""" + + +def kubernetes_version() -> str: + version = json.loads(kubectl_out("version", "-o", "json"))["serverVersion"]["gitVersion"] + return version.removeprefix("v").split("+", 1)[0] + + +def create_bootstrap_token() -> str: + token_id, token_secret = secrets.token_hex(3), secrets.token_hex(8) + expiration = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime(time.time() + 24 * 3600)) + secret = { + "apiVersion": "v1", + "kind": "Secret", + "type": "bootstrap.kubernetes.io/token", + "metadata": {"name": f"bootstrap-token-{token_id}", "namespace": "kube-system"}, + "stringData": { + "token-id": token_id, + "token-secret": token_secret, + "expiration": expiration, + "usage-bootstrap-authentication": "true", + "usage-bootstrap-signing": "true", + "auth-extra-groups": BOOTSTRAP_GROUP, + }, + } + kubectl("apply", "-f", "-", input=json.dumps(secret), quiet=True) + return f"{token_id}.{token_secret}" + + +def register_machine(version: str) -> None: + """Add the node to the controller's machine store. The version has to match + components.kubernetes exactly, since the store is read only.""" + machine = { + "id": f"{FAKE_CLUSTER_ID}/agentPools/aksflexnodes/machines/{NODE_NAME}", + "name": NODE_NAME, + "type": "Microsoft.ContainerService/managedClusters/agentPools/machines", + "properties": { + "eTag": "1", + "provisioningState": "Succeeded", + "kubernetes": {"orchestratorVersion": version, "nodeLabels": {}, "nodeTaints": []}, + }, + } + patch = {"data": {f"{NODE_NAME}.json": json.dumps(machine)}} + kubectl("-n", "kube-system", "patch", "configmap", "aks-flex-machines", "--type=merge", + "-p", json.dumps(patch)) + + +# --------------------------------------------------------------------------------------------- +# Ignition + + +def base_config(token: str, version: str) -> dict: + ca = kubectl_out("config", "view", "--minify", "--raw", "-o", + "jsonpath={.clusters[0].cluster.certificate-authority-data}") + dns_ip = kubectl_out("-n", "kube-system", "get", "service", "kube-dns", "-o", + "jsonpath={.spec.clusterIP}") + return { + "azure": { + "bootstrapToken": {"token": token}, + "targetCluster": {"resourceId": FAKE_CLUSTER_ID, "location": "local"}, + }, + "agent": { + "nodeName": NODE_NAME, + "machineClient": {"mode": "in-cluster", "endpointUrl": PROXY_PATH}, + "requireMachineRegistration": True, + }, + "components": {"kubernetes": version}, + # Required so that reset has LocalDNS state to clean up. + "networking": {"dnsServiceIP": dns_ip, "localDNS": {"mode": "Required"}}, + "node": {"kubelet": {"clusterFQDN": f"{kind_docker_ip()}:6443", "caCertData": ca}}, + } + + +def render_ignition(agent: Path, config: dict, agent_url: str, agent_sha256: str) -> dict: + base_path = STATE / "base-config.json" + base_path.write_text(json.dumps(config, indent=2) + "\n") + base_path.chmod(0o600) + rendered = STATE / "node.ign" + rendered.unlink(missing_ok=True) + run([str(agent), "ignition", "--base-config", str(base_path), "--output", str(rendered), "--", + "--agent-url", agent_url, + "--agent-sha256", agent_sha256, + "--config-overrides", json.dumps({"aclHarness": {"quoting": QUOTING_PROBE}})]) + return json.loads(rendered.read_text()) + + +def data_url(content: str) -> str: + return "data:;base64," + base64.b64encode(content.encode()).decode() + + +def mac_address() -> str: + octets = [int(part) for part in VM_IP.split(".")] + return f"52:54:00:{octets[1]:02x}:{octets[2]:02x}:{octets[3]:02x}" + + +def add_harness_access(doc: dict, ssh_public_key: str) -> dict: + """Add what the harness needs to reach the VM. Nothing here affects the bootstrap. + + A user Ignition config displaces the image's own systemd section, so the waagent mask that + config carries is restated. The core user is given in full, because a bare name leaves it on + /sbin/nologin. The static network unit matches on MAC, since interface names depend on the + machine type. + """ + doc.setdefault("passwd", {}).setdefault("users", []).append({ + "name": SSH_USER, + "shell": "/bin/bash", + "groups": ["sudo", "systemd-journal"], + "sshAuthorizedKeys": [ssh_public_key], + }) + doc.setdefault("systemd", {}).setdefault("units", []).append( + {"name": "waagent.service", "enabled": False, "mask": True}) + network_unit = textwrap.dedent(f"""\ + [Match] + MACAddress={mac_address()} + + [Network] + Address={VM_IP}/24 + Gateway={GATEWAY} + DNS=8.8.8.8 + DNS=8.8.4.4 + """) + files = doc.setdefault("storage", {}).setdefault("files", []) + files.append({"path": "/etc/hostname", "mode": 0o644, "overwrite": True, + "contents": {"source": data_url(f"{NODE_NAME}\n")}}) + files.append({"path": "/etc/systemd/network/10-acl-harness.network", "mode": 0o644, + "overwrite": True, "contents": {"source": data_url(network_unit)}}) + return doc + + +def initramfs_ip_karg() -> str: + """Static address for Ignition's fetch in the initramfs. The interface is named because an + empty device field matches lo first, and every fetch is refused.""" + return f"ip={VM_IP}::{GATEWAY}:24:{NODE_NAME}:eth0:none:8.8.8.8:8.8.4.4" + + +class FileServer: + """Serve an explicit set of files on the bridge. Only these are exposed, since the state + directory also holds the SSH key and the base config.""" + + def __init__(self, files: dict[str, Path]): + self.files = files + served = files + + class Handler(http.server.BaseHTTPRequestHandler): + def do_GET(self) -> None: # noqa: N802 - required name + path = served.get(self.path.lstrip("/").split("?", 1)[0]) + if path is None: + self.send_error(404) + return + body = path.read_bytes() + self.send_response(200) + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + + def log_message(self, *args) -> None: + pass + + self.httpd = http.server.ThreadingHTTPServer((GATEWAY, SERVE_PORT), Handler) + self.thread = threading.Thread(target=self.httpd.serve_forever, daemon=True) + + def __enter__(self) -> "FileServer": + self.thread.start() + return self + + def __exit__(self, *exc) -> None: + self.httpd.shutdown() + self.httpd.server_close() + + +# --------------------------------------------------------------------------------------------- +# VM + + +def ovmf_firmware() -> tuple[Path, Path]: + for code, variables in OVMF_CANDIDATES: + if Path(code).is_file() and Path(variables).is_file(): + return Path(code), Path(variables) + die("OVMF firmware not found; install the ovmf or edk2-ovmf package") + raise AssertionError + + +def ensure_ssh_key() -> str: + SSH_KEY.parent.mkdir(parents=True, exist_ok=True) + if not SSH_KEY.exists(): + run(["ssh-keygen", "-t", "ed25519", "-N", "", "-q", "-f", str(SSH_KEY)]) + return SSH_KEY.with_suffix(".pub").read_text().strip() + + +def launch_vm(image: Path, config_url: str) -> None: + disk = STATE / "acl.qcow2" + disk.unlink(missing_ok=True) + info = json.loads(capture(["qemu-img", "info", "--output=json", str(image)])) + # The root partition runs to the end of the disk, so a smaller overlay stops boot. + size = max(int(info["virtual-size"]), VM_MIN_DISK) + run(["qemu-img", "create", "-q", "-f", "qcow2", "-b", str(image.resolve()), "-F", + info["format"], str(disk), str(size)]) + + # systemd-boot only marks a first boot while the image's firstboot addon exists, and + # ignition-quench deletes it afterwards. Adding the Ignition URL to another addon, rather + # than booting the kernel directly, keeps later boots from running Ignition again. + patched = ukiboot.patch_uki_cmdline_addon(disk, f"ignition.config.url={config_url} {initramfs_ip_karg()}") + log(f"patched {patched.addon} ({patched.used}/{patched.capacity} bytes)") + + code, variables_template = ovmf_firmware() + variables = STATE / "OVMF_VARS.fd" + shutil.copyfile(variables_template, variables) + + pid_file, serial_log = STATE / "qemu.pid", STATE / "serial.log" + serial_log.unlink(missing_ok=True) + run(["setsid", "qemu-system-x86_64", + "-machine", "q35", "-cpu", "host", "-accel", "kvm", + "-m", VM_MEMORY, "-smp", VM_CPUS, + "-drive", f"if=pflash,format=raw,readonly=on,file={code}", + "-drive", f"if=pflash,format=raw,file={variables}", + "-drive", f"file={disk},format=qcow2,if=virtio", + "-netdev", f"tap,id=net0,ifname={TAP},script=no,downscript=no", + "-device", f"virtio-net-pci,netdev=net0,mac={mac_address()}", + "-daemonize", "-pidfile", str(pid_file), + "-serial", f"file:{serial_log}", "-display", "none"]) + log(f"VM started (pid {pid_file.read_text().strip()}, serial log {serial_log})") + + +def vm_running() -> bool: + pid_file = STATE / "qemu.pid" + if not pid_file.exists(): + return False + try: + os.kill(int(pid_file.read_text().strip()), 0) + except (OSError, ValueError): + return False + return True + + +def stop_vm() -> None: + pid_file = STATE / "qemu.pid" + if not vm_running(): + pid_file.unlink(missing_ok=True) + return + pid = int(pid_file.read_text().strip()) + log(f"stopping VM (pid {pid})") + os.kill(pid, 15) + for _ in range(10): + time.sleep(1) + if not vm_running(): + break + else: + os.kill(pid, 9) + pid_file.unlink(missing_ok=True) + + +def ssh(command: str, *, check: bool = True, timeout: float = 120) -> subprocess.CompletedProcess[str]: + try: + result = subprocess.run(["ssh", *SSH_OPTS, f"{SSH_USER}@{VM_IP}", command], + capture_output=True, text=True, timeout=timeout) + except subprocess.TimeoutExpired: + result = subprocess.CompletedProcess(["ssh"], 255, "", f"timed out: {command}") + if check and result.returncode != 0: + die(f"on the VM, {command!r} exited {result.returncode}: {result.stderr.strip()}") + return result + + +def ssh_out(command: str) -> str: + return ssh(command).stdout.strip() + + +def wait_for_ssh(timeout: float = 600) -> None: + def reachable() -> bool: + if not vm_running(): + die(f"the VM exited; see {STATE / 'serial.log'}") + return ssh("true", check=False, timeout=15).returncode == 0 + + wait_until(f"SSH on {VM_IP}", timeout, reachable, 5) + + +def unit_property(unit: str, prop: str) -> str: + return ssh_out(f"systemctl show {unit} -p {prop} --value") + + +def wait_for_bootstrap(timeout: float = 1800) -> None: + """Wait for the first-boot unit to finish. It retries on failure, so a failed attempt is + reported but not final.""" + log(f"waiting for {BOOTSTRAP_UNIT}") + seen_restarts = 0 + + def finished() -> bool: + nonlocal seen_restarts + # One property per call: systemctl prints several in its own order, not the one asked for. + active = ssh(f"systemctl show {BOOTSTRAP_UNIT} -p ActiveState --value", check=False).stdout.strip() + restarts_out = ssh(f"systemctl show {BOOTSTRAP_UNIT} -p NRestarts --value", check=False).stdout.strip() + if not active or not restarts_out.isdigit(): + return False + restarts = int(restarts_out) + if restarts > seen_restarts: + seen_restarts = restarts + log(f"{BOOTSTRAP_UNIT} failed and is retrying (restart {restarts}):") + print(ssh(f"sudo journalctl -u {BOOTSTRAP_UNIT} -b --no-pager -n 30", check=False).stdout) + return active == "active" + + wait_until(f"{BOOTSTRAP_UNIT} to complete", timeout, finished, 15) + log(f"{BOOTSTRAP_UNIT} completed") + + +def wait_for_node_ready(timeout: float = 900) -> None: + log(f"waiting for node {NODE_NAME} to be Ready") + + def ready() -> bool: + result = kubectl("get", "node", NODE_NAME, "-o", + "jsonpath={.status.conditions[?(@.type=='Ready')].status}", + check=False, quiet=True) + return result.returncode == 0 and result.stdout.strip() == "True" + + wait_until(f"node {NODE_NAME} to be Ready", timeout, ready, 10) + + +# --------------------------------------------------------------------------------------------- +# Checks + + +class Checks: + def __init__(self) -> None: + self.failures: list[str] = [] + + def expect(self, ok: bool, description: str, detail: str = "") -> None: + if ok: + log(f"ok: {description}") + else: + self.failures.append(description) + log(f"FAIL: {description}{': ' + detail if detail else ''}") + + def paths_absent(self, description: str, paths: list[str]) -> None: + present = [p for p in paths if path_exists(p)] + self.expect(not present, description, ", ".join(present)) + + def paths_present(self, description: str, paths: list[str]) -> None: + missing = [p for p in paths if not path_exists(p)] + self.expect(not missing, description, "missing " + ", ".join(missing)) + + def note(self, message: str) -> None: + log(f"not exercised: {message}") + + def done(self) -> None: + if self.failures: + die(f"{len(self.failures)} check(s) failed: " + "; ".join(self.failures)) + log("all checks passed") + + +def path_exists(path: str) -> bool: + return ssh(f"sudo test -e {path} -o -L {path}", check=False).returncode == 0 + + +# What reset has to remove, by what it is. Each is checked to exist before reset, so that its +# absence afterwards means something. +RESET_REMOVES = { + "the agent, recovery, and first-boot units": [ + f"/etc/systemd/system/{AGENT_UNIT}", + "/etc/systemd/system/aks-flex-node-agent-recovery.service", + f"/etc/systemd/system/{BOOTSTRAP_UNIT}", + ], + "the config, credentials, and logs": ["/etc/aks-flex-node", "/var/log/aks-flex-node"], + "the host helpers under the host root": [ + f"{HOST_ROOT}/bin/unbounded-agent-nspawn-lifecycle", + f"{HOST_ROOT}/lib/aks-flex-node/aks-flex-node-recovery.sh", + ], + "the LocalDNS files": [ + "/etc/systemd/system/unbounded-localdns-network.service", + f"{HOST_ROOT}/libexec/unbounded-localdns-network", + "/sys/class/net/localdns", + ], +} + + +def check_first_boot(checks: Checks) -> None: + result = unit_property(BOOTSTRAP_UNIT, "Result") + active = unit_property(BOOTSTRAP_UNIT, "ActiveState") + checks.expect(active == "active" and result == "success", + f"{BOOTSTRAP_UNIT} succeeded on first boot", f"ActiveState={active} Result={result}") + checks.expect(unit_property(BOOTSTRAP_UNIT, "InvocationID") != "", + f"{BOOTSTRAP_UNIT} ran in this boot") + checks.paths_absent("bootstrap.sh, which carries the token, was removed after bootstrap", + ["/etc/aks-flex-node/first-boot/bootstrap.sh"]) + + installed = ssh(f"sudo test -x {HOST_ROOT}/bin/aks-flex-node", check=False).returncode == 0 + checks.expect(installed, f"the agent is installed under {HOST_ROOT}") + root_is_dir = ssh(f"sudo test -d {HOST_ROOT} -a ! -L {HOST_ROOT}", check=False).returncode == 0 + checks.expect(root_is_dir, f"{HOST_ROOT} is a directory, not a link to the read-only /usr/local") + checks.paths_absent("nothing was installed under the read-only /usr/local", + ["/usr/local/bin/aks-flex-node", "/usr/local/lib/aks-flex-node"]) + probe = ssh_out("sudo jq -r .aclHarness.quoting /etc/aks-flex-node/config.json") + checks.expect(probe == QUOTING_PROBE, "an argument with $, %, quotes, and backslashes reached " + "bootstrap.sh unchanged", f"got {probe!r}") + checks.expect(unit_property(AGENT_UNIT, "ActiveState") == "active", f"{AGENT_UNIT} is active") + + +def check_workload(checks: Checks) -> None: + pod = { + "apiVersion": "v1", + "kind": "Pod", + "metadata": {"name": "acl-harness-probe", "namespace": "default"}, + "spec": { + "nodeName": NODE_NAME, + "restartPolicy": "Never", + "tolerations": [{"operator": "Exists"}], + "containers": [{"name": "probe", "image": "docker.io/library/busybox:1.36", + "command": ["sh", "-c", "echo acl-harness-ok && sleep 3600"]}], + }, + } + kubectl("delete", "pod", "acl-harness-probe", "--ignore-not-found", "--wait=true", quiet=True) + kubectl("apply", "-f", "-", input=json.dumps(pod), quiet=True) + ready = kubectl("wait", "pod/acl-harness-probe", "--for=condition=Ready", "--timeout=300s", + check=False, quiet=True).returncode == 0 + checks.expect(ready, "a pod runs on the node") + logs = kubectl("logs", "acl-harness-probe", check=False, quiet=True) + checks.expect("acl-harness-ok" in logs.stdout, "kubectl logs reaches the kubelet", + logs.stderr.strip()) + kubectl("delete", "pod", "acl-harness-probe", "--ignore-not-found", "--wait=false", quiet=True) + + +def check_reboot(checks: Checks) -> None: + """The first-boot unit must not run again once the agent is installed, and Ignition must not + run again at all: its config server is down now, so a second run would stop the boot.""" + log("rebooting the VM") + ssh("sudo systemctl reboot", check=False, timeout=15) + wait_until("the VM to go down", 120, lambda: ssh("true", check=False, timeout=5).returncode != 0, 3) + wait_for_ssh() + checks.expect(unit_property(BOOTSTRAP_UNIT, "ConditionResult") == "no", + f"{BOOTSTRAP_UNIT} is skipped after a reboot") + checks.expect(unit_property(BOOTSTRAP_UNIT, "InvocationID") == "", + f"{BOOTSTRAP_UNIT} did not run after a reboot") + wait_for_node_ready() + checks.expect(unit_property(AGENT_UNIT, "ActiveState") == "active", + f"{AGENT_UNIT} is active after a reboot") + + +def reset_node(mode: str) -> None: + if mode == "cli": + log("resetting with aks-flex-node reset") + ssh(f"sudo {HOST_ROOT}/bin/aks-flex-node reset", timeout=600) + else: + log("resetting with an AgentReset MachineOperation") + operation = { + "apiVersion": "unbounded-cloud.io/v1alpha3", + "kind": "MachineOperation", + "metadata": {"name": "acl-harness-reset"}, + "spec": {"machineRef": NODE_NAME, "operationKind": "AgentReset", + "ttlSecondsAfterFinished": 3600}, + } + kubectl("delete", "machineoperation", "acl-harness-reset", "--ignore-not-found", quiet=True) + kubectl("apply", "-f", "-", input=json.dumps(operation)) + kubectl("wait", "machineoperation/acl-harness-reset", "--for=jsonpath={.status.phase}=Complete", + "--timeout=600s") + + def uninstalled() -> bool: + return ssh(f"test -e /etc/systemd/system/{AGENT_UNIT}", check=False).returncode != 0 + + wait_until(f"{AGENT_UNIT} to be removed", 300, uninstalled, 5) + kubectl("delete", "node", NODE_NAME, "--ignore-not-found", "--wait=false", quiet=True) + + +def localdns_table_exists() -> bool: + return ssh("sudo nft list table ip unbounded_localdns", check=False).returncode == 0 + + +def machine_exists(machine: str) -> bool: + return ssh(f"sudo machinectl show {machine}", check=False).returncode == 0 + + +def check_before_reset(checks: Checks) -> bool: + """Check that what reset removes exists. Returns whether the first-boot unit is active.""" + for what, paths in RESET_REMOVES.items(): + checks.paths_present(f"{what} exist before reset", paths) + checks.expect(localdns_table_exists(), "the LocalDNS nft table exists before reset") + checks.expect(machine_exists("kube1"), "the kube1 machine exists before reset") + return unit_property(BOOTSTRAP_UNIT, "ActiveState") == "active" + + +def check_reset(checks: Checks, bootstrap_was_active: bool) -> None: + for what, paths in RESET_REMOVES.items(): + checks.paths_absent(f"reset removed {what}", paths) + checks.expect(not localdns_table_exists(), "reset removed the LocalDNS nft table") + for machine in ("kube1", "kube2"): + checks.expect(not machine_exists(machine), f"reset removed the {machine} machine") + + # A first-boot unit that stays active after its file is gone cannot be started again. + if bootstrap_was_active: + checks.expect(unit_property(BOOTSTRAP_UNIT, "ActiveState") != "active", + f"reset stopped {BOOTSTRAP_UNIT}, which was active") + else: + checks.note(f"{BOOTSTRAP_UNIT} was not active before reset, since the reboot skipped it; " + "run `test --no-reboot` to cover stopping it") + + +# --------------------------------------------------------------------------------------------- +# Commands + + +def check_prerequisites() -> None: + missing = [tool for tool in ("docker", "kind", "kubectl", "go", "qemu-system-x86_64", "qemu-img", + "qemu-nbd", "ssh", "ssh-keygen", "ip", "iptables", "nsenter", "setsid") + if shutil.which(tool) is None] + if missing: + die("missing tools: " + ", ".join(missing)) + if not os.access("/dev/kvm", os.R_OK | os.W_OK): + die("/dev/kvm is not accessible") + ovmf_firmware() + if run(["sudo", "-n", "true"], check=False, capture=True).returncode != 0: + die("sudo needs a password; run `sudo -v` in this terminal first") + + +def cmd_up(args: argparse.Namespace) -> None: + image = Path(args.image).expanduser() + if not image.is_file(): + die(f"image not found: {image}") + check_prerequisites() + stop_vm() + agent, controller = build_binaries() + public_key = ensure_ssh_key() + + create_network() + create_cluster(args.node_image) + install_controller(controller) + version = kubernetes_version() + register_machine(version) + token = create_bootstrap_token() + kubectl("delete", "node", NODE_NAME, "--ignore-not-found", quiet=True) + + serve_base = f"http://{GATEWAY}:{SERVE_PORT}" + doc = render_ignition(agent, base_config(token, version), f"{serve_base}/{AGENT_ARCHIVE}", + sha256(STATE / AGENT_ARCHIVE)) + ignition = STATE / IGNITION_NAME + ignition.write_text(json.dumps(add_harness_access(doc, public_key), indent=2) + "\n") + ignition.chmod(0o600) + + # The config server only runs while the VM provisions. Later boots must not need it. + with FileServer({IGNITION_NAME: ignition, AGENT_ARCHIVE: STATE / AGENT_ARCHIVE}): + launch_vm(image, f"{serve_base}/{IGNITION_NAME}") + wait_for_ssh() + wait_for_bootstrap() + wait_for_node_ready() + log(f"node {NODE_NAME} is Ready; run `{Path(sys.argv[0]).name} test` next") + + +def cmd_test(args: argparse.Namespace) -> None: + if not vm_running(): + die("the VM is not running; run `up` first") + checks = Checks() + check_first_boot(checks) + check_workload(checks) + if args.reboot: + check_reboot(checks) + if args.reset != "none": + bootstrap_was_active = check_before_reset(checks) + reset_node(args.reset) + check_reset(checks, bootstrap_was_active) + checks.done() + + +def cmd_down(_: argparse.Namespace) -> None: + stop_vm() + delete_network() + if shutil.which("kind") and cluster_exists(): + run(["kind", "delete", "cluster", "--name", CLUSTER]) + shutil.rmtree(STATE, ignore_errors=True) + log("removed the VM, network, cluster, and state") + + +def cmd_ssh(args: argparse.Namespace) -> None: + os.execvp("ssh", ["ssh", *SSH_OPTS, f"{SSH_USER}@{VM_IP}", *args.command]) + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__.split("\n\n", 1)[0]) + commands = parser.add_subparsers(dest="command", required=True) + + up = commands.add_parser("up", help="build, create the cluster and network, and boot the VM") + up.add_argument("--image", default=os.environ.get("ACL_IMAGE", ""), required="ACL_IMAGE" not in os.environ, + help="Azure Container Linux qcow2 image (or ACL_IMAGE)") + up.add_argument("--node-image", default=os.environ.get("ACL_KIND_NODE_IMAGE", ""), + help="kind node image, for another Kubernetes version") + up.set_defaults(func=cmd_up) + + test = commands.add_parser("test", help="check the node, reboot it, reset it, and check cleanup") + test.add_argument("--reset", choices=["agent-reset", "cli", "none"], default="agent-reset", + help="how to reset the node at the end (default: an AgentReset MachineOperation)") + test.add_argument("--no-reboot", dest="reboot", action="store_false", + help="skip the reboot, so that reset runs while the first-boot unit is active") + test.set_defaults(func=cmd_test) + + down = commands.add_parser("down", help="remove the VM, network, cluster, and state") + down.set_defaults(func=cmd_down) + + ssh_cmd = commands.add_parser("ssh", help="open a shell on the VM, or run a command there") + ssh_cmd.add_argument("command", nargs=argparse.REMAINDER) + ssh_cmd.set_defaults(func=cmd_ssh) + + args = parser.parse_args() + args.func(args) + + +if __name__ == "__main__": + main() diff --git a/hack/acl/ukiboot.py b/hack/acl/ukiboot.py new file mode 100644 index 00000000..b472aed5 --- /dev/null +++ b/hack/acl/ukiboot.py @@ -0,0 +1,618 @@ +#!/usr/bin/env python3 +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +"""Add kernel command line arguments to a Unified Kernel Image disk. + +Azure Container Linux is a Flatcar-derived image: an EFI system partition holds +a UKI that shim and systemd-boot load, /usr is a dm-verity btrfs image mounted +read-only, and first-boot provisioning is Ignition rather than cloud-init. QEMU +has no way to append to the command line of a UKI booted that way, and the +command line is where an Ignition config source and early networking are named. + +The image's own boot chain has to be left intact, because Ignition's +once-only behavior depends on it. systemd-stub assembles the command line from +the UKI's .cmdline section plus every addon in its .extra.d directory, and +ignition-quench.service deletes firstboot.addon.efi after a successful first +boot so that systemd-boot stops appending flatcar.first_boot. Booting the +kernel and initrd directly with -append bypasses that, which makes every boot +look like a first boot: Ignition re-runs, re-fetches, and the boot fails. + +So instead of replacing the boot chain, this appends to it. The .cmdline +sections of the shipped addons are padded well beyond their contents, so an +addon can be extended in place: no cluster allocation, no directory entry +changes, just bytes rewritten inside an existing file and the section header's +VirtualSize adjusted to match. + +Writes go through qemu-nbd over a unix socket, so no loop device, no nbd kernel +module and no privileges are involved. Point this at a qcow2 overlay and the +backing image is untouched. +""" +from __future__ import annotations + +import os +import re +import socket +import struct +import subprocess +import sys +import tempfile +import time +from dataclasses import dataclass +from pathlib import Path + +# NBD protocol constants (fixed newstyle handshake). +NBD_OPT_GO = 7 +NBD_REP_ACK = 1 +NBD_REP_INFO = 3 +NBD_INFO_EXPORT = 0 +NBD_CMD_READ = 0 +NBD_CMD_WRITE = 1 +NBD_CMD_FLUSH = 3 +NBD_FLAG_C_FIXED_NEWSTYLE = 1 +NBD_REQUEST_MAGIC = 0x25609513 +NBD_SIMPLE_REPLY_MAGIC = 0x67446698 +NBD_OPT_REPLY_MAGIC = 0x3E889045565A9 +NBD_REP_ERROR_BIT = 0x80000000 + +EFI_SYSTEM_PARTITION_TYPE = "c12a7328-f81f-11d2-ba4b-00a0c93ec93b" + + +class NbdClient: + """Minimal NBD client: one export, random-access reads and writes.""" + + def __init__(self, sock_path: str): + self.sock = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) + self.sock.connect(sock_path) + self.size = self._handshake() + self._handle = 0 + + def _recv(self, n: int) -> bytes: + buf = b"" + while len(buf) < n: + chunk = self.sock.recv(n - len(buf)) + if not chunk: + raise EOFError("NBD connection closed") + buf += chunk + return buf + + def _handshake(self) -> int: + if self._recv(8) != b"NBDMAGIC": + raise RuntimeError("not an NBD server") + if self._recv(8) != b"IHAVEOPT": + raise RuntimeError("server does not speak fixed newstyle NBD") + self._recv(2) # handshake flags + self.sock.sendall(struct.pack(">I", NBD_FLAG_C_FIXED_NEWSTYLE)) + + payload = struct.pack(">I", 0) + struct.pack(">H", 0) # default export, no info requests + self.sock.sendall(b"IHAVEOPT" + struct.pack(">II", NBD_OPT_GO, len(payload)) + payload) + + size = 0 + while True: + magic, option, rep_type, length = struct.unpack(">QIII", self._recv(20)) + if magic != NBD_OPT_REPLY_MAGIC: + raise RuntimeError(f"bad NBD option reply magic {magic:#x}") + data = self._recv(length) if length else b"" + if rep_type == NBD_REP_INFO and len(data) >= 10: + if struct.unpack(">H", data[:2])[0] == NBD_INFO_EXPORT: + size = struct.unpack(">Q", data[2:10])[0] + elif rep_type == NBD_REP_ACK: + return size + elif rep_type & NBD_REP_ERROR_BIT: + raise RuntimeError(f"NBD option {option} rejected ({rep_type:#x}): {data!r}") + + def _request(self, cmd: int, offset: int, length: int, data: bytes = b"") -> None: + self._handle += 1 + self.sock.sendall(struct.pack( + ">IHHQQI", NBD_REQUEST_MAGIC, 0, cmd, self._handle, offset, length)) + if data: + self.sock.sendall(data) + + def _reply(self, offset: int) -> None: + magic, error, _handle = struct.unpack(">IIQ", self._recv(16)) + if magic != NBD_SIMPLE_REPLY_MAGIC: + raise RuntimeError(f"bad NBD reply magic {magic:#x}") + if error: + raise RuntimeError(f"NBD error {error} at offset {offset}") + + def read(self, offset: int, length: int) -> bytes: + out = bytearray() + while length > 0: + n = min(length, 4 << 20) + self._request(NBD_CMD_READ, offset, n) + self._reply(offset) + out += self._recv(n) + offset += n + length -= n + return bytes(out) + + def write(self, offset: int, data: bytes) -> None: + view = memoryview(data) + while view: + chunk = view[: 4 << 20] + self._request(NBD_CMD_WRITE, offset, len(chunk), bytes(chunk)) + self._reply(offset) + offset += len(chunk) + view = view[len(chunk):] + + def flush(self) -> None: + self._request(NBD_CMD_FLUSH, 0, 0) + self._reply(0) + + def close(self) -> None: + try: + self.sock.close() + except OSError: + # Best effort: the caller is already tearing down, and the export + # has been flushed by this point. A failure to close a socket that + # is about to be discarded is not worth masking the reason the + # caller is unwinding. + pass + + +class NbdServer: + """qemu-nbd serving a disk image on a unix socket in a temporary directory.""" + + def __init__(self, image: str, image_format: str = "qcow2", writable: bool = False): + self._dir = tempfile.mkdtemp(prefix="ukiboot-") + self.sock_path = os.path.join(self._dir, "nbd.sock") + self.proc: subprocess.Popen[bytes] | None = None + + args = ["qemu-nbd", "--persistent", "--format", image_format, + "--socket", self.sock_path] + if not writable: + args.append("--read-only") + args.append(image) + + # __exit__ does not run when the constructor raises, so every failure + # here cleans up itself. + try: + self.proc = subprocess.Popen(args, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE) + + deadline = time.time() + 15 + while time.time() < deadline: + if os.path.exists(self.sock_path): + return + if self.proc.poll() is not None: + err = self.proc.stderr.read().decode("utf-8", "replace") if self.proc.stderr else "" + raise RuntimeError(f"qemu-nbd exited: {err}") + time.sleep(0.05) + raise RuntimeError("qemu-nbd did not create its socket in time") + except BaseException: + self.close() + raise + + def close(self) -> None: + if self.proc is not None: + self.proc.terminate() + try: + self.proc.wait(timeout=10) + except subprocess.TimeoutExpired: + self.proc.kill() + # Reap it: without this the killed qemu-nbd stays a zombie for the + # lifetime of the harness, which can outlast many VM cycles. + self.proc.wait() + for cleanup in (lambda: os.unlink(self.sock_path), lambda: os.rmdir(self._dir)): + try: + cleanup() + except FileNotFoundError: + # Already gone, which is the ordinary case when qemu-nbd + # removed its own socket on exit. + pass + except OSError as exc: + # Report rather than raise. close() runs from __exit__ and from + # the startup failure path above, so raising here would replace + # the reason the caller is unwinding with a cleanup detail, + # which is how the actual failure gets lost. + print(f"warning: ukiboot cleanup failed: {exc}", file=sys.stderr) + + def __enter__(self) -> "NbdServer": + return self + + def __exit__(self, *_exc: object) -> None: + self.close() + + +@dataclass(frozen=True) +class Partition: + name: str + type_guid: str + first_lba: int + last_lba: int + + @property + def offset(self) -> int: + return self.first_lba * 512 + + +def read_partitions(dev: NbdClient) -> list[Partition]: + header = dev.read(512, 512) + if header[:8] != b"EFI PART": + raise RuntimeError("disk has no GPT") + entries_lba = struct.unpack_from(" str: + d1, d2, d3 = struct.unpack_from(" bytes: + return self.dev.read(self.base + offset, length) + + def _cluster_runs(self, cluster: int) -> list[tuple[int, int]]: + """Collapse a cluster chain into contiguous (partition offset, length) runs.""" + chain = [] + while 2 <= cluster < 0x0FFFFFF8: + chain.append(cluster) + cluster = struct.unpack_from(" list[tuple[int, int]]: + """Map a byte range of a file to absolute (disk offset, length) ranges. + + A file is not necessarily contiguous, so a range can span several runs. + Returning them lets a caller read or write the range without assuming + it lies in one piece. + """ + out: list[tuple[int, int]] = [] + remaining = length + pos = 0 + for run_start, run_len in self._cluster_runs(cluster): + if remaining <= 0: + break + run_end = pos + run_len + if run_end > offset: + skip = max(0, offset - pos) + take = min(run_len - skip, remaining) + out.append((self.base + run_start + skip, take)) + remaining -= take + pos = run_end + if remaining > 0: + raise RuntimeError("range extends past the end of the file") + return out + + def read_file(self, cluster: int, size: int, offset: int = 0, + length: int | None = None) -> bytes: + """Read a byte range of a file without materializing the whole file.""" + if length is None: + length = size - offset + length = max(0, min(length, size - offset)) + if length == 0: + return b"" + return b"".join(self.dev.read(start, count) + for start, count in self.map_ranges(cluster, offset, length)) + + def write_file(self, cluster: int, size: int, offset: int, data: bytes) -> None: + """Overwrite a byte range of a file in place. + + The file keeps its length and its clusters; only the bytes change. That + is what makes this safe without a FAT allocator. + """ + if offset + len(data) > size: + raise RuntimeError("in-place write would extend the file") + view = memoryview(data) + for start, count in self.map_ranges(cluster, offset, len(data)): + self.dev.write(start, bytes(view[:count])) + view = view[count:] + + def list_dir(self, cluster: int) -> list[tuple[str, int, int, int]]: + """Return (name, attributes, start cluster, size) for each entry.""" + data = b"".join(self.dev.read(self.base + start, length) + for start, length in self._cluster_runs(cluster)) + entries: list[tuple[str, int, int, int]] = [] + long_name: list[tuple[int, str]] = [] + for i in range(0, len(data), 32): + entry = data[i:i + 32] + if len(entry) < 32 or entry[0] == 0: + break + if entry[0] == 0xE5: + long_name = [] + continue + if entry[11] == 0x0F: + text = (entry[1:11] + entry[14:26] + entry[28:32]).decode("utf-16-le", "ignore") + long_name.append((entry[0] & 0x3F, text.split("\x00")[0])) + continue + if long_name: + name = "".join(text for _, text in sorted(long_name)) + else: + stem = entry[0:8].decode("ascii", "replace").rstrip() + ext = entry[8:11].decode("ascii", "replace").rstrip() + name = f"{stem}.{ext}" if ext else stem + long_name = [] + start = ((struct.unpack_from(" tuple[int, int] | None: + """Resolve a path to (start cluster, size). FAT lookups ignore case.""" + cluster = self.root_cluster + parts = [p for p in path.split("/") if p] + for index, part in enumerate(parts): + for name, _attr, start, size in self.list_dir(cluster): + if name.lower() != part.lower(): + continue + if index == len(parts) - 1: + return start, size + cluster = start + break + else: + return None + return None + + def list_names(self, path: str) -> list[str]: + found = self.lookup(path) + if not found: + return [] + return [name for name, _attr, _start, _size in self.list_dir(found[0]) + if name not in (".", "..")] + + +def pe_section_table_offset(header: bytes) -> tuple[int, int, int]: + """Return (section table offset, section count, optional header size).""" + lfanew = struct.unpack_from(" dict[str, tuple[int, int, int, int]]: + """Return {section: (virtual size, virtual address, raw size, raw pointer)}.""" + table, count, _opt = pe_section_table_offset(header) + out = {} + for i in range(count): + entry = header[table + i * 40: table + (i + 1) * 40] + out[entry[0:8].rstrip(b"\x00").decode()] = struct.unpack_from(" int: + """Return the byte offset of a section's VirtualSize field in the header. + + systemd-stub reads VirtualSize bytes from the .cmdline section, so a longer + command line is truncated unless this field is updated to match. It is the + one write ukiboot makes outside the section body, and the only one that + lands in a PE header, where being four bytes out overwrites the section's + VirtualAddress instead and produces an executable that loads its command + line from nowhere. + """ + table, _count, _opt = pe_section_table_offset(header) + + return table + _pe_section_index(header, name) * 40 + 8 + + +def _pe_section_index(header: bytes, name: str) -> int: + table, count, _opt = pe_section_table_offset(header) + for i in range(count): + entry = header[table + i * 40: table + (i + 1) * 40] + if entry[0:8].rstrip(b"\x00").decode() == name: + return i + raise RuntimeError(f"PE image has no {name} section") + + +@dataclass(frozen=True) +class PatchedAddon: + """Where the patch landed, for logging and verification.""" + + addon: str + cmdline: str + used: int + capacity: int + + +def single_uki(names: list[str], image: Path) -> str: + """Return the one UKI under /EFI/Linux. + + systemd-boot picks among several by its own rules, so choosing one here + could patch an image that is not the one that boots, and Ignition would + then get no config URL. + """ + ukis = sorted(n for n in names if n.lower().endswith(".efi")) + if len(ukis) != 1: + raise RuntimeError(f"{image} has {len(ukis)} UKIs under /EFI/Linux, expected one: {ukis}") + return ukis[0] + + +def fit_cmdline(current: str, extra: str, raw_size: int) -> bytes | None: + """Return the merged command line encoded, or None if it and its NUL + terminator do not fit in a section of raw_size bytes. Sizes are in encoded + bytes, since that is what the section holds.""" + encoded = f"{current} {extra}".strip().encode() + if len(encoded) + 1 > raw_size: + return None + return encoded + + +def patch_uki_cmdline_addon(image: Path, extra_args: str, + image_format: str = "qcow2") -> PatchedAddon: + """Append kernel command line arguments to a UKI addon on the image's ESP. + + Chooses the largest .cmdline addon that can hold the addition, appends to + its existing contents, and updates the section's VirtualSize so + systemd-stub reads the longer string. firstboot.addon.efi is never chosen: + ignition-quench.service deletes it after a successful first boot, which is + exactly the mechanism that stops Ignition re-running, and the addition has + to survive that. + """ + with NbdServer(str(image), image_format, writable=True) as server: + dev = NbdClient(server.sock_path) + try: + esp = next((p for p in read_partitions(dev) + if p.type_guid == EFI_SYSTEM_PARTITION_TYPE), None) + if esp is None: + raise RuntimeError(f"{image} has no EFI system partition") + fat = Fat32(dev, esp.offset) + + addon_dir = f"/EFI/Linux/{single_uki(fat.list_names('/EFI/Linux'), image)}.extra.d" + + best = None + for addon in sorted(fat.list_names(addon_dir)): + if not addon.lower().endswith(".efi") or addon == "firstboot.addon.efi": + continue + entry = fat.lookup(f"{addon_dir}/{addon}") + if entry is None: + continue + cluster, size = entry + header = fat.read_file(cluster, size, 0, min(size, 8192)) + sections = pe_sections(header) + if ".cmdline" not in sections: + continue + vsize, _vaddr, rsize, rptr = sections[".cmdline"] + current = fat.read_file(cluster, size, rptr, min(vsize, rsize) if vsize else rsize) + current = current.split(b"\x00")[0].decode("utf-8", "replace").strip() + merged = fit_cmdline(current, extra_args, rsize) + if merged is None: + continue + if best is None or rsize > best[0]: + best = (rsize, addon, cluster, size, header, rptr, merged) + + if best is None: + raise RuntimeError( + f"no addon under {addon_dir} has room for {len(extra_args)} more bytes") + + rsize, addon, cluster, size, header, rptr, merged = best + + # Rewrite the section body, NUL-padded to its full raw size so no + # remnant of the previous contents is left behind. + body = merged + b"\x00" * (rsize - len(merged)) + fat.write_file(cluster, size, rptr, body) + + # systemd-stub reads VirtualSize bytes, so a longer string is + # truncated unless the header agrees. + vsize_offset = pe_section_vsize_offset(header, ".cmdline") + fat.write_file(cluster, size, vsize_offset, struct.pack(" str: + """Return the command line systemd-stub would assemble for the image's UKI. + + This is the UKI's own .cmdline plus every addon in its .extra.d directory, + in the order systemd-stub reads them. Used to check what a patch produced. + """ + with NbdServer(str(image), image_format) as server: + dev = NbdClient(server.sock_path) + try: + esp = next((p for p in read_partitions(dev) + if p.type_guid == EFI_SYSTEM_PARTITION_TYPE), None) + if esp is None: + raise RuntimeError(f"{image} has no EFI system partition") + fat = Fat32(dev, esp.offset) + + uki_name = single_uki(fat.list_names("/EFI/Linux"), image) + + parts: list[str] = [] + + def section_text(cluster: int, size: int) -> str | None: + header = fat.read_file(cluster, size, 0, min(size, 8192)) + sections = pe_sections(header) + if ".cmdline" not in sections: + return None + vsize, _vaddr, rsize, rptr = sections[".cmdline"] + raw = fat.read_file(cluster, size, rptr, min(vsize, rsize) if vsize else rsize) + return raw.split(b"\x00")[0].decode("utf-8", "replace").strip() + + entry = fat.lookup(f"/EFI/Linux/{uki_name}") + if entry: + text = section_text(*entry) + if text: + parts.append(text) + + addon_dir = f"/EFI/Linux/{uki_name}.extra.d" + for addon in sorted(fat.list_names(addon_dir)): + if not addon.lower().endswith(".efi"): + continue + found = fat.lookup(f"{addon_dir}/{addon}") + if found is None: + continue + text = section_text(*found) + if text: + parts.append(text) + + return re.sub(r"\s+", " ", " ".join(parts)).strip() + finally: + dev.close() + + +def main() -> None: + import argparse + + parser = argparse.ArgumentParser(description=__doc__.split("\n", maxsplit=1)[0]) + parser.add_argument("image", type=Path) + parser.add_argument("--append", help="kernel command line arguments to add") + parser.add_argument("--show", action="store_true", help="print the assembled command line") + args = parser.parse_args() + + if args.append: + result = patch_uki_cmdline_addon(args.image, args.append) + print(f"patched: {result.addon} ({result.used}/{result.capacity} bytes)") + print(f"cmdline: {result.cmdline}") + + if args.show or not args.append: + print(f"assembled: {read_uki_cmdline(args.image)}") + + +if __name__ == "__main__": + main() diff --git a/hack/docs/url-contract.json b/hack/docs/url-contract.json index b7df308c..ae517982 100644 --- a/hack/docs/url-contract.json +++ b/hack/docs/url-contract.json @@ -476,6 +476,7 @@ "bootstrap-token-expired", "clean-up", "flow", + "hosts-provisioned-by-ignition", "operator-guide-bootstrap-an-aks-flex-node", "preflight-reports-insufficient-disk-space", "prerequisites", @@ -516,6 +517,17 @@ "verify-node-state" ] }, + { + "path": "hack/acl/README.md", + "protectAnchors": true, + "anchors": [ + "azure-container-linux-harness", + "requirements", + "settings", + "troubleshooting", + "usage" + ] + }, { "path": "hack/e2e/README.md", "protectAnchors": true, diff --git a/hack/e2e/README.md b/hack/e2e/README.md index 654563e9..a05718df 100644 --- a/hack/e2e/README.md +++ b/hack/e2e/README.md @@ -223,7 +223,7 @@ Run it against an already joined environment: The `nspawn-lifecycle` command validates the host integration exported by the shared Unbounded lifecycle library: 1. Read each node's persisted active machine and require it to be `kube1` or `kube2`. -2. Verify `/usr/local/bin/unbounded-agent-nspawn-lifecycle` is executable and accepts the generated CLI shape. +2. Verify `/opt/unbounded/bin/unbounded-agent-nspawn-lifecycle` is executable and accepts the generated CLI shape. 3. Verify the generated pre-start and post-start systemd hooks invoke that helper with the active machine. 4. Add a marker to the token node's generated `.nspawn` config and invoke `pre-start`, proving the AKS Flex persisted-config loader regenerates the file. 5. Invoke `reconcile`, verify the active machine receives a new leader PID, wait for the Kubernetes node to return Ready, and run a smoke workload. diff --git a/hack/e2e/lib/agent-upgrade.sh b/hack/e2e/lib/agent-upgrade.sh index 63a2d704..1c1e95a0 100644 --- a/hack/e2e/lib/agent-upgrade.sh +++ b/hack/e2e/lib/agent-upgrade.sh @@ -31,16 +31,35 @@ sudo install -d -m 0755 "${work}" sudo install -m 0755 /tmp/aks-flex-node-e2e-upgrade-binary "${work}/aks-flex-node-linux-amd64" sudo tar -C "${work}" -czf "${work}/success.tar.gz" aks-flex-node-linux-amd64 +# It answers host-root the way a current release does, so verification accepts +# it and the failure comes from the daemon. cat >/tmp/aks-flex-node-e2e-broken <<'BROKEN' #!/bin/sh if [ "${1:-}" = "version" ]; then echo "e2e-forced-daemon-failure" exit 0 fi +if [ "${1:-}" = "host-root" ]; then + readlink -m /opt/unbounded + exit 0 +fi exit 42 BROKEN sudo install -m 0755 /tmp/aks-flex-node-e2e-broken "${work}/aks-flex-node-linux-amd64" sudo tar -C "${work}" -czf "${work}/failure.tar.gz" aks-flex-node-linux-amd64 + +# A release before the host root has no host-root command. +cat >/tmp/aks-flex-node-e2e-legacy <<'LEGACY' +#!/bin/sh +if [ "${1:-}" = "version" ]; then + echo "e2e-release-before-the-host-root" + exit 0 +fi +echo "Error: unknown command \"${1:-}\" for \"aks-flex-node\"" >&2 +exit 1 +LEGACY +sudo install -m 0755 /tmp/aks-flex-node-e2e-legacy "${work}/aks-flex-node-linux-amd64" +sudo tar -C "${work}" -czf "${work}/legacy.tar.gz" aks-flex-node-linux-amd64 check_dir="$(mktemp -d)" sudo tar -C "${check_dir}" -xzf "${work}/failure.tar.gz" sudo "${check_dir}/aks-flex-node-linux-amd64" version >/dev/null @@ -107,7 +126,7 @@ _agent_upgrade_snapshot() { local vm_ip="$1" remote_exec "${vm_ip}" 'bash -s' <<'REMOTE' set -euo pipefail -current="$(sudo readlink -f /usr/local/lib/aks-flex-node/aks-flex-node-current)" +current="$(sudo readlink -f /opt/unbounded/lib/aks-flex-node/aks-flex-node-current)" machine="$(sudo python3 - <<'PY' import json with open('/etc/aks-flex-node/daemon-state.json', encoding='utf-8') as stream: @@ -169,8 +188,10 @@ _agent_upgrade_direct_activation() { set -euo pipefail work=/opt/aks-flex-node-e2e-upgrade candidate="${work}/aks-flex-node-direct-candidate" -current_link=/usr/local/lib/aks-flex-node/aks-flex-node-current -last_good_link=/usr/local/lib/aks-flex-node/aks-flex-node-last-good +current_link=/opt/unbounded/lib/aks-flex-node/aks-flex-node-current +last_good_link=/opt/unbounded/lib/aks-flex-node/aks-flex-node-last-good +# The unit names the link under the resolved host root. +unit_current="$(readlink -m /opt/unbounded/lib/aks-flex-node)/aks-flex-node-current" service=/etc/systemd/system/aks-flex-node-agent.service sudo cp "${work}/aks-flex-node-linux-amd64" "${candidate}" @@ -188,9 +209,9 @@ sudo "${candidate}" agent-upgrade | tee /tmp/direct-agent-upgrade.log current_after="$(sudo readlink -f "${current_link}")" [[ "${current_after}" != "${current_before}" ]] [[ "$(sudo readlink -f "${last_good_link}")" == "${current_before}" ]] -[[ "$(sudo readlink -f /usr/local/bin/aks-flex-node)" == "${current_after}" ]] +[[ "$(sudo readlink -f /opt/unbounded/bin/aks-flex-node)" == "${current_after}" ]] [[ "$(sudo sha256sum "${candidate}" | awk '{print $1}')" == "$(sudo sha256sum "${current_after}" | awk '{print $1}')" ]] -sudo grep -Fq "ExecStart=${current_link} agent" "${service}" +sudo grep -Fq "ExecStart=${unit_current} agent" "${service}" sudo systemctl is-active --quiet aks-flex-node-agent.service pid="$(sudo systemctl show --property MainPID --value aks-flex-node-agent.service)" [[ "$(sudo readlink -f "/proc/${pid}/exe")" == "${current_after}" ]] @@ -277,8 +298,118 @@ agent_upgrade_e2e() { _agent_upgrade_assert_synchronized "${vm_ip}" validate_node_joined "${vm_name}" + # A release before the host root would look for its files under /usr/local, + # where this host has none, so verification refuses it and nothing changes. + local legacy_op="agent-upgrade-legacy-${suffix}" legacy_digest refused_snapshot refused_binary_digest + legacy_digest="$(_agent_upgrade_digest "${vm_ip}" legacy.tar.gz)" + _agent_upgrade_apply "${legacy_op}" "${vm_name}" legacy.tar.gz "${legacy_digest}" "legacy-${suffix}" + _agent_upgrade_wait_phase "${legacy_op}" Failed + if ! kubectl get machineoperation "${legacy_op}" -o jsonpath='{.status.message}' | grep -q "predates the host root"; then + log_error "AgentUpgrade to a release before the host root was not refused by verification" + kubectl get machineoperation "${legacy_op}" -o yaml || true + return 1 + fi + refused_snapshot="$(_agent_upgrade_snapshot "${vm_ip}")" + IFS='|' read -r _ _ refused_binary_digest _ <<<"${refused_snapshot}" + if [[ "${refused_binary_digest}" != "${retry_binary_digest}" ]]; then + log_error "A refused AgentUpgrade changed the active binary: ${refused_snapshot}" + return 1 + fi + remote_exec "${vm_ip}" 'sudo systemctl is-active --quiet aks-flex-node-agent.service' + _agent_upgrade_direct_activation "${vm_name}" "${vm_ip}" smoke_test "${vm_name}" "agent-upgrade" log_success "Managed AgentUpgrade success/rollback/retry and direct host activation E2E passed" } + +# --------------------------------------------------------------------------- +# host_root_migration_e2e - Move a host installed by a release before the host +# root to the build under test +# --------------------------------------------------------------------------- +_host_root_migration_uninstall() { + local vm_ip="$1" + remote_copy "${REPO_ROOT}/scripts/uninstall.sh" "${vm_ip}" "/tmp/aks-flex-node-uninstall.sh" + remote_exec "${vm_ip}" 'bash -s' <<'REMOTE' +set -euo pipefail +sudo bash /tmp/aks-flex-node-uninstall.sh --force +for path in /opt/unbounded /usr/local/bin/aks-flex-node /usr/local/lib/aks-flex-node /etc/aks-flex-node; do + if [[ -e "${path}" || -L "${path}" ]]; then + echo "uninstall left ${path}" >&2 + exit 1 + fi +done +REMOTE +} + +# _host_root_migration_assert checks the host root. "legacy" is a host the older +# release installed, with nothing at /opt/unbounded. "migrated" is the same host +# once this build has run: /opt/unbounded links to /usr/local, and the unit and +# links the older release wrote are unchanged, so it could still be returned to. +_host_root_migration_assert() { + local vm_ip="$1" state="$2" + remote_exec "${vm_ip}" "STATE=${state} bash -s" <<'REMOTE' +set -euo pipefail +service=/etc/systemd/system/aks-flex-node-agent.service +case "${STATE}" in + legacy) + if [[ -e /opt/unbounded || -L /opt/unbounded ]]; then + echo "a release before the host root created /opt/unbounded" >&2 + exit 1 + fi + ;; + migrated) + if [[ ! -L /opt/unbounded || "$(readlink /opt/unbounded)" != /usr/local ]]; then + echo "/opt/unbounded is not a link to /usr/local: $(ls -ld /opt/unbounded 2>&1)" >&2 + exit 1 + fi + ;; +esac +[[ "$(readlink /usr/local/bin/aks-flex-node)" == /usr/local/lib/aks-flex-node/aks-flex-node-current ]] || + { echo "compatibility link is $(readlink /usr/local/bin/aks-flex-node)" >&2; exit 1; } +sudo grep -Fq "ExecStart=/usr/local/lib/aks-flex-node/aks-flex-node-current agent" "${service}" || + { echo "agent unit does not run the legacy current link:" >&2; sudo cat "${service}" >&2; exit 1; } +sudo systemctl is-active --quiet aks-flex-node-agent.service +REMOTE +} + +host_root_migration_e2e() { + local release="${E2E_LEGACY_RELEASE:-v0.2.0}" + log_section "Host Root Migration from ${release}" + local vm_name vm_ip suffix success_digest op snapshot slot + vm_name="$(state_get token_vm_name)" + vm_ip="$(state_get token_vm_ip)" + suffix="$(date +%s)" + + # Start over from a host the older release installed. + node_unjoin_token + _host_root_migration_uninstall "${vm_ip}" + node_join_token "${release}" + validate_node_joined "${vm_name}" + _host_root_migration_assert "${vm_ip}" legacy + + # The older daemon performs the upgrade. This build's daemon then starts on a + # host it did not install, and links the host root to the older layout. + _agent_upgrade_ensure_api + remote_exec "${vm_ip}" 'sudo systemctl restart aks-flex-node-agent.service' + validate_node_joined "${vm_name}" + _agent_upgrade_prepare_server "${vm_ip}" + success_digest="$(_agent_upgrade_digest "${vm_ip}" success.tar.gz)" + op="host-root-migration-${suffix}" + _agent_upgrade_apply "${op}" "${vm_name}" success.tar.gz "${success_digest}" "migration-${suffix}" + _agent_upgrade_wait_phase "${op}" Complete + validate_node_joined "${vm_name}" + _host_root_migration_assert "${vm_ip}" migrated + + snapshot="$(_agent_upgrade_snapshot "${vm_ip}")" + IFS='|' read -r slot _ _ _ <<<"${snapshot}" + if [[ "${slot}" != /usr/local/lib/aks-flex-node/* ]]; then + log_error "The migrated host is not running from the older layout: ${snapshot}" + return 1 + fi + _agent_upgrade_assert_synchronized "${vm_ip}" + _agent_upgrade_validate_kubelet_auth "${vm_name}" "${vm_ip}" + smoke_test "${vm_name}" "host-root-migration" + + log_success "Host root migration from ${release} passed" +} diff --git a/hack/e2e/lib/arm-machine-registration.sh b/hack/e2e/lib/arm-machine-registration.sh index d488d1d3..f511ad71 100644 --- a/hack/e2e/lib/arm-machine-registration.sh +++ b/hack/e2e/lib/arm-machine-registration.sh @@ -318,7 +318,7 @@ _reset_arm_registration_host() { remote_exec "${vm_ip}" "sudo bash -s" <<'REMOTE' || return 1 set -euo pipefail -/usr/local/bin/aks-flex-node reset +/opt/unbounded/bin/aks-flex-node reset systemctl stop aks-flex-node-msi-arm-registration.service 2>/dev/null || true systemctl reset-failed aks-flex-node-msi-arm-registration.service 2>/dev/null || true REMOTE @@ -338,9 +338,12 @@ _reset_previous_arm_registration_host() { remote_exec "${vm_ip}" "sudo bash -s" <<'REMOTE' || return 1 set -euo pipefail -if command -v /usr/local/bin/aks-flex-node >/dev/null 2>&1; then - /usr/local/bin/aks-flex-node reset -fi +for binary in /opt/unbounded/bin/aks-flex-node /usr/local/bin/aks-flex-node; do + if [[ -x "${binary}" ]]; then + "${binary}" reset + break + fi +done systemctl stop aks-flex-node-msi-arm-registration.service 2>/dev/null || true systemctl reset-failed aks-flex-node-msi-arm-registration.service 2>/dev/null || true REMOTE diff --git a/hack/e2e/lib/node-join-token.sh b/hack/e2e/lib/node-join-token.sh index 1c8a413d..77f07c0d 100644 --- a/hack/e2e/lib/node-join-token.sh +++ b/hack/e2e/lib/node-join-token.sh @@ -16,10 +16,12 @@ readonly _E2E_NODE_JOIN_TOKEN_LOADED=1 source "$(dirname "${BASH_SOURCE[0]}")/common.sh" # --------------------------------------------------------------------------- -# node_join_token - Join the Token VM +# node_join_token - Join the Token VM, with the build under test or, given a +# release tag, with that release # --------------------------------------------------------------------------- node_join_token() { - log_section "Joining Token Node" + local release="${1:-}" + log_section "Joining Token Node${release:+ with ${release}}" local start start=$(timer_start) @@ -96,7 +98,7 @@ node_join_token() { # Step 3: Publish the AKS Machine goal and deploy the agent. machine_configmap_upsert "$(state_get token_vm_name)" "${E2E_KUBERNETES_VERSION}" "${E2E_KUBERNETES_VERSION}" - _deploy_and_start_agent "${vm_ip}" "${config_file}" "aks-flex-node-token" + _deploy_and_start_agent "${vm_ip}" "${config_file}" "aks-flex-node-token" "${release}" log_success "Token node joined in $(timer_elapsed "${start}")s" } diff --git a/hack/e2e/lib/node-join.sh b/hack/e2e/lib/node-join.sh index 1d7a6c83..4c7c6bba 100755 --- a/hack/e2e/lib/node-join.sh +++ b/hack/e2e/lib/node-join.sh @@ -26,10 +26,13 @@ source "$(dirname "${BASH_SOURCE[0]}")/controller.sh" # --------------------------------------------------------------------------- # Internal: upload binary & config, start agent on a VM # --------------------------------------------------------------------------- +# An optional fourth argument names an AKS Flex Node release to install from +# GitHub instead of the build under test. _deploy_and_start_agent() { local vm_ip="$1" local config_file="$2" local unit_name="$3" + local release="${4:-}" log_info "Uploading binary and config to ${vm_ip}..." remote_copy "${E2E_BINARY}" "${vm_ip}" "/tmp/aks-flex-node-binary" @@ -37,12 +40,32 @@ _deploy_and_start_agent() { remote_copy "${REPO_ROOT}/scripts/install.sh" "${vm_ip}" "/tmp/aks-flex-node-install.sh" log_info "Installing and starting flex node agent on ${vm_ip}..." - remote_exec "${vm_ip}" "UNIT_NAME=${unit_name} E2E_NODE_JOIN_TIMEOUT=${E2E_NODE_JOIN_TIMEOUT} E2E_KUBERNETES_VERSION=${E2E_KUBERNETES_VERSION} bash -s" <<'REMOTE' + remote_exec "${vm_ip}" "UNIT_NAME=${unit_name} RELEASE=${release} E2E_NODE_JOIN_TIMEOUT=${E2E_NODE_JOIN_TIMEOUT} E2E_KUBERNETES_VERSION=${E2E_KUBERNETES_VERSION} bash -s" <<'REMOTE' set -euo pipefail -managed_current=/usr/local/lib/aks-flex-node/aks-flex-node-current -if [[ -e "${managed_current}" || -L "${managed_current}" ]]; then - echo "Existing managed layout found; activating the separately staged E2E candidate..." +# The agent is installed under the host root. A layout an earlier join left is +# found there, or under /usr/local on a host an older release installed, where +# activating the candidate links the host root to it. +host_root=/opt/unbounded +managed_current="" +for root in "${host_root}" /usr/local; do + if [[ -e "${root}/lib/aks-flex-node/aks-flex-node-current" || -L "${root}/lib/aks-flex-node/aks-flex-node-current" ]]; then + managed_current="${root}/lib/aks-flex-node/aks-flex-node-current" + break + fi +done +if [[ -n "${RELEASE}" ]]; then + echo "Installing AKS Flex Node ${RELEASE} from its GitHub release..." + sudo AKS_FLEX_NODE_VERSION="${RELEASE}" bash /tmp/aks-flex-node-install.sh --yes + # A release that predates the host root installs under /usr/local. + if ! sudo /usr/local/bin/aks-flex-node host-root >/dev/null 2>&1; then + host_root=/usr/local + fi +elif [[ -n "${managed_current}" && -e /etc/systemd/system/aks-flex-node-agent.service ]]; then + # Only a layout whose service is still installed needs the upgrade path. + # Reset removes the service and the config directory but keeps the layout, + # and install.sh replaces such a layout and recreates the directories. + echo "Existing managed layout found at ${managed_current}; activating the separately staged E2E candidate..." sudo chmod 0755 /tmp/aks-flex-node-binary sudo /tmp/aks-flex-node-binary agent-upgrade else @@ -51,7 +74,7 @@ else bash /tmp/aks-flex-node-install.sh --yes fi -sudo /usr/local/bin/aks-flex-node version +sudo "${host_root}/bin/aks-flex-node" version sudo cp /tmp/config.json /etc/aks-flex-node/ @@ -81,7 +104,7 @@ echo "Running preflight checks before bootstrap..." set +e { echo "=== preflight ${UNIT_NAME} $(date -Is) ===" - sudo /usr/local/bin/aks-flex-node preflight --config /etc/aks-flex-node/config.json --output text + sudo "${host_root}/bin/aks-flex-node" preflight --config /etc/aks-flex-node/config.json --output text preflight_rc=$? echo "=== preflight ${UNIT_NAME} exit ${preflight_rc} ===" exit "${preflight_rc}" @@ -101,7 +124,7 @@ sudo systemd-run \ --unit="${UNIT_NAME}" \ --description="AKS Flex Node E2E (${UNIT_NAME})" \ --remain-after-exit \ - /usr/local/bin/aks-flex-node bootstrap --config /etc/aks-flex-node/config.json + "${host_root}/bin/aks-flex-node" bootstrap --config /etc/aks-flex-node/config.json echo "Waiting up to ${E2E_NODE_JOIN_TIMEOUT}s for aks-flex-node-agent.service to start..." deadline=$((SECONDS + E2E_NODE_JOIN_TIMEOUT)) @@ -322,6 +345,49 @@ validate_no_policy_routing_state() { fi } +# LocalDNS state used to survive reset, because reset skipped the library's +# LocalDNS cleanup. The MSI node enables LocalDNS, so this is exercised there. +validate_no_localdns_state() { + local path + + # Under both roots: reset does not depend on the host root having been migrated. + for path in /etc/systemd/system/unbounded-localdns-network.service \ + /opt/unbounded/libexec/unbounded-localdns-network \ + /usr/local/libexec/unbounded-localdns-network; do + if [[ -e "${path}" ]]; then + echo "reset cleanup LocalDNS file ${path} still exists" + exit 1 + fi + done + + if [[ -e /sys/class/net/localdns ]]; then + echo "reset cleanup LocalDNS interface localdns still exists" + ip link show localdns || true + exit 1 + fi + + if command -v nft >/dev/null 2>&1 && sudo nft list table ip unbounded_localdns >/dev/null 2>&1; then + echo "reset cleanup LocalDNS nft table unbounded_localdns still exists" + sudo nft list table ip unbounded_localdns || true + exit 1 + fi +} + +validate_no_host_helpers() { + local path + + for path in /opt/unbounded/bin/unbounded-agent-nspawn-lifecycle \ + /usr/local/bin/unbounded-agent-nspawn-lifecycle \ + /opt/unbounded/lib/aks-flex-node/aks-flex-node-recovery.sh \ + /usr/local/lib/aks-flex-node/aks-flex-node-recovery.sh \ + /etc/systemd/system/aks-flex-node-agent-recovery.service; do + if [[ -e "${path}" ]]; then + echo "reset cleanup helper ${path} still exists" + exit 1 + fi + done +} + deadline=$((SECONDS + E2E_NODE_JOIN_TIMEOUT)) while systemctl list-unit-files aks-flex-node-agent.service --no-legend | grep -q '^aks-flex-node-agent.service'; do if (( SECONDS >= deadline )); then @@ -346,6 +412,8 @@ validate_no_overlay_interfaces validate_no_legacy_cni_rules validate_no_wireguard_keys validate_no_policy_routing_state +validate_no_localdns_state +validate_no_host_helpers for path in /etc/aks-flex-node /var/log/aks-flex-node; do if [[ -e "${path}" ]]; then diff --git a/hack/e2e/lib/nspawn-lifecycle.sh b/hack/e2e/lib/nspawn-lifecycle.sh index 0daeff1a..ce407ea6 100644 --- a/hack/e2e/lib/nspawn-lifecycle.sh +++ b/hack/e2e/lib/nspawn-lifecycle.sh @@ -25,7 +25,7 @@ _validate_nspawn_lifecycle_contract() { remote_exec "${vm_ip}" 'bash -s' <<'REMOTE' set -euo pipefail -helper="/usr/local/bin/unbounded-agent-nspawn-lifecycle" +helper="/opt/unbounded/bin/unbounded-agent-nspawn-lifecycle" state_file="/etc/aks-flex-node/daemon-state.json" machine="$(sudo python3 - <<'PY' import json @@ -69,7 +69,7 @@ _reconcile_nspawn_lifecycle() { remote_exec "${vm_ip}" "E2E_NSPAWN_LIFECYCLE_TIMEOUT=${timeout} bash -s" <<'REMOTE' set -euo pipefail -helper="/usr/local/bin/unbounded-agent-nspawn-lifecycle" +helper="/opt/unbounded/bin/unbounded-agent-nspawn-lifecycle" machine="$(sudo python3 - <<'PY' import json with open("/etc/aks-flex-node/daemon-state.json", encoding="utf-8") as state: diff --git a/hack/e2e/run.sh b/hack/e2e/run.sh index 0dd1ccc5..09253208 100755 --- a/hack/e2e/run.sh +++ b/hack/e2e/run.sh @@ -29,6 +29,7 @@ # smoke Run smoke tests only (pods on flex nodes) # nspawn-lifecycle Validate generated lifecycle hooks and restart reconciliation # agent-upgrade Validate managed binary upgrade, rollback, and retry +# host-root-migration Upgrade a token node installed by a release before the host root # upgrade-drift Run controller-machine Kubernetes version drift repave test # logs Collect logs from VMs # cleanup Tear down Azure resources @@ -142,7 +143,7 @@ usage() { parse_args() { while [[ $# -gt 0 ]]; do case "$1" in - all|infra|arm-registration|join|join-msi|join-token|join-offline|join-kubeadm|join-arc|unjoin|unjoin-msi|unjoin-token|unjoin-offline|unjoin-kubeadm|unjoin-arc|validate|validate-absent|smoke|nspawn-lifecycle|agent-upgrade|upgrade-drift|logs|cleanup|runner-cleanup|status) + all|infra|arm-registration|join|join-msi|join-token|join-offline|join-kubeadm|join-arc|unjoin|unjoin-msi|unjoin-token|unjoin-offline|unjoin-kubeadm|unjoin-arc|validate|validate-absent|smoke|nspawn-lifecycle|agent-upgrade|host-root-migration|upgrade-drift|logs|cleanup|runner-cleanup|status) COMMAND="$1"; shift ;; -g|--resource-group) export E2E_RESOURCE_GROUP="$2"; shift 2 ;; -l|--location) export E2E_LOCATION="$2"; shift 2 ;; @@ -220,6 +221,9 @@ cmd_all() { # ── Managed host agent binary upgrade ───────────────────────────────── agent_upgrade_e2e + # ── Upgrade from a release before the host root ─────────────────────── + host_root_migration_e2e + # ── Controller-backed machine repave after agent upgrade ─────────────── upgrade_drift_all @@ -354,6 +358,11 @@ main() { ensure_cluster_dependencies agent_upgrade_e2e ;; + host-root-migration) + ensure_binary + ensure_cluster_dependencies + host_root_migration_e2e + ;; upgrade-drift) ensure_binary ensure_cluster_dependencies diff --git a/pkg/cmd/hostroot/hostroot.go b/pkg/cmd/hostroot/hostroot.go new file mode 100644 index 00000000..e5641003 --- /dev/null +++ b/pkg/cmd/hostroot/hostroot.go @@ -0,0 +1,29 @@ +// Package hostroot provides the hidden host-root command. +package hostroot + +import ( + "fmt" + + "github.com/spf13/cobra" + + "github.com/Azure/AKSFlexNode/pkg/daemon" +) + +// NewCommand returns the host-root command. It prints where this release keeps +// its host-side files on this host, as it will once the host root is migrated, +// and changes nothing. +// +// Its existence is what the install scripts and AgentUpgrade check for: a +// release without it predates the host root and installs under /usr/local. +func NewCommand() *cobra.Command { + return &cobra.Command{ + Use: "host-root", + Short: "Print the directory that holds the agent's host-side files", + Hidden: true, + Args: cobra.NoArgs, + RunE: func(cmd *cobra.Command, _ []string) error { + _, err := fmt.Fprintln(cmd.OutOrStdout(), daemon.PlannedHostRoot()) + return err + }, + } +} diff --git a/pkg/cmd/hostroot/hostroot_test.go b/pkg/cmd/hostroot/hostroot_test.go new file mode 100644 index 00000000..bca468b0 --- /dev/null +++ b/pkg/cmd/hostroot/hostroot_test.go @@ -0,0 +1,48 @@ +package hostroot + +import ( + "bytes" + "testing" + + "github.com/Azure/AKSFlexNode/pkg/daemon" +) + +// TestHostRootCommand pins the output the install scripts and the AgentUpgrade +// downgrade guard read. The guard compares it with the root the running agent +// resolved, so anything but the bare path on one line refuses every upgrade. +func TestHostRootCommand(t *testing.T) { + t.Parallel() + + tests := []struct { + name string + args []string + want string + wantErr bool + }{ + {name: "prints the planned root", args: nil, want: daemon.PlannedHostRoot() + "\n"}, + {name: "rejects arguments", args: []string{"extra"}, wantErr: true}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + + var out bytes.Buffer + cmd := NewCommand() + cmd.SetOut(&out) + cmd.SetErr(&bytes.Buffer{}) + cmd.SetArgs(tt.args) + + err := cmd.Execute() + if (err != nil) != tt.wantErr { + t.Fatalf("Execute() error = %v, wantErr %v", err, tt.wantErr) + } + if !tt.wantErr && out.String() != tt.want { + t.Fatalf("output = %q, want %q", out.String(), tt.want) + } + if !cmd.Hidden { + t.Fatal("host-root is for the agent's own tooling, not for operators") + } + }) + } +} diff --git a/pkg/cmd/ignition/ignition.go b/pkg/cmd/ignition/ignition.go new file mode 100644 index 00000000..cbdf7031 --- /dev/null +++ b/pkg/cmd/ignition/ignition.go @@ -0,0 +1,263 @@ +// Package ignition implements `aks-flex-node ignition`, which renders an Ignition config that +// bootstraps a host on first boot. Hosts such as Azure Container Linux are provisioned only by +// Ignition and have a read-only /usr, so the agent goes under a host prefix. +package ignition + +import ( + "bytes" + "encoding/json" + "errors" + "fmt" + "io" + "os" + "path" + "slices" + "strings" + "unicode" + + "github.com/spf13/cobra" + + "github.com/Azure/AKSFlexNode/pkg/config" +) + +// bootstrapValueFlags and bootstrapSwitches are the bootstrap.sh options that take a value and +// those that do not. TestBootstrapFlagsMatchTheScript keeps them in step with the embedded script. +var ( + bootstrapValueFlags = []string{ + "--auth", + "--msi-client-id", + "--sp-tenant-id", + "--sp-client-id", + "--sp-client-secret-file", + "--sp-client-certificate-file", + "--agent-url", + "--agent-version", + "--agent-sha256", + "--bootstrap-data-api-version", + "--cluster-resource-id", + "--agent-pool-name", + "--resource-manager-endpoint", + "--bootstrap-oci-image", + "--bootstrap-offline-artifacts-source", + "--config-overrides", + "--install-dir", + "--config-path", + } + bootstrapSwitches = []string{"--fetch-bootstrap-data"} +) + +// ownedBootstrapFlags are bootstrap.sh options this command sets itself, with the reason a caller +// cannot pass them. +var ownedBootstrapFlags = map[string]string{ + "--install-dir": "the agent chooses its install directory", + "--config-path": "the agent unit reads the default config path", + "--sp-client-secret-file": "use this command's --sp-client-secret-file, which also writes the file to the host", + "--sp-client-certificate-file": "use this command's --sp-client-certificate-file, which also writes the file to the host", +} + +type options struct { + baseConfigPath string + spClientSecretFile string + spClientCertificateFile string + outputPath string +} + +// NewCommand returns the ignition command. +func NewCommand() *cobra.Command { + var opts options + + cmd := &cobra.Command{ + Use: "ignition [flags] -- BOOTSTRAP_ARGS...", + Short: "Render an Ignition config that bootstraps the host on first boot", + Long: `Render an Ignition config for hosts provisioned by Ignition, such as Azure Container Linux. + +The config writes bootstrap.sh and any service principal credential to the host, and enables a +unit that runs bootstrap.sh with BOOTSTRAP_ARGS once the network is up. The unit retries until the +agent is installed and does not run after that. The script, which carries the base config, is +removed once bootstrap succeeds. + +The agent is installed under /opt/unbounded, which is writable on these hosts. The output contains +the base config and credentials, so treat it as a secret.`, + Example: ` aks-flex-node ignition --base-config base.json -o node.ign -- \ + --auth msi --agent-version v0.1.0 --fetch-bootstrap-data`, + RunE: func(cmd *cobra.Command, args []string) error { + return run(cmd, opts, args) + }, + } + + flags := cmd.Flags() + flags.StringVar(&opts.baseConfigPath, "base-config", "", "Base config JSON for bootstrap.sh; without it bootstrap.sh needs --fetch-bootstrap-data") + flags.StringVar(&opts.spClientSecretFile, "sp-client-secret-file", "", "Service principal client secret file to write to the host") + flags.StringVar(&opts.spClientCertificateFile, "sp-client-certificate-file", "", "Service principal client certificate file to write to the host") + flags.StringVarP(&opts.outputPath, "output", "o", "-", "Output path, or - for stdout; a file is created with mode 0600") + cmd.MarkFlagsMutuallyExclusive("sp-client-secret-file", "sp-client-certificate-file") + + return cmd +} + +func run(cmd *cobra.Command, opts options, bootstrapArgs []string) error { + if dash := cmd.ArgsLenAtDash(); len(bootstrapArgs) > 0 && dash != 0 { + return errors.New("put bootstrap.sh options after --") + } + + in, err := buildRenderInput(opts, bootstrapArgs) + if err != nil { + return err + } + out, err := render(in) + if err != nil { + return err + } + + return writeOutput(cmd.OutOrStdout(), opts.outputPath, out) +} + +// buildRenderInput validates everything before anything is rendered, so a mistake is reported +// here rather than by a host that fails on first boot. +func buildRenderInput(opts options, bootstrapArgs []string) (renderInput, error) { + seen, err := validateBootstrapArgs(bootstrapArgs) + if err != nil { + return renderInput{}, err + } + if !seen["--agent-url"] && !seen["--agent-version"] { + return renderInput{}, errors.New("bootstrap.sh needs --agent-url or --agent-version after --") + } + + var in renderInput + in.bootstrapArgs = bootstrapArgs + + if opts.baseConfigPath != "" { + in.baseConfig, err = loadBaseConfig(opts.baseConfigPath) + if err != nil { + return renderInput{}, err + } + } else if !seen["--fetch-bootstrap-data"] { + return renderInput{}, errors.New("without --base-config, bootstrap.sh needs --fetch-bootstrap-data after --") + } + + in.credential, err = loadCredential(opts) + if err != nil { + return renderInput{}, err + } + + return in, nil +} + +// validateBootstrapArgs checks the arguments against bootstrap.sh's parser and returns the options +// that were given. +func validateBootstrapArgs(args []string) (map[string]bool, error) { + seen := map[string]bool{} + for i := 0; i < len(args); i++ { + arg := args[i] + if reason, ok := ownedBootstrapFlags[arg]; ok { + return nil, fmt.Errorf("bootstrap.sh option %s cannot be passed through: %s", arg, reason) + } + + switch { + case slices.Contains(bootstrapSwitches, arg): + case slices.Contains(bootstrapValueFlags, arg): + if i+1 >= len(args) || args[i+1] == "" { + return nil, fmt.Errorf("bootstrap.sh option %s requires a value", arg) + } + i++ + if err := validateUnitArgument(arg, args[i]); err != nil { + return nil, err + } + default: + return nil, fmt.Errorf("unknown bootstrap.sh option %q", arg) + } + seen[arg] = true + } + + return seen, nil +} + +// validateUnitArgument rejects values that cannot be written into the unit's command line. +func validateUnitArgument(flag, value string) error { + if strings.ContainsFunc(value, unicode.IsControl) { + return fmt.Errorf("value of bootstrap.sh option %s contains a control character", flag) + } + + return nil +} + +// loadBaseConfig returns the base config as compact JSON. +func loadBaseConfig(configPath string) ([]byte, error) { + raw, err := os.ReadFile(configPath) // #nosec G304 -- the operator names the file to embed + if err != nil { + return nil, fmt.Errorf("read base config: %w", err) + } + + var object map[string]json.RawMessage + if err := json.Unmarshal(raw, &object); err != nil || object == nil { + return nil, fmt.Errorf("base config %s must be a JSON object", configPath) + } + + var compact bytes.Buffer + if err := json.Compact(&compact, raw); err != nil { + return nil, fmt.Errorf("base config %s: %w", configPath, err) + } + + return compact.Bytes(), nil +} + +// loadCredential reads the service principal credential to write to the host. The checks are the +// ones bootstrap.sh and the agent apply on the host, made here so that they fail early. +func loadCredential(opts options) (*credential, error) { + switch { + case opts.spClientSecretFile != "": + content, err := config.LoadServicePrincipalCredentialFile(opts.spClientSecretFile) + if err != nil { + return nil, err + } + if len(bytes.TrimSpace(content)) == 0 { + return nil, errors.New("service principal client secret file is empty") + } + + return &credential{ + flag: "--sp-client-secret-file", + hostPath: path.Join(credentialsDir, "sp-client-secret"), + content: content, + }, nil + case opts.spClientCertificateFile != "": + if err := config.ValidateServicePrincipalCertificateFile(opts.spClientCertificateFile); err != nil { + return nil, err + } + content, err := config.LoadServicePrincipalCredentialFile(opts.spClientCertificateFile) + if err != nil { + return nil, err + } + // PKCS#12 is recognized by its suffix, so the host copy keeps it. + name := "sp-client-certificate" + if strings.HasSuffix(strings.ToLower(opts.spClientCertificateFile), ".pfx") { + name += ".pfx" + } + + return &credential{ + flag: "--sp-client-certificate-file", + hostPath: path.Join(credentialsDir, name), + content: content, + }, nil + default: + return nil, nil + } +} + +func writeOutput(stdout io.Writer, outputPath string, out []byte) error { + if outputPath == "-" { + _, err := stdout.Write(out) + return err + } + + // O_EXCL keeps an existing file, possibly readable by others, from being reused for secrets. + f, err := os.OpenFile(outputPath, os.O_WRONLY|os.O_CREATE|os.O_EXCL, 0o600) // #nosec G304 -- the operator names the output file + if err != nil { + return fmt.Errorf("create output: %w", err) + } + if _, err := f.Write(out); err != nil { + _ = f.Close() + return fmt.Errorf("write output: %w", err) + } + + return f.Close() +} diff --git a/pkg/cmd/ignition/ignition_test.go b/pkg/cmd/ignition/ignition_test.go new file mode 100644 index 00000000..f4d55676 --- /dev/null +++ b/pkg/cmd/ignition/ignition_test.go @@ -0,0 +1,694 @@ +package ignition + +import ( + "bytes" + "compress/gzip" + "crypto/rand" + "crypto/rsa" + "crypto/x509" + "crypto/x509/pkix" + "encoding/base64" + "encoding/json" + "encoding/pem" + "io" + "math/big" + "os" + "os/exec" + "path/filepath" + "regexp" + "slices" + "strings" + "testing" + "time" + + "github.com/Azure/AKSFlexNode/pkg/daemon" + "github.com/Azure/AKSFlexNode/scripts" +) + +var msiArgs = []string{"--auth", "msi", "--agent-version", "v0.1.0", "--fetch-bootstrap-data"} + +// TestBootstrapFlagsMatchTheScript keeps the pass-through validation in step with bootstrap.sh, so +// an option added there is not refused here, and one removed there is not accepted here. +func TestBootstrapFlagsMatchTheScript(t *testing.T) { + t.Parallel() + + valueArm := regexp.MustCompile(`(?m)^([ \t]*)(--auth\|[a-z0-9|-]+)\)[ \t]*$`).FindStringSubmatch(scripts.Bootstrap) + if valueArm == nil { + t.Fatal("bootstrap.sh no longer lists its value options in one case arm starting with --auth") + } + scriptValueFlags := strings.Split(valueArm[2], "|") + + // Switches are the other arms of the same case statement, so at the same indentation. Deeper + // arms belong to the case that assigns the values. + var scriptSwitches []string + switchArm := regexp.MustCompile(`(?m)^` + valueArm[1] + `(--[a-z][a-z0-9-]*)\)[ \t]*$`) + for _, m := range switchArm.FindAllStringSubmatch(scripts.Bootstrap, -1) { + scriptSwitches = append(scriptSwitches, m[1]) + } + + for _, tt := range []struct { + name string + script, here []string + }{ + {name: "value options", script: scriptValueFlags, here: bootstrapValueFlags}, + {name: "switches", script: scriptSwitches, here: bootstrapSwitches}, + } { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + + script, here := slices.Sorted(slices.Values(tt.script)), slices.Sorted(slices.Values(tt.here)) + if !slices.Equal(script, here) { + t.Fatalf("bootstrap.sh has %q, the ignition command has %q", script, here) + } + }) + } +} + +func TestOwnedBootstrapFlagsAreScriptFlags(t *testing.T) { + t.Parallel() + + for flag := range ownedBootstrapFlags { + if !slices.Contains(bootstrapValueFlags, flag) { + t.Errorf("owned flag %s is not a bootstrap.sh value option", flag) + } + } +} + +func TestValidateBootstrapArgs(t *testing.T) { + t.Parallel() + + tests := []struct { + name string + args []string + wantErr string + wantSeen []string + }{ + { + name: "value options and switches", + args: []string{"--auth", "msi", "--fetch-bootstrap-data", "--config-overrides", `{"node":{"labels":{"a":"b"}}}`}, + wantSeen: []string{"--auth", "--config-overrides", "--fetch-bootstrap-data"}, + }, + { + name: "a value that looks like an option is taken as the value, as bootstrap.sh does", + args: []string{"--agent-pool-name", "--auth"}, + wantSeen: []string{"--agent-pool-name"}, + }, + {name: "unknown option", args: []string{"--nope"}, wantErr: `unknown bootstrap.sh option "--nope"`}, + {name: "help is not passed through", args: []string{"--help"}, wantErr: "unknown bootstrap.sh option"}, + {name: "positional argument", args: []string{"msi"}, wantErr: `unknown bootstrap.sh option "msi"`}, + {name: "missing value", args: []string{"--auth"}, wantErr: "--auth requires a value"}, + {name: "empty value", args: []string{"--auth", "", "--fetch-bootstrap-data"}, wantErr: "--auth requires a value"}, + {name: "newline in value", args: []string{"--config-overrides", "{\n}"}, wantErr: "control character"}, + {name: "host prefix is not an option", args: []string{"--host-prefix", "/opt/x"}, wantErr: `unknown bootstrap.sh option "--host-prefix"`}, + {name: "install dir is owned", args: []string{"--install-dir", "/opt/x/bin"}, wantErr: "cannot be passed through"}, + {name: "config path is owned", args: []string{"--config-path", "/etc/x.json"}, wantErr: "cannot be passed through"}, + {name: "secret file is owned", args: []string{"--sp-client-secret-file", "/etc/s"}, wantErr: "writes the file to the host"}, + {name: "certificate file is owned", args: []string{"--sp-client-certificate-file", "/etc/c"}, wantErr: "writes the file to the host"}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + + seen, err := validateBootstrapArgs(tt.args) + if tt.wantErr != "" { + if err == nil || !strings.Contains(err.Error(), tt.wantErr) { + t.Fatalf("validateBootstrapArgs() error = %v, want %q", err, tt.wantErr) + } + return + } + if err != nil { + t.Fatalf("validateBootstrapArgs() error = %v", err) + } + var got []string + for flag := range seen { + got = append(got, flag) + } + if slices.Sort(got); !slices.Equal(got, tt.wantSeen) { + t.Fatalf("seen = %q, want %q", got, tt.wantSeen) + } + }) + } +} + +func TestBuildRenderInput(t *testing.T) { + t.Parallel() + + tests := []struct { + name string + baseConfig string + args []string + wantErr string + wantBase string + }{ + {name: "bootstrap data only", args: msiArgs}, + { + name: "base config is compacted", + baseConfig: "{\n \"agent\": {\"logLevel\": \"debug\"}\n}\n", + args: []string{"--agent-url", "https://example.com/a.tar.gz"}, + wantBase: `{"agent":{"logLevel":"debug"}}`, + }, + { + name: "an agent source is required", + args: []string{"--auth", "msi", "--fetch-bootstrap-data"}, + wantErr: "--agent-url or --agent-version", + }, + { + name: "without a base config bootstrap data must be fetched", + args: []string{"--auth", "msi", "--agent-version", "v0.1.0"}, + wantErr: "needs --fetch-bootstrap-data", + }, + {name: "base config must be an object", baseConfig: `["a"]`, args: msiArgs, wantErr: "must be a JSON object"}, + {name: "base config must not be null", baseConfig: `null`, args: msiArgs, wantErr: "must be a JSON object"}, + {name: "base config must be JSON", baseConfig: `{"a":`, args: msiArgs, wantErr: "must be a JSON object"}, + {name: "bad pass-through fails before anything is read", baseConfig: `{"a":`, args: []string{"--nope"}, wantErr: "unknown bootstrap.sh option"}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + + var opts options + if tt.baseConfig != "" { + opts.baseConfigPath = filepath.Join(t.TempDir(), "base.json") + if err := os.WriteFile(opts.baseConfigPath, []byte(tt.baseConfig), 0o600); err != nil { + t.Fatal(err) + } + } + + in, err := buildRenderInput(opts, tt.args) + if tt.wantErr != "" { + if err == nil || !strings.Contains(err.Error(), tt.wantErr) { + t.Fatalf("buildRenderInput() error = %v, want %q", err, tt.wantErr) + } + return + } + if err != nil { + t.Fatalf("buildRenderInput() error = %v", err) + } + if string(in.baseConfig) != tt.wantBase { + t.Errorf("baseConfig = %q, want %q", in.baseConfig, tt.wantBase) + } + if !slices.Equal(in.bootstrapArgs, tt.args) { + t.Errorf("bootstrapArgs = %q, want %q", in.bootstrapArgs, tt.args) + } + }) + } +} + +func TestLoadCredential(t *testing.T) { + t.Parallel() + + tests := []struct { + name string + secret bool + fileName string + write func(t *testing.T, path string) + mode os.FileMode + wantErr string + wantFlag string + wantHostPath string + }{ + { + name: "client secret", + secret: true, + fileName: "secret", + write: writeFile("s3cret\n"), + mode: 0o600, + wantFlag: "--sp-client-secret-file", + wantHostPath: "/etc/aks-flex-node/credentials/sp-client-secret", + }, + {name: "readable secret is refused", secret: true, fileName: "secret", write: writeFile("s3cret"), mode: 0o644, wantErr: "group or other"}, + {name: "empty secret is refused", secret: true, fileName: "secret", write: writeFile(" \n"), mode: 0o600, wantErr: "empty"}, + { + name: "client certificate", + fileName: "client.pem", + write: writeTestClientCertificate, + mode: 0o600, + wantFlag: "--sp-client-certificate-file", + wantHostPath: "/etc/aks-flex-node/credentials/sp-client-certificate", + }, + { + name: "pfx suffix is kept because PKCS#12 is recognized by it", + fileName: "client.PFX", + write: writeTestClientCertificate, + mode: 0o600, + wantFlag: "--sp-client-certificate-file", + wantHostPath: "/etc/aks-flex-node/credentials/sp-client-certificate.pfx", + }, + {name: "invalid certificate is refused", fileName: "client.pem", write: writeFile("not a certificate"), mode: 0o600, wantErr: "certificate"}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + + localPath := filepath.Join(t.TempDir(), tt.fileName) + tt.write(t, localPath) + if err := os.Chmod(localPath, tt.mode); err != nil { + t.Fatal(err) + } + var opts options + if tt.secret { + opts.spClientSecretFile = localPath + } else { + opts.spClientCertificateFile = localPath + } + + cred, err := loadCredential(opts) + if tt.wantErr != "" { + if err == nil || !strings.Contains(err.Error(), tt.wantErr) { + t.Fatalf("loadCredential() error = %v, want %q", err, tt.wantErr) + } + return + } + if err != nil { + t.Fatalf("loadCredential() error = %v", err) + } + want, err := os.ReadFile(localPath) + if err != nil { + t.Fatal(err) + } + if cred.flag != tt.wantFlag || cred.hostPath != tt.wantHostPath || !bytes.Equal(cred.content, want) { + t.Fatalf("credential = {%s %s %d bytes}, want {%s %s %d bytes}", + cred.flag, cred.hostPath, len(cred.content), tt.wantFlag, tt.wantHostPath, len(want)) + } + }) + } +} + +func TestLoadCredentialWithoutOne(t *testing.T) { + t.Parallel() + + cred, err := loadCredential(options{}) + if err != nil || cred != nil { + t.Fatalf("loadCredential() = %v, %v, want nil, nil", cred, err) + } +} + +func TestRender(t *testing.T) { + t.Parallel() + + secret := &credential{ + flag: "--sp-client-secret-file", + hostPath: "/etc/aks-flex-node/credentials/sp-client-secret", + content: []byte("s3cret\n"), + } + + tests := []struct { + name string + in renderInput + wantScript string + }{ + { + name: "without a base config the script is written as is", + in: renderInput{bootstrapArgs: msiArgs}, + wantScript: scripts.Bootstrap, + }, + { + name: "base config and credential", + in: renderInput{ + baseConfig: []byte(`{"azure":{"bootstrapToken":{"token":"abcdef.0123456789abcdef"}}}`), + credential: secret, + bootstrapArgs: []string{"--auth", "service-principal", "--agent-version", "v0.1.0"}, + }, + wantScript: strings.Replace(scripts.Bootstrap, baseConfigPlaceholder, + `{"azure":{"bootstrapToken":{"token":"abcdef.0123456789abcdef"}}}`, 1), + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + + out, err := render(tt.in) + if err != nil { + t.Fatalf("render() error = %v", err) + } + var cfg ignitionConfig + if err := json.Unmarshal(out, &cfg); err != nil { + t.Fatalf("output is not JSON: %v", err) + } + if cfg.Ignition.Version != "3.4.0" { + t.Errorf("version = %q, want 3.4.0", cfg.Ignition.Version) + } + + wantDirs := []directory{{Path: "/etc/aks-flex-node/first-boot", Mode: 0o700}} + if tt.in.credential != nil { + wantDirs = append(wantDirs, directory{Path: "/etc/aks-flex-node/credentials", Mode: 0o700}) + } + if !slices.Equal(cfg.Storage.Directories, wantDirs) { + t.Errorf("directories = %+v, want %+v", cfg.Storage.Directories, wantDirs) + } + + wantFiles := 1 + if tt.in.credential != nil { + wantFiles = 2 + } + if len(cfg.Storage.Files) != wantFiles { + t.Fatalf("got %d files, want %d", len(cfg.Storage.Files), wantFiles) + } + script := cfg.Storage.Files[0] + if script.Path != "/etc/aks-flex-node/first-boot/bootstrap.sh" || script.Mode != 0o700 || + script.Contents.Compression != "gzip" || script.Overwrite == nil || !*script.Overwrite { + t.Errorf("script entry = %+v", script) + } + if got := decodeSource(t, script.Contents.Source, true); got != tt.wantScript { + t.Errorf("script content differs from the embedded script with the base config in place") + } + + wantArgs := append([]string(nil), tt.in.bootstrapArgs...) + if tt.in.credential != nil { + cred := cfg.Storage.Files[1] + if cred.Path != tt.in.credential.hostPath || cred.Mode != 0o600 || cred.Contents.Compression != "" { + t.Errorf("credential entry = %+v", cred) + } + if got := decodeSource(t, cred.Contents.Source, false); got != string(tt.in.credential.content) { + t.Errorf("credential content = %q, want %q", got, tt.in.credential.content) + } + wantArgs = append(wantArgs, tt.in.credential.flag, tt.in.credential.hostPath) + } + + if len(cfg.Systemd.Units) != 1 { + t.Fatalf("got %d units, want 1", len(cfg.Systemd.Units)) + } + u := cfg.Systemd.Units[0] + if u.Name != daemon.FirstBootUnitName || u.Enabled == nil || !*u.Enabled { + t.Errorf("unit = %s enabled=%v, want %s enabled", u.Name, u.Enabled, daemon.FirstBootUnitName) + } + if u.Contents != firstBootUnit(wantArgs) { + t.Errorf("unit contents do not run bootstrap.sh with %q:\n%s", wantArgs, u.Contents) + } + }) + } +} + +// TestRenderedModesAreDecimal pins the encoding: Ignition takes modes as JSON integers. +func TestRenderedModesAreDecimal(t *testing.T) { + t.Parallel() + + out, err := render(renderInput{bootstrapArgs: msiArgs}) + if err != nil { + t.Fatal(err) + } + for _, want := range []string{`"mode": 448`, `"version": "3.4.0"`} { + if !bytes.Contains(out, []byte(want)) { + t.Errorf("output does not contain %s", want) + } + } +} + +func TestFirstBootUnit(t *testing.T) { + t.Parallel() + + want := `[Unit] +Description=Bootstrap AKS Flex Node +Wants=network-online.target +After=network-online.target nss-lookup.target +ConditionPathExists=!/etc/systemd/system/aks-flex-node-agent.service +AssertPathExists=/etc/aks-flex-node/first-boot/bootstrap.sh +StartLimitIntervalSec=0 + +[Service] +Type=oneshot +RemainAfterExit=yes +Restart=on-failure +RestartSec=10s +RestartSteps=10 +RestartMaxDelaySec=300 +ExecStart=/bin/bash /etc/aks-flex-node/first-boot/bootstrap.sh "--auth" "msi" "--config-overrides" "{\"a\":\"$$HOME %%h\"}" +ExecStartPost=/bin/rm -f /etc/aks-flex-node/first-boot/bootstrap.sh + +[Install] +WantedBy=multi-user.target +` + got := firstBootUnit([]string{"--auth", "msi", "--config-overrides", `{"a":"$HOME %h"}`}) + if got != want { + t.Fatalf("firstBootUnit() =\n%s\nwant\n%s", got, want) + } +} + +func TestSystemdQuote(t *testing.T) { + t.Parallel() + + tests := []struct { + name, arg, want string + }{ + {name: "plain", arg: "msi", want: `"msi"`}, + {name: "spaces stay in one argument", arg: "a b", want: `"a b"`}, + {name: "double quote", arg: `say "hi"`, want: `"say \"hi\""`}, + {name: "backslash", arg: `a\b`, want: `"a\\b"`}, + {name: "variable is not expanded", arg: "$HOME${X}", want: `"$$HOME$${X}"`}, + {name: "specifier is not expanded", arg: "%h%%", want: `"%%h%%%%"`}, + {name: "single quote needs nothing", arg: "it's", want: `"it's"`}, + {name: "empty", arg: "", want: `""`}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + + if got := systemdQuote(tt.arg); got != tt.want { + t.Fatalf("systemdQuote(%q) = %s, want %s", tt.arg, got, tt.want) + } + }) + } +} + +func TestPopulateBootstrapScript(t *testing.T) { + t.Parallel() + + tests := []struct { + name, script, base, want, wantErr string + }{ + {name: "replaces the placeholder", script: "a\n" + baseConfigPlaceholder + "\nb\n", base: `{"x":1}`, want: "a\n{\"x\":1}\nb\n"}, + {name: "no base config keeps the placeholder", script: "a\n" + baseConfigPlaceholder + "\n", want: "a\n" + baseConfigPlaceholder + "\n"}, + {name: "missing placeholder", script: "a\n", base: `{}`, wantErr: "0 base config placeholders"}, + {name: "two placeholders", script: baseConfigPlaceholder + baseConfigPlaceholder, base: `{}`, wantErr: "2 base config placeholders"}, + {name: "multi-line base config", script: baseConfigPlaceholder, base: "{\n}", wantErr: "compact"}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + + got, err := populateBootstrapScript(tt.script, []byte(tt.base)) + if tt.wantErr != "" { + if err == nil || !strings.Contains(err.Error(), tt.wantErr) { + t.Fatalf("populateBootstrapScript() error = %v, want %q", err, tt.wantErr) + } + return + } + if err != nil || got != tt.want { + t.Fatalf("populateBootstrapScript() = %q, %v, want %q", got, err, tt.want) + } + }) + } +} + +// TestPopulatedScriptReturnsTheBaseConfig runs the populated script's own reader, so the +// substitution is checked against how bash parses the heredoc rather than against a string. +func TestPopulatedScriptReturnsTheBaseConfig(t *testing.T) { + t.Parallel() + + bash, err := exec.LookPath("bash") + if err != nil { + t.Skip("bash is not available") + } + + tests := []struct { + name string + base string + }{ + {name: "plain", base: `{"agent":{"logLevel":"info"}}`}, + {name: "shell syntax stays literal", base: `{"a":"$HOME $(id) ` + "`id`" + ` 'q' \\ \"d\""}`}, + {name: "heredoc delimiter inside a value", base: `{"a":"AKS_FLEX_NODE_EMBEDDED_CONFIG"}`}, + {name: "unicode", base: `{"a":"caf\u00e9 ` + "\u00e9" + `"}`}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + + var compact bytes.Buffer + if err := json.Compact(&compact, []byte(tt.base)); err != nil { + t.Fatalf("test base config is not JSON: %v", err) + } + script, err := populateBootstrapScript(scripts.Bootstrap, compact.Bytes()) + if err != nil { + t.Fatal(err) + } + scriptPath := filepath.Join(t.TempDir(), "bootstrap.sh") + if err := os.WriteFile(scriptPath, []byte(script), 0o600); err != nil { + t.Fatal(err) + } + + // Sourcing defines the functions without running main. + out, err := exec.CommandContext(t.Context(), bash, "-c", `source "$1" && write_embedded_base_config`, "bash", scriptPath).Output() + if err != nil { + t.Fatalf("reading the base config back failed: %v", err) + } + if got := strings.TrimSuffix(string(out), "\n"); got != compact.String() { + t.Fatalf("base config read back = %s, want %s", got, compact.String()) + } + }) + } +} + +func TestWriteOutput(t *testing.T) { + t.Parallel() + + t.Run("stdout", func(t *testing.T) { + t.Parallel() + + var stdout bytes.Buffer + if err := writeOutput(&stdout, "-", []byte("config")); err != nil || stdout.String() != "config" { + t.Fatalf("writeOutput() = %v, stdout %q", err, stdout.String()) + } + }) + + t.Run("new file is private", func(t *testing.T) { + t.Parallel() + + outputPath := filepath.Join(t.TempDir(), "node.ign") + if err := writeOutput(io.Discard, outputPath, []byte("config")); err != nil { + t.Fatal(err) + } + info, err := os.Stat(outputPath) + if err != nil { + t.Fatal(err) + } + if info.Mode().Perm() != 0o600 { + t.Fatalf("mode = %o, want 600", info.Mode().Perm()) + } + }) + + t.Run("existing file is not reused", func(t *testing.T) { + t.Parallel() + + outputPath := filepath.Join(t.TempDir(), "node.ign") + if err := os.WriteFile(outputPath, []byte("old"), 0o644); err != nil { + t.Fatal(err) + } + if err := writeOutput(io.Discard, outputPath, []byte("config")); err == nil { + t.Fatal("writeOutput() reused an existing file") + } + if got, _ := os.ReadFile(outputPath); string(got) != "old" { + t.Fatalf("existing file was changed to %q", got) + } + }) +} + +func TestCommand(t *testing.T) { + t.Parallel() + + tests := []struct { + name string + args []string + wantErr string + }{ + {name: "bootstrap.sh options after --", args: append([]string{"--"}, msiArgs...)}, + {name: "bootstrap.sh options before --", args: append([]string{"--auth"}, msiArgs...), wantErr: "unknown flag"}, + {name: "positional arguments before --", args: append([]string{"msi", "--"}, msiArgs...), wantErr: "after --"}, + {name: "no bootstrap.sh options", args: nil, wantErr: "--agent-url or --agent-version"}, + { + name: "one credential at a time", + args: append([]string{"--sp-client-secret-file", "a", "--sp-client-certificate-file", "b", "--"}, msiArgs...), + wantErr: "none of the others can be", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + + cmd := NewCommand() + var stdout bytes.Buffer + cmd.SetOut(&stdout) + cmd.SetErr(io.Discard) + cmd.SetArgs(tt.args) + + err := cmd.ExecuteContext(t.Context()) + if tt.wantErr != "" { + if err == nil || !strings.Contains(err.Error(), tt.wantErr) { + t.Fatalf("Execute() error = %v, want %q", err, tt.wantErr) + } + return + } + if err != nil { + t.Fatalf("Execute() error = %v", err) + } + var cfg ignitionConfig + if err := json.Unmarshal(stdout.Bytes(), &cfg); err != nil || len(cfg.Systemd.Units) != 1 { + t.Fatalf("stdout is not the rendered config: %v", err) + } + }) + } +} + +func decodeSource(t *testing.T, source string, gzipped bool) string { + t.Helper() + + encoded, ok := strings.CutPrefix(source, "data:;base64,") + if !ok { + t.Fatalf("source %.40q is not a base64 data URL", source) + } + raw, err := base64.StdEncoding.DecodeString(encoded) + if err != nil { + t.Fatalf("decode data URL: %v", err) + } + if !gzipped { + return string(raw) + } + zr, err := gzip.NewReader(bytes.NewReader(raw)) + if err != nil { + t.Fatalf("gunzip: %v", err) + } + out, err := io.ReadAll(zr) + if err != nil { + t.Fatalf("gunzip: %v", err) + } + + return string(out) +} + +func writeFile(content string) func(t *testing.T, path string) { + return func(t *testing.T, path string) { + t.Helper() + + if err := os.WriteFile(path, []byte(content), 0o600); err != nil { + t.Fatal(err) + } + } +} + +func writeTestClientCertificate(t *testing.T, path string) { + t.Helper() + + privateKey, err := rsa.GenerateKey(rand.Reader, 2048) + if err != nil { + t.Fatalf("rsa.GenerateKey: %v", err) + } + template := &x509.Certificate{ + SerialNumber: big.NewInt(1), + Subject: pkix.Name{CommonName: "test-client"}, + NotBefore: time.Now().Add(-time.Hour), + NotAfter: time.Now().Add(time.Hour), + KeyUsage: x509.KeyUsageDigitalSignature, + } + certificate, err := x509.CreateCertificate(rand.Reader, template, template, &privateKey.PublicKey, privateKey) + if err != nil { + t.Fatalf("x509.CreateCertificate: %v", err) + } + privateKeyData, err := x509.MarshalPKCS8PrivateKey(privateKey) + if err != nil { + t.Fatalf("x509.MarshalPKCS8PrivateKey: %v", err) + } + data := append( + pem.EncodeToMemory(&pem.Block{Type: "CERTIFICATE", Bytes: certificate}), + pem.EncodeToMemory(&pem.Block{Type: "PRIVATE KEY", Bytes: privateKeyData})..., + ) + if err := os.WriteFile(path, data, 0o600); err != nil { + t.Fatalf("os.WriteFile: %v", err) + } +} diff --git a/pkg/cmd/ignition/render.go b/pkg/cmd/ignition/render.go new file mode 100644 index 00000000..55deac85 --- /dev/null +++ b/pkg/cmd/ignition/render.go @@ -0,0 +1,245 @@ +package ignition + +import ( + "bytes" + "compress/gzip" + "encoding/base64" + "encoding/json" + "fmt" + "path" + "strings" + + "k8s.io/utils/ptr" + + "github.com/Azure/AKSFlexNode/pkg/config" + "github.com/Azure/AKSFlexNode/pkg/daemon" + "github.com/Azure/AKSFlexNode/scripts" +) + +// The Ignition types are written out rather than taken from github.com/coreos/ignition, which +// would bring in the whole specification for the few fields used here. The version is pinned and +// asserted by tests. +const specVersion = "3.4.0" + +const ( + modeSecretDir = 0o700 + modeSecretFile = 0o600 + modeScript = 0o700 + + // baseConfigPlaceholder is the line in bootstrap.sh that stands for the base config. + baseConfigPlaceholder = "__AKS_FLEX_NODE_BASE_CONFIG_JSON__" +) + +var ( + // Both directories are under the config directory, so reset removes them with it. + firstBootDir = path.Join(config.ConfigDir, "first-boot") + bootstrapPath = path.Join(firstBootDir, "bootstrap.sh") + credentialsDir = path.Join(config.ConfigDir, "credentials") +) + +type ignitionConfig struct { + Ignition ignitionVersion `json:"ignition"` + Storage storage `json:"storage"` + Systemd systemd `json:"systemd"` +} + +type ignitionVersion struct { + Version string `json:"version"` +} + +type storage struct { + Directories []directory `json:"directories,omitempty"` + Files []file `json:"files,omitempty"` +} + +type directory struct { + Path string `json:"path"` + Mode int `json:"mode"` +} + +type file struct { + Path string `json:"path"` + Mode int `json:"mode"` + Overwrite *bool `json:"overwrite,omitempty"` + Contents contents `json:"contents"` +} + +type contents struct { + Source string `json:"source"` + Compression string `json:"compression,omitempty"` +} + +type systemd struct { + Units []unit `json:"units"` +} + +type unit struct { + Name string `json:"name"` + Enabled *bool `json:"enabled,omitempty"` + Contents string `json:"contents"` +} + +// credential is a file written to the host for bootstrap.sh to use, and the bootstrap.sh flag +// that points at it. +type credential struct { + flag string + hostPath string + content []byte +} + +// renderInput is everything the Ignition config is built from, already validated. +type renderInput struct { + // baseConfig is compact JSON, or empty to leave bootstrap.sh without one. + baseConfig []byte + credential *credential + bootstrapArgs []string +} + +func render(in renderInput) ([]byte, error) { + script, err := populateBootstrapScript(scripts.Bootstrap, in.baseConfig) + if err != nil { + return nil, err + } + scriptSource, err := gzipDataURL([]byte(script)) + if err != nil { + return nil, err + } + + args := append([]string(nil), in.bootstrapArgs...) + cfg := ignitionConfig{ + Ignition: ignitionVersion{Version: specVersion}, + Storage: storage{ + Directories: []directory{{Path: firstBootDir, Mode: modeSecretDir}}, + Files: []file{{ + Path: bootstrapPath, + Mode: modeScript, + Overwrite: ptr.To(true), + Contents: contents{Source: scriptSource, Compression: "gzip"}, + }}, + }, + } + if in.credential != nil { + cfg.Storage.Directories = append(cfg.Storage.Directories, directory{Path: credentialsDir, Mode: modeSecretDir}) + cfg.Storage.Files = append(cfg.Storage.Files, file{ + Path: in.credential.hostPath, + Mode: modeSecretFile, + Overwrite: ptr.To(true), + Contents: contents{Source: dataURL(in.credential.content)}, + }) + args = append(args, in.credential.flag, in.credential.hostPath) + } + cfg.Systemd.Units = []unit{{ + Name: daemon.FirstBootUnitName, + Enabled: ptr.To(true), + Contents: firstBootUnit(args), + }} + + out, err := json.MarshalIndent(cfg, "", " ") + if err != nil { + return nil, fmt.Errorf("encode Ignition config: %w", err) + } + + return append(out, '\n'), nil +} + +// populateBootstrapScript puts the base config in place of the placeholder, the way a published +// script is populated. Without a base config the placeholder stays, and bootstrap.sh starts from +// an empty config and fetches bootstrap data. +func populateBootstrapScript(script string, baseConfig []byte) (string, error) { + if n := strings.Count(script, baseConfigPlaceholder); n != 1 { + return "", fmt.Errorf("bootstrap.sh has %d base config placeholders, want 1", n) + } + if len(baseConfig) == 0 { + return script, nil + } + // The placeholder sits in a quoted heredoc, so the JSON is taken literally. It must stay on + // one line so that it cannot end the heredoc early. + if bytes.ContainsAny(baseConfig, "\r\n") { + return "", fmt.Errorf("base config must be compact JSON") + } + + return strings.Replace(script, baseConfigPlaceholder, string(baseConfig), 1), nil +} + +// firstBootUnit returns the unit that runs bootstrap.sh until the agent is installed. +func firstBootUnit(args []string) string { + execStart := make([]string, 0, len(args)+2) + execStart = append(execStart, "/bin/bash", bootstrapPath) + for _, arg := range args { + execStart = append(execStart, systemdQuote(arg)) + } + + var b strings.Builder + b.WriteString("[Unit]\n") + b.WriteString("Description=Bootstrap AKS Flex Node\n") + b.WriteString("Wants=network-online.target\n") + b.WriteString("After=network-online.target nss-lookup.target\n") + // Once the agent unit exists the agent runs the node, and bootstrap.sh would fail its + // preflight check for an existing deployment. Skipping on that condition, rather than on a + // marker of our own, also means a failure after the agent is installed is not retried. + b.WriteString("ConditionPathExists=!" + daemon.ServiceUnitPath + "\n") + // An assertion rather than a condition: a missing script is a provisioning error and should + // fail the unit where it can be seen, not skip it. + b.WriteString("AssertPathExists=" + bootstrapPath + "\n") + // Retry for as long as it takes; the backoff below bounds the rate. + b.WriteString("StartLimitIntervalSec=0\n\n") + b.WriteString("[Service]\n") + b.WriteString("Type=oneshot\n") + b.WriteString("RemainAfterExit=yes\n") + // Early boot failures, such as DNS not answering yet when network-online.target is reached, + // are retried. The delay grows from 10s to 5min. systemd older than 254 ignores RestartSteps + // and RestartMaxDelaySec and keeps the fixed delay. + b.WriteString("Restart=on-failure\n") + b.WriteString("RestartSec=10s\n") + b.WriteString("RestartSteps=10\n") + b.WriteString("RestartMaxDelaySec=300\n") + b.WriteString("ExecStart=" + strings.Join(execStart, " ") + "\n") + // The script carries the base config, including any bootstrap token. The installed config + // has what the agent needs, so the copy goes once bootstrap succeeds. + b.WriteString("ExecStartPost=/bin/rm -f " + bootstrapPath + "\n\n") + b.WriteString("[Install]\n") + b.WriteString("WantedBy=multi-user.target\n") + + return b.String() +} + +// systemdQuote quotes one argument of an ExecStart line. systemd expands % specifiers and $ +// variables even inside double quotes, so both are doubled. Control characters are rejected by +// validation before this is reached. +func systemdQuote(arg string) string { + var b strings.Builder + b.WriteByte('"') + for _, r := range arg { + switch r { + case '\\', '"': + b.WriteByte('\\') + b.WriteRune(r) + case '%': + b.WriteString("%%") + case '$': + b.WriteString("$$") + default: + b.WriteRune(r) + } + } + b.WriteByte('"') + + return b.String() +} + +func dataURL(content []byte) string { + return "data:;base64," + base64.StdEncoding.EncodeToString(content) +} + +func gzipDataURL(content []byte) (string, error) { + var buf bytes.Buffer + zw := gzip.NewWriter(&buf) + if _, err := zw.Write(content); err != nil { + return "", fmt.Errorf("compress bootstrap.sh: %w", err) + } + if err := zw.Close(); err != nil { + return "", fmt.Errorf("compress bootstrap.sh: %w", err) + } + + return dataURL(buf.Bytes()), nil +} diff --git a/pkg/cmd/nspawnlifecycle/nspawn_lifecycle.go b/pkg/cmd/nspawnlifecycle/nspawn_lifecycle.go index 3335cf0c..387b7a0e 100644 --- a/pkg/cmd/nspawnlifecycle/nspawn_lifecycle.go +++ b/pkg/cmd/nspawnlifecycle/nspawn_lifecycle.go @@ -10,6 +10,7 @@ import ( "github.com/spf13/cobra" flexconfig "github.com/Azure/AKSFlexNode/pkg/config" + "github.com/Azure/AKSFlexNode/pkg/daemon" agentconfig "github.com/Azure/unbounded/pkg/agent/config" "github.com/Azure/unbounded/pkg/agent/goalstates" sharedlifecycle "github.com/Azure/unbounded/pkg/agent/nspawnlifecycle" @@ -73,6 +74,12 @@ func newPhaseCommand( return err } + // The hooks regenerate the units that invoke them, and name the + // helper under the host root, so it has to lead to the files. + if err := daemon.MigrateHostRoot(log); err != nil { + return err + } + lifecycle, err := factory(log) if err != nil { return fmt.Errorf("create nspawn lifecycle: %w", err) diff --git a/pkg/cmd/start/start.go b/pkg/cmd/start/start.go index fd3a44a4..7237d3f6 100644 --- a/pkg/cmd/start/start.go +++ b/pkg/cmd/start/start.go @@ -50,6 +50,9 @@ func NewCommand() *cobra.Command { } func runStart(ctx context.Context, cfg *config.Config, logger *slog.Logger) error { + if err := daemon.MigrateHostRoot(logger); err != nil { + return err + } goal, err := aksmachine.GoalStateFromConfig(cfg) if err != nil { return fmt.Errorf("build goal state from config: %w", err) @@ -81,6 +84,9 @@ func runStart(ctx context.Context, cfg *config.Config, logger *slog.Logger) erro return fmt.Errorf("bootstrap failed to resolve goal state: %w", err) } + if err := daemon.PrepareHostRoot(ctx, logger); err != nil { + return fmt.Errorf("bootstrap failed: %w", err) + } tasks := phases.Serial(logger, daemon.SetupHost(cfg, logger), daemon.StartNode(cfg, logger, machineName, gs, containerImageArchives, stateStore, state), diff --git a/pkg/config/adapter.go b/pkg/config/adapter.go index 08fcb2cc..d4c29dc9 100644 --- a/pkg/config/adapter.go +++ b/pkg/config/adapter.go @@ -5,10 +5,12 @@ import ( "fmt" "log/slog" "maps" + "path/filepath" "runtime" agentconfig "github.com/Azure/unbounded/pkg/agent/config" "github.com/Azure/unbounded/pkg/agent/goalstates" + "github.com/Azure/unbounded/pkg/agent/hostroot" clientcmdapi "k8s.io/client-go/tools/clientcmd/api" ) @@ -18,6 +20,10 @@ const ( // starting the kubelet so that exec credential plugins can invoke it. flexNodeBinaryPath = "/usr/local/bin/aks-flex-node" + // hostBinaryRelativePath is the aks-flex-node compatibility link on the + // host, relative to the host root; see HostBinaryPath. + hostBinaryRelativePath = "bin/aks-flex-node" + // aksAADServerID is the Azure AD server application ID for AKS. aksAADServerID = "6dae42f8-4368-4678-94ff-3960e28e3630" @@ -203,6 +209,13 @@ func ResolveMachineGoalState(ctx context.Context, log *slog.Logger, cfg *Config, return agentCfg, gs, containerImageArchives, nil } +// HostBinaryPath returns the aks-flex-node compatibility link on the host, +// under the resolved host root. Clients that run on the host, rather than the +// kubelet inside the machine, invoke the exec credential plugin through it. +func HostBinaryPath() string { + return filepath.Join(hostroot.Resolve(), hostBinaryRelativePath) +} + // buildExecCredential creates an ExecConfig that invokes the aks-flex-node // binary as a credential plugin. The binary's `token kubelogin` subcommand // uses kubelogin to obtain an Azure AD token for the AKS API server. diff --git a/pkg/daemon/agent_upgrade.go b/pkg/daemon/agent_upgrade.go index a18d7617..cb35fe57 100644 --- a/pkg/daemon/agent_upgrade.go +++ b/pkg/daemon/agent_upgrade.go @@ -23,6 +23,7 @@ import ( "github.com/Azure/unbounded/pkg/agent/agentbinary" agentdaemon "github.com/Azure/unbounded/pkg/agent/daemon" "github.com/Azure/unbounded/pkg/agent/goalstates" + "github.com/Azure/unbounded/pkg/agent/hostroot" ) const ( @@ -32,16 +33,11 @@ const ( var errAgentUpgradeAlreadyPending = errors.New("AgentUpgrade operation is already pending") +// defaultAgentUpgradePaths returns the host-side agent upgrade layout under the +// resolved host root. Callers that change the host migrate it first; see +// MigrateHostRoot. func defaultAgentUpgradePaths() agentUpgradePaths { - const binaryDir = "/usr/local/lib/aks-flex-node" - return agentUpgradePaths{ - BinaryPath: "/usr/local/bin/aks-flex-node", - BluePath: filepath.Join(binaryDir, "aks-flex-node-blue"), - GreenPath: filepath.Join(binaryDir, "aks-flex-node-green"), - CurrentPath: filepath.Join(binaryDir, "aks-flex-node-current"), - LastGoodPath: filepath.Join(binaryDir, "aks-flex-node-last-good"), - SignalPath: "/etc/aks-flex-node/agent-upgrade-signal.json", - } + return agentUpgradePathsUnder(hostroot.Resolve()) } type agentUpgradeRequest struct { @@ -386,6 +382,9 @@ func synchronizeNspawnAgentBinary(sourcePath, machine string) error { // RecoverAgentUpgrade records failure and restores both host and active nspawn // binaries. It is invoked by the systemd recovery unit through last-good. func RecoverAgentUpgrade(ctx context.Context, message string) error { + if err := MigrateHostRoot(slog.Default()); err != nil { + return err + } paths := defaultAgentUpgradePaths() signals := agentUpgradeSignalStore{path: paths.SignalPath} if err := signals.recordFailure(message); err != nil { diff --git a/pkg/daemon/agent_upgrade_test.go b/pkg/daemon/agent_upgrade_test.go index 37a0f679..d5053ef6 100644 --- a/pkg/daemon/agent_upgrade_test.go +++ b/pkg/daemon/agent_upgrade_test.go @@ -499,3 +499,65 @@ func TestFilesHaveEqualSHA256(t *testing.T) { t.Fatalf("filesHaveEqualSHA256 = %v, %v", equal, err) } } + +// TestAgentUpgradePathsUnder covers the host layout used by agent upgrade. +// +// Under the legacy root it must reproduce the absolute paths released versions +// used. A migrated host resolves the host root to the legacy root, and the +// links and units those versions wrote name these paths. +func TestAgentUpgradePathsUnder(t *testing.T) { + t.Parallel() + + tests := []struct { + name string + root string + wantBinary string + wantBlue string + wantCurrent string + wantLastGood string + }{ + { + name: "host root", + root: "/opt/unbounded", + wantBinary: "/opt/unbounded/bin/aks-flex-node", + wantBlue: "/opt/unbounded/lib/aks-flex-node/aks-flex-node-blue", + wantCurrent: "/opt/unbounded/lib/aks-flex-node/aks-flex-node-current", + wantLastGood: "/opt/unbounded/lib/aks-flex-node/aks-flex-node-last-good", + }, + { + name: "legacy root keeps the released layout", + root: "/usr/local", + wantBinary: "/usr/local/bin/aks-flex-node", + wantBlue: "/usr/local/lib/aks-flex-node/aks-flex-node-blue", + wantCurrent: "/usr/local/lib/aks-flex-node/aks-flex-node-current", + wantLastGood: "/usr/local/lib/aks-flex-node/aks-flex-node-last-good", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + + paths := agentUpgradePathsUnder(tt.root) + if paths.BinaryPath != tt.wantBinary { + t.Errorf("BinaryPath = %q, want %q", paths.BinaryPath, tt.wantBinary) + } + if paths.BluePath != tt.wantBlue { + t.Errorf("BluePath = %q, want %q", paths.BluePath, tt.wantBlue) + } + if paths.CurrentPath != tt.wantCurrent { + t.Errorf("CurrentPath = %q, want %q", paths.CurrentPath, tt.wantCurrent) + } + if paths.LastGoodPath != tt.wantLastGood { + t.Errorf("LastGoodPath = %q, want %q", paths.LastGoodPath, tt.wantLastGood) + } + + // The signal must stay on a filesystem that is writable even when + // /usr is read-only, and survive a rollback to a binary that + // predates the host root, so it does not move with the root. + if paths.SignalPath != "/etc/aks-flex-node/agent-upgrade-signal.json" { + t.Errorf("SignalPath = %q, want it to stay under /etc", paths.SignalPath) + } + }) + } +} diff --git a/pkg/daemon/daemon.go b/pkg/daemon/daemon.go index 36f473fe..414ae7d6 100644 --- a/pkg/daemon/daemon.go +++ b/pkg/daemon/daemon.go @@ -38,6 +38,12 @@ func Run(ctx context.Context, cfg *config.Config, log *slog.Logger) error { // logs unless the process explicitly configures its global logger. ctrl.SetLogger(logr.FromSlogHandler(log.Handler())) + // After an AgentUpgrade from a release that predates the host root, this is + // the first time the new agent runs on the host. + if err := MigrateHostRoot(log); err != nil { + return err + } + // Existing direct-file installations may predate the recovery units. Keep // the binary layout and systemd rollback assets converged on every startup. if err := ensureAgentUpgradeServiceAssets(ctx, log, cfg); err != nil { diff --git a/pkg/daemon/daemon_test.go b/pkg/daemon/daemon_test.go index 54964bea..fb22214c 100644 --- a/pkg/daemon/daemon_test.go +++ b/pkg/daemon/daemon_test.go @@ -42,8 +42,14 @@ func TestBootstrapCredentialRESTConfigExecCredential(t *testing.T) { if restCfg.ExecProvider == nil { t.Fatalf("ExecProvider = nil, want exec credential") } - if restCfg.ExecProvider.Command != "/usr/local/bin/aks-flex-node" { - t.Fatalf("ExecProvider.Command = %q", restCfg.ExecProvider.Command) + // This client runs on the host, where the binary is under the host root. The + // kubelet's credential keeps naming the binary inside the machine, which is + // not a path a fresh host has. + if restCfg.ExecProvider.Command != config.HostBinaryPath() { + t.Fatalf("ExecProvider.Command = %q, want %q", restCfg.ExecProvider.Command, config.HostBinaryPath()) + } + if machine := config.ToAgentConfig(cfg, "kube1").Kubelet.Auth.ExecCredential.Command; machine == restCfg.ExecProvider.Command { + t.Fatalf("host and machine credentials both run %q", machine) } } diff --git a/pkg/daemon/host_agent_activation.go b/pkg/daemon/host_agent_activation.go index 3a29869d..2cefc131 100644 --- a/pkg/daemon/host_agent_activation.go +++ b/pkg/daemon/host_agent_activation.go @@ -14,6 +14,7 @@ import ( "github.com/Azure/AKSFlexNode/pkg/utils/utilexec" "github.com/Azure/unbounded/pkg/agent/agentbinary" + "github.com/Azure/unbounded/pkg/agent/hostroot" ) const ( @@ -24,8 +25,11 @@ const ( // PreflightHostAgentActivation validates a directly staged Flex agent binary // and returns the shared activation plan without changing host state. +// +// It does not migrate the host root, so it plans against the root the host +// will resolve once migrated. func PreflightHostAgentActivation(ctx context.Context, log *slog.Logger, candidatePath string) (agentbinary.ActivationPlan, error) { - service, paths, err := newFlexDaemonActivationService(log) + service, paths, err := newFlexDaemonActivationService(log, PlannedHostRoot()) if err != nil { return agentbinary.ActivationPlan{}, err } @@ -38,7 +42,10 @@ func ActivateHostAgent(ctx context.Context, log *slog.Logger, candidatePath stri if os.Geteuid() != 0 { return agentbinary.ActivationResult{}, fmt.Errorf("host agent upgrade requires root privileges") } - service, paths, err := newFlexDaemonActivationService(log) + if err := MigrateHostRoot(log); err != nil { + return agentbinary.ActivationResult{}, err + } + service, paths, err := newFlexDaemonActivationService(log, hostroot.Resolve()) if err != nil { return agentbinary.ActivationResult{}, err } @@ -65,7 +72,7 @@ type flexDaemonActivationService struct { serviceWasActive bool } -func newFlexDaemonActivationService(log *slog.Logger) (*flexDaemonActivationService, agentUpgradePaths, error) { +func newFlexDaemonActivationService(log *slog.Logger, root string) (*flexDaemonActivationService, agentUpgradePaths, error) { if log == nil { log = slog.Default() } @@ -73,7 +80,7 @@ func newFlexDaemonActivationService(log *slog.Logger) (*flexDaemonActivationServ if err != nil { return nil, agentUpgradePaths{}, err } - paths := defaultAgentUpgradePaths() + paths := agentUpgradePathsUnder(root) serviceOptions, err := installedAgentServiceOptions(systemdSystemDir) if err != nil { return nil, agentUpgradePaths{}, err @@ -83,7 +90,7 @@ func newFlexDaemonActivationService(log *slog.Logger) (*flexDaemonActivationServ paths: paths, state: state, systemdDir: systemdSystemDir, - recoveryScript: recoveryScriptPath, + recoveryScript: recoveryScriptPathUnder(root), serviceOptions: serviceOptions, inspectService: inspectAgentServiceActive, }, paths, nil diff --git a/pkg/daemon/hostroot.go b/pkg/daemon/hostroot.go new file mode 100644 index 00000000..0c1c6482 --- /dev/null +++ b/pkg/daemon/hostroot.go @@ -0,0 +1,78 @@ +package daemon + +import ( + "context" + "log/slog" + "path/filepath" + + "github.com/Azure/unbounded/pkg/agent/hostroot" +) + +// AKS Flex Node keeps its own host-side files under the host root, /opt/unbounded +// resolved through symlinks; see the hostroot package. Releases before the host +// root installed them under /usr/local. On a host installed by one of those, the +// first command that changes the host links /opt/unbounded to /usr/local, and +// every path is built from the resolved root, so the paths an older release +// wrote into links and units still compare equal to the ones built here. +const ( + binaryName = "aks-flex-node" + managedBinaryDir = "lib/aks-flex-node" + recoveryScriptName = "aks-flex-node-recovery.sh" +) + +// HostRootMarkers returns the files, relative to the host root, whose presence +// under the legacy root identifies an installation by a release before the host +// root. They are the binary layout only: files the agent library installs, such +// as the nspawn lifecycle helper, can be left behind by an older reset. +func HostRootMarkers() []string { + return []string{ + filepath.Join("bin", binaryName), + filepath.Join(managedBinaryDir, "aks-flex-node-blue"), + filepath.Join(managedBinaryDir, "aks-flex-node-green"), + filepath.Join(managedBinaryDir, "aks-flex-node-current"), + filepath.Join(managedBinaryDir, "aks-flex-node-last-good"), + } +} + +// MigrateHostRoot links the host root to the legacy root on a host installed by +// a release before the host root. Commands that change the host call it before +// they resolve any path. +func MigrateHostRoot(log *slog.Logger) error { + return hostroot.Migrate(log, HostRootMarkers()...) +} + +// PlannedHostRoot returns the host root this host will resolve once migrated, +// without migrating it. It is for code that must not change the host. +func PlannedHostRoot() string { + return hostroot.Planned(HostRootMarkers()...) +} + +// PrepareHostRoot creates the directories under the host root that the node +// installs into. +func PrepareHostRoot(ctx context.Context, log *slog.Logger) error { + return hostroot.Prepare(ctx, log, "bin", managedBinaryDir, "libexec") +} + +// agentUpgradePathsUnder builds the upgrade layout under a resolved host root. +// +// SignalPath is under /etc rather than the root. It is state about an upgrade +// rather than part of the installed layout, and has to survive a rollback to a +// binary that predates the host root. +func agentUpgradePathsUnder(root string) agentUpgradePaths { + binaryDir := filepath.Join(root, managedBinaryDir) + + return agentUpgradePaths{ + BinaryPath: filepath.Join(root, "bin", binaryName), + BluePath: filepath.Join(binaryDir, "aks-flex-node-blue"), + GreenPath: filepath.Join(binaryDir, "aks-flex-node-green"), + CurrentPath: filepath.Join(binaryDir, "aks-flex-node-current"), + LastGoodPath: filepath.Join(binaryDir, "aks-flex-node-last-good"), + SignalPath: "/etc/aks-flex-node/agent-upgrade-signal.json", + } +} + +// recoveryScriptPathUnder returns where the recovery script is installed under a +// resolved host root. +func recoveryScriptPathUnder(root string) string { + return filepath.Join(root, managedBinaryDir, recoveryScriptName) +} diff --git a/pkg/daemon/hostroot_test.go b/pkg/daemon/hostroot_test.go new file mode 100644 index 00000000..84d7bd6b --- /dev/null +++ b/pkg/daemon/hostroot_test.go @@ -0,0 +1,50 @@ +package daemon + +import ( + "path/filepath" + "slices" + "testing" + + "github.com/Azure/unbounded/pkg/agent/hostroot" +) + +// TestHostRootMarkersAreTheReleasedBinaryLayout ties the markers to the layout +// released versions installed under the legacy root. A marker that does not +// match it would leave those hosts unmigrated, and the new agent would install a +// second layout beside the one systemd runs. +func TestHostRootMarkersAreTheReleasedBinaryLayout(t *testing.T) { + t.Parallel() + + legacy := agentUpgradePathsUnder(hostroot.LegacyPath) + markers := HostRootMarkers() + + tests := []struct { + name string + path string + marker bool + }{ + {name: "compatibility link", path: legacy.BinaryPath, marker: true}, + {name: "blue slot", path: legacy.BluePath, marker: true}, + {name: "green slot", path: legacy.GreenPath, marker: true}, + {name: "current link", path: legacy.CurrentPath, marker: true}, + {name: "last-good link", path: legacy.LastGoodPath, marker: true}, + // Not part of the binary layout, and an older reset can leave the + // library helpers behind. + {name: "recovery script", path: recoveryScriptPathUnder(hostroot.LegacyPath), marker: false}, + {name: "nspawn lifecycle helper", path: "/usr/local/bin/unbounded-agent-nspawn-lifecycle", marker: false}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + + rel, err := filepath.Rel(hostroot.LegacyPath, tt.path) + if err != nil { + t.Fatal(err) + } + if got := slices.Contains(markers, rel); got != tt.marker { + t.Fatalf("%s is a marker = %v, want %v", rel, got, tt.marker) + } + }) + } +} diff --git a/pkg/daemon/lifecycle.go b/pkg/daemon/lifecycle.go index 6862fd91..fd302a7c 100644 --- a/pkg/daemon/lifecycle.go +++ b/pkg/daemon/lifecycle.go @@ -14,15 +14,30 @@ import ( "github.com/Azure/AKSFlexNode/pkg/config" "github.com/Azure/AKSFlexNode/pkg/utils/utilexec" "github.com/Azure/AKSFlexNode/pkg/utils/utilio" + "github.com/Azure/unbounded/pkg/agent/hostroot" "github.com/Azure/unbounded/pkg/agent/phases" ) const ( ServiceUnitName = "aks-flex-node-agent.service" recoveryServiceUnitName = "aks-flex-node-agent-recovery.service" - recoveryScriptPath = "/usr/local/lib/aks-flex-node/aks-flex-node-recovery.sh" - systemdSystemDir = "/etc/systemd/system" - arcSystemdDependency = "himdsd.service" + // embeddedRecoveryScriptPath is the path the embedded recovery unit names, + // the one releases before the host root installed to. It is replaced with + // the path under the resolved host root when the unit is written. Keep it + // in sync with the embedded asset. + embeddedRecoveryScriptPath = "/usr/local/lib/aks-flex-node/aks-flex-node-recovery.sh" + // systemdSystemDir is not under the host root. Units must live where + // systemd looks for them, and /etc is writable even when /usr is not. + systemdSystemDir = "/etc/systemd/system" + arcSystemdDependency = "himdsd.service" + + // ServiceUnitPath is where the agent unit is installed. The first-boot unit + // only runs while it is absent. + ServiceUnitPath = systemdSystemDir + "/" + ServiceUnitName + // FirstBootUnitName is the oneshot unit that `aks-flex-node ignition` + // installs to run bootstrap.sh on first boot. It is shared with reset, + // which has to remove it. + FirstBootUnitName = "aks-flex-node-bootstrap.service" ) //go:embed assets/aks-flex-node-agent.service @@ -64,13 +79,15 @@ func (t *installServiceTask) Do(ctx context.Context) error { } func ensureAgentUpgradeServiceAssets(ctx context.Context, log *slog.Logger, cfg *config.Config) error { + root := hostroot.Resolve() + return ensureAgentUpgradeServiceAssetsAt( ctx, log, - defaultAgentUpgradePaths(), + agentUpgradePathsUnder(root), agentServiceOptionsFromConfig(cfg), systemdSystemDir, - recoveryScriptPath, + recoveryScriptPathUnder(root), utilexec.ReloadSystemd, ) } @@ -114,14 +131,8 @@ func desiredAgentServiceAssets(binaryPaths agentUpgradePaths, serviceOptions age if err != nil { return nil, err } - recoveryServiceContent := bytes.ReplaceAll(recoveryServiceUnitContent, []byte(recoveryScriptPath), []byte(recoveryScript)) - recoveryContent := recoveryScriptContent - for oldPath, newPath := range map[string]string{ - defaultAgentUpgradePaths().LastGoodPath: binaryPaths.LastGoodPath, - defaultAgentUpgradePaths().SignalPath: binaryPaths.SignalPath, - } { - recoveryContent = bytes.ReplaceAll(recoveryContent, []byte(oldPath), []byte(newPath)) - } + recoveryServiceContent := bytes.ReplaceAll(recoveryServiceUnitContent, []byte(embeddedRecoveryScriptPath), []byte(recoveryScript)) + recoveryContent := renderRecoveryScript(binaryPaths) // Publish dependencies before the main unit that references OnFailure, so an // interrupted update never leaves systemd pointing at missing recovery assets. return []agentServiceAsset{ @@ -164,6 +175,27 @@ func renderAgentServiceUnit(currentBinaryPath string, serviceOptions agentServic return content.Bytes(), nil } +// renderRecoveryScript points the embedded recovery script at the binary layout +// for this host. +// +// The embedded script contains the paths under the legacy root as literals, so +// those literals are what gets replaced. Left alone on a host installed under +// the host root, the script reads last-good from /usr/local, where there is no +// binary, so a failed upgrade cannot be rolled back. +func renderRecoveryScript(binaryPaths agentUpgradePaths) []byte { + embedded := agentUpgradePathsUnder(hostroot.LegacyPath) + content := recoveryScriptContent + + for oldPath, newPath := range map[string]string{ + embedded.LastGoodPath: binaryPaths.LastGoodPath, + embedded.SignalPath: binaryPaths.SignalPath, + } { + content = bytes.ReplaceAll(content, []byte(oldPath), []byte(newPath)) + } + + return content +} + func writeAgentServiceAssets(binaryPaths agentUpgradePaths, serviceOptions agentServiceOptions, systemdDir, recoveryScript, currentBinaryPath string) error { assets, err := desiredAgentServiceAssets(binaryPaths, serviceOptions, systemdDir, recoveryScript, currentBinaryPath) if err != nil { @@ -181,14 +213,61 @@ type uninstallServiceTask struct { log *slog.Logger } -// UninstallService returns a task that stops, disables, removes, and reloads the systemd unit. +// UninstallService returns a task that stops, disables, removes, and reloads +// the systemd unit. func UninstallService(log *slog.Logger) phases.Task { return &uninstallServiceTask{log: log} } +// uninstallPaths returns the files UninstallService removes. The recovery +// script is removed under the resolved host root and under the legacy root, so +// that uninstalling does not depend on the host root having been migrated. +func uninstallPaths() []string { + return uninstallPathsUnder(hostroot.Resolve(), hostroot.LegacyPath) +} + +func uninstallPathsUnder(root, legacy string) []string { + paths := []string{ + filepath.Join(systemdSystemDir, ServiceUnitName), + filepath.Join(systemdSystemDir, recoveryServiceUnitName), + recoveryScriptPathUnder(root), + } + if root != legacy { + paths = append(paths, recoveryScriptPathUnder(legacy)) + } + + return append(paths, agentUpgradePathsUnder(root).SignalPath) +} + +// removeIfPresent removes a file and treats its absence as success. +// +// It checks first. On a read-only filesystem, such as /usr on Azure Container +// Linux, unlinking a path that does not exist returns EROFS rather than ENOENT, +// so the sweep of the legacy root would otherwise fail reset on a file that was +// never there. Lstat so a dangling symlink still counts as present. +func removeIfPresent(path string) error { + return removeIfPresentWith(path, os.Lstat, os.Remove) +} + +func removeIfPresentWith(path string, lstat func(string) (os.FileInfo, error), remove func(string) error) error { + if _, err := lstat(path); errors.Is(err, os.ErrNotExist) { + return nil + } + if err := remove(path); err != nil && !errors.Is(err, os.ErrNotExist) { + return fmt.Errorf("remove %s: %w", path, err) + } + + return nil +} + func (t *uninstallServiceTask) Name() string { return "uninstall-service" } func (t *uninstallServiceTask) Do(ctx context.Context) error { + // Before the agent is stopped: in the daemon's reset paths, stopping the + // agent ends the process running this task. + if err := removeFirstBootUnit(ctx, t.log, systemdSystemDir, runSystemctl); err != nil { + return err + } if err := utilexec.StopService(ctx, t.log, ServiceUnitName); err != nil { t.log.Warn("failed to stop service (may not be running)", "unit", ServiceUnitName, "error", err) } @@ -196,14 +275,9 @@ func (t *uninstallServiceTask) Do(ctx context.Context) error { t.log.Warn("failed to disable service (may not be enabled)", "unit", ServiceUnitName, "error", err) } - for _, path := range []string{ - filepath.Join(systemdSystemDir, ServiceUnitName), - filepath.Join(systemdSystemDir, recoveryServiceUnitName), - recoveryScriptPath, - defaultAgentUpgradePaths().SignalPath, - } { - if err := os.Remove(path); err != nil && !os.IsNotExist(err) { - return fmt.Errorf("remove %s: %w", path, err) + for _, path := range uninstallPaths() { + if err := removeIfPresent(path); err != nil { + return err } } @@ -214,3 +288,32 @@ func (t *uninstallServiceTask) Do(ctx context.Context) error { t.log.Info("systemd service uninstalled", "unit", ServiceUnitName) return nil } + +func runSystemctl(ctx context.Context, log *slog.Logger, args ...string) error { + return utilexec.RunCmd(ctx, log, utilexec.Systemctl(), args...) +} + +// removeFirstBootUnit disables, stops, and removes the first-boot bootstrap +// unit, if the host was provisioned with one. +// +// --now matters. The unit is a oneshot with RemainAfterExit=yes, so without a +// stop it stays active after its file is gone, and starting it again after the +// host is provisioned anew does nothing. +func removeFirstBootUnit( + ctx context.Context, + log *slog.Logger, + unitDir string, + systemctl func(context.Context, *slog.Logger, ...string) error, +) error { + unitPath := filepath.Join(unitDir, FirstBootUnitName) + if _, err := os.Lstat(unitPath); errors.Is(err, os.ErrNotExist) { + return nil + } + + log.Info("removing first-boot bootstrap unit", "unit", FirstBootUnitName) + if err := systemctl(ctx, log, "disable", "--now", FirstBootUnitName); err != nil { + return fmt.Errorf("systemctl disable --now %s: %w", FirstBootUnitName, err) + } + + return removeIfPresent(unitPath) +} diff --git a/pkg/daemon/lifecycle_test.go b/pkg/daemon/lifecycle_test.go index b4a404fc..bfcbb2bc 100644 --- a/pkg/daemon/lifecycle_test.go +++ b/pkg/daemon/lifecycle_test.go @@ -3,11 +3,16 @@ package daemon import ( "bytes" "context" + "errors" "log/slog" "os" "path/filepath" + "slices" "strings" + "syscall" "testing" + + "github.com/Azure/unbounded/pkg/agent/hostroot" ) func TestEnsureAgentUpgradeServiceAssetsMigratesExistingInstallation(t *testing.T) { @@ -92,7 +97,7 @@ func TestRenderAgentServiceUnitGolden(t *testing.T) { for name, tt := range tests { t.Run(name, func(t *testing.T) { t.Parallel() - got, err := renderAgentServiceUnit(defaultAgentUpgradePaths().BinaryPath, tt.serviceOptions) + got, err := renderAgentServiceUnit(agentUpgradePathsUnder(hostroot.Path).BinaryPath, tt.serviceOptions) if err != nil { t.Fatalf("renderAgentServiceUnit() error = %v", err) } @@ -134,7 +139,7 @@ func TestInstalledAgentServiceOptions(t *testing.T) { t.Parallel() systemdDir := t.TempDir() if tt.writeUnit { - unit, err := renderAgentServiceUnit(defaultAgentUpgradePaths().BinaryPath, tt.serviceOptions) + unit, err := renderAgentServiceUnit(agentUpgradePathsUnder(hostroot.Path).BinaryPath, tt.serviceOptions) if err != nil { t.Fatalf("renderAgentServiceUnit() error = %v", err) } @@ -156,7 +161,7 @@ func TestInstalledAgentServiceOptions(t *testing.T) { func TestAgentServiceIncludesUpgradeRecovery(t *testing.T) { t.Parallel() - serviceContent, err := renderAgentServiceUnit(defaultAgentUpgradePaths().BinaryPath, agentServiceOptions{}) + serviceContent, err := renderAgentServiceUnit(agentUpgradePathsUnder(hostroot.Path).BinaryPath, agentServiceOptions{}) if err != nil { t.Fatalf("renderAgentServiceUnit() error = %v", err) } @@ -170,8 +175,8 @@ func TestAgentServiceIncludesUpgradeRecovery(t *testing.T) { if !strings.Contains(service, "OnFailure="+recoveryServiceUnitName) { t.Fatalf("service does not activate %s on failure", recoveryServiceUnitName) } - if !strings.Contains(string(recoveryServiceUnitContent), "ExecStart="+recoveryScriptPath) { - t.Fatalf("recovery service does not execute %s", recoveryScriptPath) + if !strings.Contains(string(recoveryServiceUnitContent), "ExecStart="+embeddedRecoveryScriptPath) { + t.Fatalf("recovery service does not execute %s", embeddedRecoveryScriptPath) } script := string(recoveryScriptContent) for _, expected := range []string{ @@ -188,3 +193,291 @@ func TestAgentServiceIncludesUpgradeRecovery(t *testing.T) { } } } + +// TestRecoveryScriptPathUnder covers the recovery script location. It lives +// beside the blue/green binaries under the host root, and under the legacy root +// it is the path baked into the embedded recovery unit. +func TestRecoveryScriptPathUnder(t *testing.T) { + t.Parallel() + + tests := []struct { + name string + root string + want string + }{ + { + name: "legacy root matches the embedded placeholder", + root: hostroot.LegacyPath, + want: embeddedRecoveryScriptPath, + }, + { + name: "host root", + root: "/opt/unbounded", + want: "/opt/unbounded/lib/aks-flex-node/aks-flex-node-recovery.sh", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + + if got := recoveryScriptPathUnder(tt.root); got != tt.want { + t.Errorf("recoveryScriptPathUnder(%q) = %q, want %q", tt.root, got, tt.want) + } + }) + } +} + +// TestRenderRecoveryScriptFollowsTheHostRoot covers the recovery script on a +// host installed under the host root. The script restores the last-good binary +// after a failed upgrade, so if it still names the /usr/local path, rollback +// fails there. +func TestRenderRecoveryScriptFollowsTheHostRoot(t *testing.T) { + t.Parallel() + + tests := []struct { + name string + root string + lastGood string + }{ + { + name: "legacy root is unchanged", + root: hostroot.LegacyPath, + lastGood: "/usr/local/lib/aks-flex-node/aks-flex-node-last-good", + }, + { + name: "host root", + root: "/opt/unbounded", + lastGood: "/opt/unbounded/lib/aks-flex-node/aks-flex-node-last-good", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + + script := string(renderRecoveryScript(agentUpgradePathsUnder(tt.root))) + + if !strings.Contains(script, "readlink -f "+tt.lastGood) { + t.Fatalf("recovery script does not read %s:\n%s", tt.lastGood, script) + } + if tt.root != hostroot.LegacyPath && strings.Contains(script, "/usr/local/") { + t.Fatalf("recovery script still names /usr/local:\n%s", script) + } + }) + } +} + +// TestEmbeddedRecoveryScriptNamesTheLegacyPaths pins the literals +// renderRecoveryScript replaces. If the asset and the legacy layout drift +// apart, the replacement silently matches nothing. +func TestEmbeddedRecoveryScriptNamesTheLegacyPaths(t *testing.T) { + t.Parallel() + + legacy := agentUpgradePathsUnder(hostroot.LegacyPath) + script := string(recoveryScriptContent) + + for _, path := range []string{legacy.LastGoodPath, legacy.SignalPath} { + if !strings.Contains(script, path) { + t.Fatalf("embedded recovery script does not contain %s", path) + } + } +} + +// TestUninstallPathsSweepBothRoots covers the files uninstall removes. The +// recovery script is removed under the legacy root too, so uninstalling does +// not depend on the host root having been migrated. +func TestUninstallPathsSweepBothRoots(t *testing.T) { + t.Parallel() + + tests := []struct { + name string + root string + want []string + }{ + { + name: "host root also sweeps the legacy root", + root: "/opt/unbounded", + want: []string{ + "/opt/unbounded/lib/aks-flex-node/aks-flex-node-recovery.sh", + "/usr/local/lib/aks-flex-node/aks-flex-node-recovery.sh", + }, + }, + { + name: "migrated host sweeps the legacy root once", + root: hostroot.LegacyPath, + want: []string{"/usr/local/lib/aks-flex-node/aks-flex-node-recovery.sh"}, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + + got := uninstallPathsUnder(tt.root, hostroot.LegacyPath) + want := append(tt.want, + filepath.Join(systemdSystemDir, ServiceUnitName), + filepath.Join(systemdSystemDir, recoveryServiceUnitName), + "/etc/aks-flex-node/agent-upgrade-signal.json", + ) + slices.Sort(got) + slices.Sort(want) + if !slices.Equal(got, want) { + t.Fatalf("uninstallPathsUnder(%q) = %v, want %v", tt.root, got, want) + } + }) + } +} + +// TestRemoveIfPresentSkipsAbsentFiles covers uninstall on a read-only /usr. +// Unlinking a missing file there returns EROFS rather than ENOENT, so the +// remove must not be attempted at all for an absent file. +func TestRemoveIfPresentSkipsAbsentFiles(t *testing.T) { + t.Parallel() + + tests := []struct { + name string + lstat func(string) (os.FileInfo, error) + wantErr bool + removed bool + }{ + { + name: "absent file is not removed", + lstat: func(string) (os.FileInfo, error) { return nil, os.ErrNotExist }, + wantErr: false, + removed: false, + }, + { + name: "present file that cannot be removed is an error", + lstat: func(string) (os.FileInfo, error) { return nil, nil }, + wantErr: true, + removed: true, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + + removed := false + err := removeIfPresentWith("/usr/local/lib/aks-flex-node/aks-flex-node-recovery.sh", tt.lstat, func(string) error { + removed = true + return syscall.EROFS + }) + + if (err != nil) != tt.wantErr { + t.Fatalf("removeIfPresentWith() error = %v, wantErr %v", err, tt.wantErr) + } + if removed != tt.removed { + t.Fatalf("remove called = %v, want %v", removed, tt.removed) + } + }) + } +} + +// TestRemoveIfPresentRemovesDanglingSymlink pins Lstat over Stat. +func TestRemoveIfPresentRemovesDanglingSymlink(t *testing.T) { + t.Parallel() + + dir := t.TempDir() + link := filepath.Join(dir, "link") + if err := os.Symlink(filepath.Join(dir, "missing"), link); err != nil { + t.Fatal(err) + } + + if err := removeIfPresent(link); err != nil { + t.Fatalf("removeIfPresent() error = %v", err) + } + if _, err := os.Lstat(link); !errors.Is(err, os.ErrNotExist) { + t.Fatalf("dangling symlink was not removed: %v", err) + } +} + +func TestRemoveFirstBootUnit(t *testing.T) { + t.Parallel() + + tests := []struct { + name string + setup func(t *testing.T, unitPath string) + systemctlErr error + wantErr bool + wantCalls []string + wantRemoved bool + }{ + { + name: "absent unit is left alone", + setup: func(*testing.T, string) {}, + wantCalls: nil, + wantRemoved: true, + }, + { + name: "installed unit is disabled, stopped, and removed", + setup: func(t *testing.T, unitPath string) { + if err := os.WriteFile(unitPath, []byte("[Unit]\n"), 0o644); err != nil { + t.Fatal(err) + } + }, + wantCalls: []string{"disable --now " + FirstBootUnitName}, + wantRemoved: true, + }, + { + name: "dangling unit link is still disabled and removed", + setup: func(t *testing.T, unitPath string) { + if err := os.Symlink(unitPath+".missing", unitPath); err != nil { + t.Fatal(err) + } + }, + wantCalls: []string{"disable --now " + FirstBootUnitName}, + wantRemoved: true, + }, + { + name: "unit that cannot be disabled is kept and reported", + setup: func(t *testing.T, unitPath string) { + if err := os.WriteFile(unitPath, []byte("[Unit]\n"), 0o644); err != nil { + t.Fatal(err) + } + }, + systemctlErr: errors.New("systemctl failed"), + wantErr: true, + wantCalls: []string{"disable --now " + FirstBootUnitName}, + wantRemoved: false, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + + dir := t.TempDir() + unitPath := filepath.Join(dir, FirstBootUnitName) + tt.setup(t, unitPath) + + var calls []string + err := removeFirstBootUnit(t.Context(), discardLogger(), dir, func(_ context.Context, _ *slog.Logger, args ...string) error { + calls = append(calls, strings.Join(args, " ")) + return tt.systemctlErr + }) + + if (err != nil) != tt.wantErr { + t.Fatalf("removeFirstBootUnit() error = %v, wantErr %v", err, tt.wantErr) + } + if !slices.Equal(calls, tt.wantCalls) { + t.Fatalf("systemctl calls = %q, want %q", calls, tt.wantCalls) + } + _, statErr := os.Lstat(unitPath) + if removed := errors.Is(statErr, os.ErrNotExist); removed != tt.wantRemoved { + t.Fatalf("unit removed = %v, want %v", removed, tt.wantRemoved) + } + }) + } +} + +// TestFirstBootUnitIsConditionedOnTheAgentUnit pins the path the first-boot unit checks, which +// has to be where the agent unit is actually written. +func TestFirstBootUnitIsConditionedOnTheAgentUnit(t *testing.T) { + t.Parallel() + + if want := filepath.Join(systemdSystemDir, ServiceUnitName); ServiceUnitPath != want { + t.Fatalf("ServiceUnitPath = %q, want %q", ServiceUnitPath, want) + } +} diff --git a/pkg/daemon/reset.go b/pkg/daemon/reset.go index 74883ae3..4eb78a0d 100644 --- a/pkg/daemon/reset.go +++ b/pkg/daemon/reset.go @@ -1,6 +1,7 @@ package daemon import ( + "context" "log/slog" "github.com/Azure/AKSFlexNode/pkg/config" @@ -10,6 +11,12 @@ import ( "github.com/Azure/unbounded/pkg/agent/phases/reset" ) +// ResetNode returns the task that removes the node runtime from the host. +// +// Helpers the agent library installs under the host root, the nspawn lifecycle +// helper and the LocalDNS network helper, are removed under the resolved host +// root and under the legacy root, so reset does not depend on the host root +// having been migrated. The managed agent binaries are intentionally kept. func ResetNode(log *slog.Logger) phases.Task { return phases.Serial(log, phases.Parallel(log, @@ -21,12 +28,59 @@ func ResetNode(log *slog.Logger) phases.Task { reset.CleanupMachine(log, goalstates.NSpawnMachineKube2), ), phases.Parallel(log, - reset.RemoveNetworkInterfaces(log), + // CleanupNetwork also removes the LocalDNS unit, nft table, dummy + // interface and helper, which reset previously left behind. + reset.CleanupNetwork(log), reset.RemoveWireGuardKeys(log), - reset.CleanupRoutes(log), cleanupLegacyBridgeCNI(log), ), + removeHostHelpers(log), reset.ReloadSystemd(log), config.RemoveRuntimeDirs(log), ) } + +// hostHelperPaths returns every location the agent library may have installed +// its host helpers to. CleanupNetwork removes the LocalDNS helper under the +// resolved root only. +func hostHelperPaths() []string { + return hostHelperPathsFor(goalstates.ResolveHostPaths(), goalstates.LegacyHostPaths()) +} + +func hostHelperPathsFor(resolved, legacy goalstates.HostPaths) []string { + paths := []string{resolved.NSpawnLifecycleBinary} + if legacy.Root != resolved.Root { + paths = append(paths, legacy.NSpawnLifecycleBinary, legacy.LocalDNSNetworkHelper) + } + + return paths +} + +type removeFilesTask struct { + name string + log *slog.Logger + paths []string +} + +// removeHostHelpers removes the nspawn lifecycle helper, and the LocalDNS +// helper under the legacy root. The machines and units that invoke them are +// gone by the time this runs, and the next start installs them again. +func removeHostHelpers(log *slog.Logger) phases.Task { + return &removeFilesTask{ + name: "remove-host-helpers", + log: log, + paths: hostHelperPaths(), + } +} + +func (t *removeFilesTask) Name() string { return t.name } + +func (t *removeFilesTask) Do(context.Context) error { + for _, path := range t.paths { + if err := removeIfPresent(path); err != nil { + return err + } + } + + return nil +} diff --git a/pkg/daemon/reset_test.go b/pkg/daemon/reset_test.go new file mode 100644 index 00000000..fd569ed0 --- /dev/null +++ b/pkg/daemon/reset_test.go @@ -0,0 +1,103 @@ +package daemon + +import ( + "errors" + "log/slog" + "os" + "path/filepath" + "slices" + "strings" + "testing" + + "github.com/Azure/unbounded/pkg/agent/goalstates" +) + +// TestResetNodeCleansUpLocalDNS pins that reset runs the library's full +// network cleanup. It used to call the interface and route cleanups directly +// and skip the LocalDNS one, so every reset of a LocalDNS node left the unit, +// the nft table, the dummy interface and the helper on the host. +func TestResetNodeCleansUpLocalDNS(t *testing.T) { + t.Parallel() + + name := ResetNode(slog.New(slog.DiscardHandler)).Name() + + for _, want := range []string{ + "cleanup-localdns-rules", + "remove-network-interfaces", + "cleanup-routes", + "remove-host-helpers", + } { + if !strings.Contains(name, want) { + t.Fatalf("ResetNode does not run %s: %s", want, name) + } + } +} + +// TestHostHelperPaths covers the library helpers reset removes. Reset does not +// migrate the host root, so the legacy root is swept too, including the LocalDNS +// helper, which CleanupNetwork removes under the resolved root only. +func TestHostHelperPaths(t *testing.T) { + t.Parallel() + + legacy := goalstates.LegacyHostPaths() + + tests := []struct { + name string + resolved goalstates.HostPaths + want []string + }{ + { + name: "host root also sweeps the legacy root", + resolved: goalstates.HostPaths{Root: "/opt/unbounded", NSpawnLifecycleBinary: "/opt/unbounded/bin/unbounded-agent-nspawn-lifecycle"}, + want: []string{ + "/opt/unbounded/bin/unbounded-agent-nspawn-lifecycle", + "/usr/local/bin/unbounded-agent-nspawn-lifecycle", + "/usr/local/libexec/unbounded-localdns-network", + }, + }, + { + name: "migrated host sweeps the legacy root once", + resolved: legacy, + want: []string{"/usr/local/bin/unbounded-agent-nspawn-lifecycle"}, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + + got := hostHelperPathsFor(tt.resolved, legacy) + if !slices.Equal(got, tt.want) { + t.Fatalf("hostHelperPathsFor(%q) = %v, want %v", tt.resolved.Root, got, tt.want) + } + }) + } +} + +// TestRemoveFilesTaskIsIdempotent runs the removal against a temp tree twice. +// Reset can be retried after a partial failure, so a second pass over files +// that are already gone must succeed. +func TestRemoveFilesTaskIsIdempotent(t *testing.T) { + t.Parallel() + + dir := t.TempDir() + present := filepath.Join(dir, "present") + if err := os.WriteFile(present, []byte("x"), 0o755); err != nil { + t.Fatal(err) + } + + task := &removeFilesTask{ + name: "test", + log: slog.New(slog.DiscardHandler), + paths: []string{present, filepath.Join(dir, "absent")}, + } + + for pass := 1; pass <= 2; pass++ { + if err := task.Do(t.Context()); err != nil { + t.Fatalf("pass %d: %v", pass, err) + } + } + if _, err := os.Lstat(present); !errors.Is(err, os.ErrNotExist) { + t.Fatalf("file was not removed: %v", err) + } +} diff --git a/pkg/daemon/testdata/aks-flex-node-agent-arc.service.golden b/pkg/daemon/testdata/aks-flex-node-agent-arc.service.golden index f2195753..a863d24b 100644 --- a/pkg/daemon/testdata/aks-flex-node-agent-arc.service.golden +++ b/pkg/daemon/testdata/aks-flex-node-agent-arc.service.golden @@ -10,7 +10,7 @@ StartLimitBurst=5 [Service] Type=simple RemainAfterExit=no -ExecStart=/usr/local/bin/aks-flex-node agent --config /etc/aks-flex-node/config.json +ExecStart=/opt/unbounded/bin/aks-flex-node agent --config /etc/aks-flex-node/config.json TimeoutStartSec=300 TimeoutStopSec=60 # Restart configuration for daemon resilience diff --git a/pkg/daemon/testdata/aks-flex-node-agent.service.golden b/pkg/daemon/testdata/aks-flex-node-agent.service.golden index 522b25d3..dd67aa83 100644 --- a/pkg/daemon/testdata/aks-flex-node-agent.service.golden +++ b/pkg/daemon/testdata/aks-flex-node-agent.service.golden @@ -10,7 +10,7 @@ StartLimitBurst=5 [Service] Type=simple RemainAfterExit=no -ExecStart=/usr/local/bin/aks-flex-node agent --config /etc/aks-flex-node/config.json +ExecStart=/opt/unbounded/bin/aks-flex-node agent --config /etc/aks-flex-node/config.json TimeoutStartSec=300 TimeoutStopSec=60 # Restart configuration for daemon resilience diff --git a/pkg/kubeauth/rest.go b/pkg/kubeauth/rest.go index 1b13ea47..eb5a25f5 100644 --- a/pkg/kubeauth/rest.go +++ b/pkg/kubeauth/rest.go @@ -39,5 +39,9 @@ func BootstrapRESTConfig(cfg *config.Config) (*rest.Config, error) { return nil, fmt.Errorf("kubernetes client requires bootstrap token or exec credential") } restCfg.ExecProvider = agentCfg.Kubelet.Auth.ExecCredential.DeepCopy() + // The agent config's credential is the kubelet's, and names the binary + // inside the machine. This client runs on the host, where the binary is + // under the host root. + restCfg.ExecProvider.Command = config.HostBinaryPath() return restCfg, nil } diff --git a/scripts/bootstrap.sh b/scripts/bootstrap.sh index b88ce3a6..4782a806 100755 --- a/scripts/bootstrap.sh +++ b/scripts/bootstrap.sh @@ -15,7 +15,13 @@ set -euo pipefail umask 077 readonly DEFAULT_REPOSITORY="Azure/AKSFlexNode" -readonly DEFAULT_INSTALL_DIR="/usr/local/bin" +# Where an agent released before the host root looks for its binary. A newer +# agent reports its own directory; see resolve_install_dir. +readonly LEGACY_INSTALL_DIR="/usr/local/bin" +# Exec-capable place to run the downloaded agent before it is installed. The +# temp dir may be on a noexec /tmp, and a failure to run there would be taken +# for an older agent. +readonly STAGING_PARENT="/var/lib/aks-flex-node" readonly DEFAULT_CONFIG_PATH="/etc/aks-flex-node/config.json" readonly DEFAULT_BOOTSTRAP_DATA_API_VERSION="2026-05-02-preview" readonly DEFAULT_AUTHORITY_HOST="https://login.microsoftonline.com" @@ -39,7 +45,11 @@ SP_CLIENT_CERTIFICATE_FILE="${AKS_FLEX_NODE_SP_CLIENT_CERTIFICATE_FILE:-}" AGENT_URL="${AKS_FLEX_NODE_AGENT_URL:-}" AGENT_VERSION="${AKS_FLEX_NODE_AGENT_VERSION:-}" AGENT_SHA256="${AKS_FLEX_NODE_AGENT_SHA256:-}" -INSTALL_DIR="${AKS_FLEX_NODE_INSTALL_DIR:-$DEFAULT_INSTALL_DIR}" +# Deprecated: the agent chooses its install directory. Kept only to reject a +# value the agent would not find. +REQUESTED_INSTALL_DIR="${AKS_FLEX_NODE_INSTALL_DIR:-}" +INSTALL_DIR="" +LEGACY_AGENT="" CONFIG_PATH="${AKS_FLEX_NODE_CONFIG_PATH:-$DEFAULT_CONFIG_PATH}" ENV_CONFIG_OVERRIDES="${AKS_FLEX_NODE_CONFIG_OVERRIDES:-}" BOOTSTRAP_OCI_IMAGE="${AKS_FLEX_NODE_BOOTSTRAP_OCI_IMAGE:-}" @@ -64,6 +74,7 @@ unset \ AKS_FLEX_NODE_BOOTSTRAP_OFFLINE_ARTIFACTS_SOURCE \ AKS_FLEX_NODE_CONFIG_OVERRIDES || true TEMP_DIR="" +STAGING_DIR="" log() { printf 'bootstrap: %s\n' "$*" >&2 @@ -104,7 +115,7 @@ Options: Override bootstrap.offlineArtifacts.source --config-overrides JSON JSON object deep-merged into the base config; repeatable and not suitable for secrets - --install-dir PATH Binary destination directory + --install-dir PATH Deprecated; the agent chooses its directory --config-path PATH Rendered config destination -h, --help Show this help @@ -174,7 +185,7 @@ parse_args() { --bootstrap-oci-image) BOOTSTRAP_OCI_IMAGE="$2" ;; --bootstrap-offline-artifacts-source) BOOTSTRAP_OFFLINE_ARTIFACTS_SOURCE="$2" ;; --config-overrides) CONFIG_OVERRIDES+=("$2") ;; - --install-dir) INSTALL_DIR="$2" ;; + --install-dir) REQUESTED_INSTALL_DIR="$2" ;; --config-path) CONFIG_PATH="$2" ;; esac shift 2 @@ -198,6 +209,9 @@ parse_args() { } cleanup() { + if [[ -n "$STAGING_DIR" && -d "$STAGING_DIR" ]]; then + rm -rf "$STAGING_DIR" + fi if [[ -n "$TEMP_DIR" && -d "$TEMP_DIR" ]]; then rm -rf "$TEMP_DIR" fi @@ -555,6 +569,34 @@ validate_archive_paths() { done < "$listing" } +# resolve_install_dir asks the downloaded agent where it keeps its files. An +# agent that answers host-root installs under the host root: /opt/unbounded, or +# /usr/local on a host an older release installed. An older agent has no such +# command and looks for its binary in /usr/local/bin. +resolve_install_dir() { + local candidate="$1" + local root + + install -d -o root -g root -m 0755 "$STAGING_PARENT" + STAGING_DIR=$(mktemp -d "$STAGING_PARENT/.bootstrap.XXXXXX") + install -o root -g root -m 0755 "$candidate" "$STAGING_DIR/aks-flex-node" + if root=$("$STAGING_DIR/aks-flex-node" host-root 2>/dev/null); then + [[ "$root" == /* && "$root" != *[[:space:]]* ]] || fatal "agent reported an invalid host root: $root" + INSTALL_DIR="${root%/}/bin" + else + INSTALL_DIR="$LEGACY_INSTALL_DIR" + LEGACY_AGENT=1 + fi + rm -rf "$STAGING_DIR" + STAGING_DIR="" + + if [[ -n "$REQUESTED_INSTALL_DIR" ]]; then + [[ "${REQUESTED_INSTALL_DIR%/}" == "$INSTALL_DIR" ]] || + fatal "--install-dir $REQUESTED_INSTALL_DIR is not $INSTALL_DIR, where this agent looks for its binary" + log "warning: --install-dir is deprecated; the agent chooses its install directory" + fi +} + download_and_install_agent() { local arch="$1" local url archive extract_dir expected candidate staged @@ -578,8 +620,15 @@ download_and_install_agent() { candidate=$(find "$extract_dir" -type f \( -name "$expected" -o -name aks-flex-node \) -print -quit) [[ -n "$candidate" ]] || fatal "agent binary not found in archive" - install -d -o root -g root -m 0755 "$INSTALL_DIR" - staged=$(mktemp "$INSTALL_DIR/.aks-flex-node.XXXXXX") + resolve_install_dir "$candidate" + # Created only when missing: an existing directory belongs to the image, and on a read-only + # /usr/local even setting its mode fails. + if ! { [[ -d "$INSTALL_DIR" ]] || install -d -o root -g root -m 0755 "$INSTALL_DIR"; } || + ! staged=$(mktemp "$INSTALL_DIR/.aks-flex-node.XXXXXX" 2>/dev/null); then + [[ -z "$LEGACY_AGENT" ]] || + fatal "this aks-flex-node release predates /opt/unbounded and needs a writable $LEGACY_INSTALL_DIR; use a newer release" + fatal "cannot write to $INSTALL_DIR" + fi install -o root -g root -m 0755 "$candidate" "$staged" mv -f "$staged" "$INSTALL_DIR/aks-flex-node" log "installed agent at $INSTALL_DIR/aks-flex-node" diff --git a/scripts/bootstrap_test.sh b/scripts/bootstrap_test.sh index fb08320d..fb73f281 100755 --- a/scripts/bootstrap_test.sh +++ b/scripts/bootstrap_test.sh @@ -6,6 +6,14 @@ if [[ $EUID -ne 0 ]]; then exec sudo -E bash "$0" "$@" fi +# Every case installs a stub agent as root. Run in a private mount namespace so /usr/local and +# /var/lib can be overlaid below: a case that installs an older release then cannot replace a real +# binary on the host, and the staging copy of the agent is not left on the host either. +if [[ -z "${BOOTSTRAP_TEST_ISOLATED:-}" ]]; then + command -v unshare >/dev/null || { echo "bootstrap_test: unshare is required" >&2; exit 1; } + exec env BOOTSTRAP_TEST_ISOLATED=1 unshare --mount --propagation private bash "$0" "$@" +fi + REPO_ROOT=$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd) SCRIPT="$REPO_ROOT/scripts/bootstrap.sh" WORK_DIR=$(mktemp -d) @@ -14,6 +22,11 @@ cleanup() { if [[ -n "$SERVER_PID" ]]; then kill "$SERVER_PID" 2>/dev/null || true fi + umount "$WORK_DIR/noexec-tmp" 2>/dev/null || true + umount /var/lib 2>/dev/null || true + # Twice: a failed legacy case leaves the read-only bind over the overlay. + umount /usr/local 2>/dev/null || true + umount /usr/local 2>/dev/null || true rm -rf "$WORK_DIR" } trap cleanup EXIT @@ -23,6 +36,24 @@ fail() { exit 1 } +# Relative paths, such as a rejected relative host root, resolve inside the work dir rather than +# wherever the test was started. +cd "$WORK_DIR" + +# Writes to /usr/local land in the overlay's upper dir, which is checked at the end. +USR_LOCAL_WRITES="$WORK_DIR/usr-local/upper" +mkdir -p "$USR_LOCAL_WRITES" "$WORK_DIR/usr-local/work" +mount -t overlay overlay \ + -o "lowerdir=/usr/local,upperdir=$USR_LOCAL_WRITES,workdir=$WORK_DIR/usr-local/work" /usr/local || + fail "could not overlay /usr/local" + +# bootstrap.sh stages the agent under /var/lib/aks-flex-node to ask it for its host root. +VAR_LIB_WRITES="$WORK_DIR/var-lib/upper" +mkdir -p "$VAR_LIB_WRITES" "$WORK_DIR/var-lib/work" +mount -t overlay overlay \ + -o "lowerdir=/var/lib,upperdir=$VAR_LIB_WRITES,workdir=$WORK_DIR/var-lib/work" /var/lib || + fail "could not overlay /var/lib" + command -v jq >/dev/null || fail "jq is required" bash -n "$SCRIPT" @@ -37,7 +68,19 @@ make_agent_archive() { mkdir -p "$dir" cat > "$dir/aks-flex-node-linux-$ARCH" <<'AGENT' #!/bin/bash +# host-root is answered before the calls are recorded: bootstrap.sh asks it from a staging copy, +# and the cases below check that every recorded call ran the installed binary. +if [[ "${1:-}" == host-root ]]; then + printf '%s\n' "$0" >> "${BOOTSTRAP_TEST_CALLS:?}.host-root" + if [[ -n "${BOOTSTRAP_TEST_LEGACY:-}" ]]; then + printf 'Error: unknown command "host-root" for "aks-flex-node"\n' >&2 + exit 1 + fi + printf '%s\n' "${BOOTSTRAP_TEST_HOST_ROOT:?}" + exit 0 +fi printf '%s\n' "$*" >> "${BOOTSTRAP_TEST_CALLS:?}" +printf '%s\n' "$0" >> "${BOOTSTRAP_TEST_CALLS}.path" if [[ "${1:-}" == fetch-bootstrap-data ]]; then output="" while (($# > 0)); do @@ -45,9 +88,13 @@ if [[ "${1:-}" == fetch-bootstrap-data ]]; then shift done [[ -n "$output" ]] || exit 24 - cat > "$output" <<'JSON' + if [[ -n "${BOOTSTRAP_TEST_FETCH_RESPONSE:-}" ]]; then + printf '%s\n' "$BOOTSTRAP_TEST_FETCH_RESPONSE" > "$output" + else + cat > "$output" <<'JSON' {"azure":{"bootstrapToken":{"token":"fresh1.0123456789abcdef"}},"components":{"kubernetes":"1.35.6"},"networking":{"dnsServiceIP":"10.0.0.10","cniVersion":"stale-cni"}} JSON + fi chmod 0600 "$output" exit 0 fi @@ -81,6 +128,7 @@ JSON chmod 0600 "$WORK_DIR/base.json" BOOTSTRAP_TEST_CALLS="$WORK_DIR/msi-calls" \ +BOOTSTRAP_TEST_HOST_ROOT="$WORK_DIR/msi" \ AKS_FLEX_NODE_BASE_CONFIG_FILE="$WORK_DIR/base.json" \ AKS_FLEX_NODE_AUTH=service-principal \ AKS_FLEX_NODE_SP_CLIENT_ID=environment-client \ @@ -95,7 +143,6 @@ AKS_FLEX_NODE_CONFIG_OVERRIDES='{"node":{"labels":{"environment":"true"}}}' \ --msi-client-id cli-msi \ --bootstrap-oci-image 'https://cli.example/rootfs.tar.gz' \ --config-overrides '{"node":{"labels":{"cli":"true"}},"bootstrap":{"offlineArtifacts":{"source":"https://generic-cli.example/ignored.tar.gz"}}}' \ - --install-dir "$WORK_DIR/msi-bin" \ --config-path "$WORK_DIR/msi-etc/config.json" >/dev/null jq -e ' @@ -112,6 +159,7 @@ grep -Fx "preflight --config $WORK_DIR/msi-etc/config.json --output text" "$WORK grep -Fx "start --config $WORK_DIR/msi-etc/config.json" "$WORK_DIR/msi-calls" >/dev/null BOOTSTRAP_TEST_CALLS="$WORK_DIR/arc-calls" \ +BOOTSTRAP_TEST_HOST_ROOT="$WORK_DIR/arc" \ AKS_FLEX_NODE_BASE_CONFIG_FILE="$WORK_DIR/base.json" \ AKS_FLEX_NODE_AGENT_URL="$AGENT_URL" \ bash "$SCRIPT" \ @@ -119,7 +167,6 @@ AKS_FLEX_NODE_AGENT_URL="$AGENT_URL" \ --fetch-bootstrap-data \ --cluster-resource-id '/subscriptions/00000000-0000-0000-0000-000000000000/resourceGroups/rg/providers/Microsoft.ContainerService/managedClusters/cluster' \ --agent-pool-name aksflexnodes \ - --install-dir "$WORK_DIR/arc-bin" \ --config-path "$WORK_DIR/arc-etc/config.json" >/dev/null jq -e ' @@ -129,10 +176,15 @@ jq -e ' .azure.bootstrapToken.token == "fresh1.0123456789abcdef" ' "$WORK_DIR/arc-etc/config.json" >/dev/null grep -E '^fetch-bootstrap-data .*--auth arc( |$)' "$WORK_DIR/arc-calls" >/dev/null +# Every command, including the bootstrap-data fetch, runs the installed binary. Running it from +# the temp dir would fail on hosts that mount /tmp noexec. +[[ "$(sort -u "$WORK_DIR/arc-calls.path")" == "$WORK_DIR/arc/bin/aks-flex-node" ]] || + fail "agent ran from $(sort -u "$WORK_DIR/arc-calls.path" | tr '\n' ' '), want the installed binary" printf 's"e\\cret\n' > "$WORK_DIR/client-secret" chmod 0600 "$WORK_DIR/client-secret" BOOTSTRAP_TEST_CALLS="$WORK_DIR/sp-calls" \ +BOOTSTRAP_TEST_HOST_ROOT="$WORK_DIR/sp" \ AKS_FLEX_NODE_BASE_CONFIG_FILE="$WORK_DIR/base.json" \ bash "$SCRIPT" \ --auth service-principal \ @@ -140,7 +192,6 @@ AKS_FLEX_NODE_BASE_CONFIG_FILE="$WORK_DIR/base.json" \ --sp-client-secret-file "$WORK_DIR/client-secret" \ --agent-url "$AGENT_URL" \ --agent-sha256 "$AGENT_SHA256" \ - --install-dir "$WORK_DIR/sp-bin" \ --config-path "$WORK_DIR/sp-etc/config.json" >/dev/null jq -e --arg secretFile "$WORK_DIR/client-secret" ' @@ -154,13 +205,13 @@ jq -e --arg secretFile "$WORK_DIR/client-secret" ' ' "$WORK_DIR/sp-etc/config.json" >/dev/null BOOTSTRAP_TEST_CALLS="$WORK_DIR/sp-inline-calls" \ +BOOTSTRAP_TEST_HOST_ROOT="$WORK_DIR/sp-inline" \ AKS_FLEX_NODE_BASE_CONFIG_FILE="$WORK_DIR/base.json" \ AKS_FLEX_NODE_SP_CLIENT_SECRET='inline-secret' \ AKS_FLEX_NODE_AGENT_URL="$AGENT_URL" \ bash "$SCRIPT" \ --auth service-principal \ --sp-client-id inline-client \ - --install-dir "$WORK_DIR/sp-inline-bin" \ --config-path "$WORK_DIR/sp-inline-etc/config.json" >/dev/null jq -e ' .azure.servicePrincipal.clientId == "inline-client" and @@ -170,12 +221,12 @@ jq -e ' ln -s "$WORK_DIR/client-secret" "$WORK_DIR/client-secret-link" if BOOTSTRAP_TEST_CALLS="$WORK_DIR/sp-link-calls" \ + BOOTSTRAP_TEST_HOST_ROOT="$WORK_DIR/sp-link" \ AKS_FLEX_NODE_BASE_CONFIG_FILE="$WORK_DIR/base.json" \ AKS_FLEX_NODE_AGENT_URL="$AGENT_URL" \ bash "$SCRIPT" --auth service-principal \ --sp-client-id link-client \ --sp-client-secret-file "$WORK_DIR/client-secret-link" \ - --install-dir "$WORK_DIR/sp-link-bin" \ --config-path "$WORK_DIR/sp-link-etc/config.json" \ >"$WORK_DIR/sp-link.log" 2>&1; then fail "symlink client-secret file was accepted" @@ -274,6 +325,7 @@ JSON chmod 0600 "$WORK_DIR/fetch-base.json" BOOTSTRAP_TEST_CALLS="$WORK_DIR/fetch-calls" \ +BOOTSTRAP_TEST_HOST_ROOT="$WORK_DIR/fetch" \ AKS_FLEX_NODE_BASE_CONFIG_FILE="$WORK_DIR/fetch-base.json" \ AKS_FLEX_NODE_IMDS_ENDPOINT="http://127.0.0.1:${port}/metadata/identity/oauth2/token" \ AKS_FLEX_NODE_ALLOW_INSECURE_TEST_ENDPOINTS=true \ @@ -287,7 +339,6 @@ AKS_FLEX_NODE_AGENT_URL="$AGENT_URL" \ --agent-pool-name aksflexnodes \ --resource-manager-endpoint "http://127.0.0.1:${port}" \ --config-overrides '{"azure":{"targetCluster":{"resourceId":"/subscriptions/wrong/resourceGroups/wrong/providers/Microsoft.ContainerService/managedClusters/wrong"},"targetAgentPoolName":"wrongpool"},"node":{"labels":{"fresh":"true"}}}' \ - --install-dir "$WORK_DIR/fetch-bin" \ --config-path "$WORK_DIR/fetch-etc/config.json" >/dev/null jq -e --arg armEndpoint "http://127.0.0.1:${port}" ' @@ -307,6 +358,7 @@ jq -e --arg armEndpoint "http://127.0.0.1:${port}" ' # The raw repository script can start from an implicit empty object when fresh # bootstrap data and the target cluster/pool are supplied. BOOTSTRAP_TEST_CALLS="$WORK_DIR/fetch-empty-base-calls" \ +BOOTSTRAP_TEST_HOST_ROOT="$WORK_DIR/fetch-empty-base" \ AKS_FLEX_NODE_IMDS_ENDPOINT="http://127.0.0.1:${port}/metadata/identity/oauth2/token" \ AKS_FLEX_NODE_ALLOW_INSECURE_TEST_ENDPOINTS=true \ AKS_FLEX_NODE_AGENT_URL="$AGENT_URL" \ @@ -316,7 +368,6 @@ AKS_FLEX_NODE_AGENT_URL="$AGENT_URL" \ --cluster-resource-id '/subscriptions/00000000-0000-0000-0000-000000000000/resourceGroups/rg/providers/Microsoft.ContainerService/managedClusters/cluster' \ --agent-pool-name aksflexnodes \ --resource-manager-endpoint "http://127.0.0.1:${port}" \ - --install-dir "$WORK_DIR/fetch-empty-base-bin" \ --config-path "$WORK_DIR/fetch-empty-base-etc/config.json" >/dev/null jq -e --arg armEndpoint "http://127.0.0.1:${port}" ' @@ -330,9 +381,9 @@ jq -e --arg armEndpoint "http://127.0.0.1:${port}" ' ' "$WORK_DIR/fetch-empty-base-etc/config.json" >/dev/null if BOOTSTRAP_TEST_CALLS="$WORK_DIR/no-base-calls" \ + BOOTSTRAP_TEST_HOST_ROOT="$WORK_DIR/no-base" \ AKS_FLEX_NODE_AGENT_URL="$AGENT_URL" \ bash "$SCRIPT" --auth msi \ - --install-dir "$WORK_DIR/no-base-bin" \ --config-path "$WORK_DIR/no-base-etc/config.json" \ >"$WORK_DIR/no-base.log" 2>&1; then fail "unpopulated embedded config was accepted without bootstrap-data fetch" @@ -366,13 +417,13 @@ JSON chmod 0600 "$WORK_DIR/fetch-sp-base.json" BOOTSTRAP_TEST_CALLS="$WORK_DIR/fetch-sp-calls" \ +BOOTSTRAP_TEST_HOST_ROOT="$WORK_DIR/fetch-sp" \ AKS_FLEX_NODE_BASE_CONFIG_FILE="$WORK_DIR/fetch-sp-base.json" \ AKS_FLEX_NODE_FETCH_BOOTSTRAP_DATA=true \ AKS_FLEX_NODE_AUTHORITY_HOST="http://127.0.0.1:${port}" \ AKS_FLEX_NODE_ALLOW_INSECURE_TEST_ENDPOINTS=true \ AKS_FLEX_NODE_AGENT_URL="$AGENT_URL" \ bash "$SCRIPT" \ - --install-dir "$WORK_DIR/fetch-sp-bin" \ --config-path "$WORK_DIR/fetch-sp-etc/config.json" >/dev/null jq -e --arg secretFile "$WORK_DIR/fetch-sp-secret" ' @@ -399,6 +450,7 @@ command -v openssl >/dev/null || fail "openssl is required by the certificate bo chmod 0600 "$WORK_DIR/client-certificate" BOOTSTRAP_TEST_CALLS="$WORK_DIR/fetch-cert-calls" \ + BOOTSTRAP_TEST_HOST_ROOT="$WORK_DIR/fetch-cert" \ AKS_FLEX_NODE_BASE_CONFIG_FILE="$WORK_DIR/fetch-base.json" \ AKS_FLEX_NODE_AUTHORITY_HOST="http://127.0.0.1:${port}" \ AKS_FLEX_NODE_ALLOW_INSECURE_TEST_ENDPOINTS=true \ @@ -413,7 +465,6 @@ command -v openssl >/dev/null || fail "openssl is required by the certificate bo --cluster-resource-id '/subscriptions/00000000-0000-0000-0000-000000000000/resourceGroups/rg/providers/Microsoft.ContainerService/managedClusters/cluster' \ --agent-pool-name aksflexnodes \ --resource-manager-endpoint "http://127.0.0.1:${port}" \ - --install-dir "$WORK_DIR/fetch-cert-bin" \ --config-path "$WORK_DIR/fetch-cert-etc/config.json" >/dev/null jq -e --arg certificateFile "$WORK_DIR/client-certificate" ' @@ -431,6 +482,7 @@ command -v openssl >/dev/null || fail "openssl is required by the certificate bo -out "$WORK_DIR/client-certificate.pfx" >/dev/null 2>&1 chmod 0600 "$WORK_DIR/client-certificate.pfx" BOOTSTRAP_TEST_CALLS="$WORK_DIR/fetch-pfx-calls" \ + BOOTSTRAP_TEST_HOST_ROOT="$WORK_DIR/fetch-pfx" \ AKS_FLEX_NODE_BASE_CONFIG_FILE="$WORK_DIR/fetch-base.json" \ AKS_FLEX_NODE_AUTHORITY_HOST="http://127.0.0.1:${port}" \ AKS_FLEX_NODE_ALLOW_INSECURE_TEST_ENDPOINTS=true \ @@ -444,11 +496,121 @@ command -v openssl >/dev/null || fail "openssl is required by the certificate bo --cluster-resource-id '/subscriptions/00000000-0000-0000-0000-000000000000/resourceGroups/rg/providers/Microsoft.ContainerService/managedClusters/cluster' \ --agent-pool-name aksflexnodes \ --resource-manager-endpoint "http://127.0.0.1:${port}" \ - --install-dir "$WORK_DIR/fetch-pfx-bin" \ --config-path "$WORK_DIR/fetch-pfx-etc/config.json" >/dev/null jq -e --arg certificateFile "$WORK_DIR/client-certificate.pfx" ' .azure.servicePrincipal.clientSecretFile == $certificateFile and .azure.bootstrapToken.token == "fresh1.0123456789abcdef" ' "$WORK_DIR/fetch-pfx-etc/config.json" >/dev/null +# The binary goes under the host root the agent reports, in directories created 0755 even under the +# script's umask, so systemd and the agent can reach it. +BOOTSTRAP_TEST_CALLS="$WORK_DIR/root-calls" \ +BOOTSTRAP_TEST_HOST_ROOT="$WORK_DIR/root/opt/unbounded" \ +AKS_FLEX_NODE_BASE_CONFIG_FILE="$WORK_DIR/base.json" \ +AKS_FLEX_NODE_AGENT_URL="$AGENT_URL" \ + bash "$SCRIPT" --auth arc --config-path "$WORK_DIR/root-etc/config.json" >/dev/null +[[ -x "$WORK_DIR/root/opt/unbounded/bin/aks-flex-node" ]] || fail "binary was not installed under the host root" +for dir in opt opt/unbounded opt/unbounded/bin; do + [[ $(stat -c '%a' "$WORK_DIR/root/$dir") == 755 ]] || fail "new directory $dir is not 0755" +done +[[ "$(sort -u "$WORK_DIR/root-calls.path")" == "$WORK_DIR/root/opt/unbounded/bin/aks-flex-node" ]] || + fail "agent ran from $(sort -u "$WORK_DIR/root-calls.path" | tr '\n' ' '), want the installed binary" + +# The host root is asked of a copy under /var/lib, not of the download in the temp dir: on a host +# that mounts /tmp noexec that could not run, and would be taken for an older release. +[[ "$(cat "$WORK_DIR/root-calls.host-root")" == /var/lib/aks-flex-node/.bootstrap.*/aks-flex-node ]] || + fail "host-root ran from $(cat "$WORK_DIR/root-calls.host-root"), want the staging copy" +mkdir -p "$WORK_DIR/noexec-tmp" +mount -t tmpfs -o noexec,mode=0700 tmpfs "$WORK_DIR/noexec-tmp" || fail "could not mount a noexec temp dir" +TMPDIR="$WORK_DIR/noexec-tmp" \ +BOOTSTRAP_TEST_CALLS="$WORK_DIR/noexec-calls" \ +BOOTSTRAP_TEST_HOST_ROOT="$WORK_DIR/noexec" \ +AKS_FLEX_NODE_BASE_CONFIG_FILE="$WORK_DIR/base.json" \ +AKS_FLEX_NODE_AGENT_URL="$AGENT_URL" \ + bash "$SCRIPT" --auth arc --config-path "$WORK_DIR/noexec-etc/config.json" >/dev/null || + fail "bootstrap failed with a noexec temp dir" +[[ -x "$WORK_DIR/noexec/bin/aks-flex-node" ]] || fail "a noexec temp dir sent the binary elsewhere" +umount "$WORK_DIR/noexec-tmp" + +# The staging copy does not outlive the run. +if compgen -G "$VAR_LIB_WRITES/aks-flex-node/.bootstrap.*" >/dev/null; then + fail "staging copies were left in /var/lib/aks-flex-node" +fi + +# --install-dir is deprecated. It is accepted only when it matches the directory the agent reports, +# since the agent cannot find a binary installed anywhere else. +BOOTSTRAP_TEST_CALLS="$WORK_DIR/matching-calls" \ +BOOTSTRAP_TEST_HOST_ROOT="$WORK_DIR/matching" \ +AKS_FLEX_NODE_BASE_CONFIG_FILE="$WORK_DIR/base.json" \ +AKS_FLEX_NODE_AGENT_URL="$AGENT_URL" \ + bash "$SCRIPT" --auth arc \ + --install-dir "$WORK_DIR/matching/bin/" \ + --config-path "$WORK_DIR/matching-etc/config.json" >"$WORK_DIR/matching.log" 2>&1 || + fail "a matching --install-dir was rejected" +grep -q "deprecated" "$WORK_DIR/matching.log" || fail "--install-dir did not warn that it is deprecated" + +if BOOTSTRAP_TEST_CALLS="$WORK_DIR/mismatch-calls" \ + BOOTSTRAP_TEST_HOST_ROOT="$WORK_DIR/mismatch" \ + AKS_FLEX_NODE_BASE_CONFIG_FILE="$WORK_DIR/base.json" \ + AKS_FLEX_NODE_AGENT_URL="$AGENT_URL" \ + bash "$SCRIPT" --auth arc \ + --install-dir "$WORK_DIR/elsewhere" \ + --config-path "$WORK_DIR/mismatch-etc/config.json" >"$WORK_DIR/mismatch.log" 2>&1; then + fail "an --install-dir that disagrees with the agent was accepted" +fi +grep -q "where this agent looks for its binary" "$WORK_DIR/mismatch.log" || fail "mismatch was not explained" +[[ ! -e "$WORK_DIR/elsewhere/aks-flex-node" && ! -e "$WORK_DIR/mismatch/bin/aks-flex-node" ]] || + fail "a binary was installed despite the mismatch" +[[ ! -s "$WORK_DIR/mismatch-calls" ]] || fail "the agent was run despite the mismatch" + +# A host root that is not an absolute path is refused before anything is installed. +if BOOTSTRAP_TEST_CALLS="$WORK_DIR/relative-calls" \ + BOOTSTRAP_TEST_HOST_ROOT="relative/root" \ + AKS_FLEX_NODE_BASE_CONFIG_FILE="$WORK_DIR/base.json" \ + AKS_FLEX_NODE_AGENT_URL="$AGENT_URL" \ + bash "$SCRIPT" --auth arc --config-path "$WORK_DIR/relative-etc/config.json" >"$WORK_DIR/relative.log" 2>&1; then + fail "a relative host root was accepted" +fi +grep -q "invalid host root" "$WORK_DIR/relative.log" || fail "relative host root was not reported" +[[ ! -e "$WORK_DIR/relative-calls" && ! -e "$WORK_DIR/relative" ]] || fail "the agent was installed under a relative host root" + +# Every case above installs under the work dir, so nothing may have been written to /usr/local. +if [[ -n "$(ls -A "$USR_LOCAL_WRITES")" ]]; then + fail "a case wrote to /usr/local: $(cd "$USR_LOCAL_WRITES" && find . -mindepth 1 | tr '\n' ' ')" +fi + +# A release before the host root has no host-root command and looks for its binary in /usr/local/bin, +# so that is where it goes. These cases write to /usr/local, so they run last and only on the overlay. +[[ "$(findmnt -n -o FSTYPE /usr/local)" == overlay ]] || fail "/usr/local is not overlaid; refusing the legacy cases" + +# On a host where /usr/local is read-only, such as Azure Container Linux, an older release cannot be +# installed at all, and the operator is told to use a newer one. +# A read-only bind over the overlay, since overlayfs cannot be remounted read-only. +{ mount --bind /usr/local /usr/local && mount -o remount,bind,ro /usr/local; } || fail "could not make /usr/local read-only" +if BOOTSTRAP_TEST_CALLS="$WORK_DIR/legacy-ro-calls" \ + BOOTSTRAP_TEST_LEGACY=1 \ + AKS_FLEX_NODE_BASE_CONFIG_FILE="$WORK_DIR/base.json" \ + AKS_FLEX_NODE_AGENT_URL="$AGENT_URL" \ + bash "$SCRIPT" --auth arc --config-path "$WORK_DIR/legacy-ro-etc/config.json" >"$WORK_DIR/legacy-ro.log" 2>&1; then + fail "an older release was installed on a read-only /usr/local" +fi +umount /usr/local || fail "could not make /usr/local writable again" +grep -q "predates /opt/unbounded" "$WORK_DIR/legacy-ro.log" || fail "a read-only /usr/local was not explained" +[[ ! -s "$WORK_DIR/legacy-ro-calls" ]] || fail "the agent was run despite a failed install" + +# /usr/local/bin belongs to the image, so its mode is left as the image set it. +legacy_bin_mode=$(stat -c '%a' /usr/local/bin) +chmod 0750 /usr/local/bin +BOOTSTRAP_TEST_CALLS="$WORK_DIR/legacy-calls" \ +BOOTSTRAP_TEST_LEGACY=1 \ +AKS_FLEX_NODE_BASE_CONFIG_FILE="$WORK_DIR/base.json" \ +AKS_FLEX_NODE_AGENT_URL="$AGENT_URL" \ + bash "$SCRIPT" --auth arc --config-path "$WORK_DIR/legacy-etc/config.json" >/dev/null +grep -q BOOTSTRAP_TEST_CALLS "$USR_LOCAL_WRITES/bin/aks-flex-node" 2>/dev/null || + fail "an older release was not installed at /usr/local/bin" +[[ "$(sort -u "$WORK_DIR/legacy-calls.path")" == /usr/local/bin/aks-flex-node ]] || + fail "an older release ran from $(sort -u "$WORK_DIR/legacy-calls.path" | tr '\n' ' ')" +[[ $(stat -c '%a' /usr/local/bin) == 750 ]] || fail "the mode of an existing /usr/local/bin was changed" +chmod "$legacy_bin_mode" /usr/local/bin + echo "bootstrap script tests passed" diff --git a/scripts/embed.go b/scripts/embed.go new file mode 100644 index 00000000..e23d5559 --- /dev/null +++ b/scripts/embed.go @@ -0,0 +1,10 @@ +// Package scripts embeds the host scripts that the aks-flex-node binary hands to hosts. +package scripts + +import _ "embed" + +// Bootstrap is bootstrap.sh, which `aks-flex-node ignition` writes to hosts that run it on first +// boot. +// +//go:embed bootstrap.sh +var Bootstrap string diff --git a/scripts/install.sh b/scripts/install.sh index 6a309f56..7c63dfee 100755 --- a/scripts/install.sh +++ b/scripts/install.sh @@ -3,8 +3,9 @@ # This script downloads and installs an AKS Flex Node binary from GitHub releases or a custom archive URL. # # Scope: initial installation and reinstall after reset. While the agent service is installed, -# /usr/local/bin/aks-flex-node is a symlink into the managed blue/green layout and must be updated -# through the agent upgrade flow. +# /bin/aks-flex-node is a symlink into the managed blue/green layout and must be updated +# through the agent upgrade flow. The host root is /opt/unbounded, or /usr/local on a host installed +# by a release before it; the downloaded binary reports which. set -euo pipefail @@ -20,8 +21,12 @@ REPO="Azure/AKSFlexNode" SERVICE_NAME="aks-flex-node" SERVICE_UNIT="aks-flex-node-agent.service" SERVICE_UNIT_PATH="/etc/systemd/system/$SERVICE_UNIT" -INSTALL_DIR="/usr/local/bin" -MANAGED_BINARY_DIR="/usr/local/lib/aks-flex-node" +# Where a release before the host root installs. resolve_install_dir replaces these with the +# directories the downloaded binary reports. +LEGACY_ROOT="/usr/local" +INSTALL_DIR="$LEGACY_ROOT/bin" +MANAGED_BINARY_DIR="$LEGACY_ROOT/lib/aks-flex-node" +LEGACY_AGENT=false AGENT_UPGRADE_LOCK_PATH="/run/aks-flex-node-agent-upgrade.lock" CONFIG_DIR="/etc/aks-flex-node" DATA_DIR="/var/lib/aks-flex-node" @@ -244,6 +249,45 @@ download_binary() { echo "$temp_dir/$binary_name" } +# resolve_install_dir asks the downloaded binary where it keeps its files. A release that answers +# host-root installs under the host root: /opt/unbounded, or /usr/local on a host an older release +# installed. An older release has no such command and installs under /usr/local. +# +# The binary runs from a copy under $DATA_DIR rather than the download directory, which may be on a +# noexec /tmp: a failure to run there would be taken for an older release. +resolve_install_dir() { + local binary_path="$1" + local staging root + + if ! mkdir -p "$DATA_DIR" || ! staging=$(mktemp -d "$DATA_DIR/.install.XXXXXX"); then + log_error "Failed to create a staging directory in $DATA_DIR" + return 1 + fi + if ! install -m 0755 "$binary_path" "$staging/aks-flex-node"; then + rm -rf -- "$staging" + log_error "Failed to stage the binary in $staging" + return 1 + fi + + if root=$("$staging/aks-flex-node" host-root 2>/dev/null); then + rm -rf -- "$staging" + if [[ "$root" != /* || "$root" == *[[:space:]]* ]]; then + log_error "The binary reported an invalid host root: $root" + return 1 + fi + root="${root%/}" + INSTALL_DIR="$root/bin" + MANAGED_BINARY_DIR="$root/lib/aks-flex-node" + return 0 + fi + + rm -rf -- "$staging" + INSTALL_DIR="$LEGACY_ROOT/bin" + MANAGED_BINARY_DIR="$LEGACY_ROOT/lib/aks-flex-node" + LEGACY_AGENT=true + log_warning "This release predates /opt/unbounded and installs under $LEGACY_ROOT" +} + is_managed_binary_link() { local target_path="$1" local current_path="$MANAGED_BINARY_DIR/aks-flex-node-current" @@ -263,9 +307,10 @@ install_binary() { log_info "Installing binary to $INSTALL_DIR..." - # Minimal and custom images aren't required to pre-create /usr/local/bin. - # Create a missing destination, but don't change an existing directory's - # ownership or mode because it can be managed by the host image owner. + # Minimal and custom images aren't required to pre-create /usr/local/bin, + # and the host root does not exist before the first installation. Create a + # missing destination, but don't change an existing directory's ownership + # or mode because it can be managed by the host image owner. if [[ ! -e "$INSTALL_DIR" ]]; then if ! install -d -o root -g root -m 0755 "$INSTALL_DIR"; then log_error "Failed to create install directory $INSTALL_DIR" @@ -315,6 +360,9 @@ install_binary() { if ! staged=$(mktemp "$INSTALL_DIR/.aks-flex-node.XXXXXX") || ! install -o root -g root -m 0755 "$binary_path" "$staged"; then log_error "Failed to stage binary in $INSTALL_DIR" + if [[ "$LEGACY_AGENT" == true ]]; then + log_error "This release predates /opt/unbounded and needs a writable $INSTALL_DIR; install a newer release" + fi exit 1 fi # -T makes a directory appearing at the target fail rather than receive the staged file. @@ -436,6 +484,7 @@ main() { # Download binary local binary_path binary_path=$(download_binary "$version" "$os" "$arch") + resolve_install_dir "$binary_path" || exit 1 # Install binary install_binary "$binary_path" diff --git a/scripts/install_test.sh b/scripts/install_test.sh index 5b33b181..4047f40e 100755 --- a/scripts/install_test.sh +++ b/scripts/install_test.sh @@ -12,6 +12,7 @@ WORK_DIR=$(mktemp -d) RUNNING_PID="" cleanup() { [[ -z "$RUNNING_PID" ]] || kill "$RUNNING_PID" 2>/dev/null || true + umount "$WORK_DIR/ro" 2>/dev/null || true rm -rf "$WORK_DIR" } trap cleanup EXIT @@ -133,4 +134,67 @@ assert_no_staged_files fi ) +# The install directories come from the downloaded binary. It is asked from a copy under DATA_DIR, +# because the download directory may be on a noexec /tmp. +DATA_DIR="$WORK_DIR/data" +HOST_ROOT_CALLS="$WORK_DIR/host-root-calls" + +# make_stub writes a binary that answers host-root with $2, or, with no $2, a release from before the +# host root, which has no such command. +make_stub() { + local path="$1" root="${2:-}" + { + printf '#!/bin/bash\n' + printf 'if [[ "${1:-}" == host-root ]]; then\n' + printf ' printf "%%s\\n" "$0" >> %q\n' "$HOST_ROOT_CALLS" + if [[ -n "$root" ]]; then + printf ' printf "%%s\\n" %q\n exit 0\n' "$root" + else + printf ' echo "Error: unknown command \\"host-root\\"" >&2\n exit 1\n' + fi + printf 'fi\nprintf "%%s\\n" "$*"\n' + } > "$path" + chmod 0755 "$path" +} + +make_stub "$WORK_DIR/current-agent" "$WORK_DIR/root/opt/unbounded/" +LEGACY_AGENT=false +resolve_install_dir "$WORK_DIR/current-agent" >/dev/null || fail "resolving a current release failed" +[[ "$INSTALL_DIR" == "$WORK_DIR/root/opt/unbounded/bin" && + "$MANAGED_BINARY_DIR" == "$WORK_DIR/root/opt/unbounded/lib/aks-flex-node" ]] || + fail "a current release was not installed under its host root: $INSTALL_DIR, $MANAGED_BINARY_DIR" +[[ "$LEGACY_AGENT" == false ]] || fail "a current release was taken for an older one" +[[ "$(tail -1 "$HOST_ROOT_CALLS")" == "$DATA_DIR"/.install.*/aks-flex-node ]] || + fail "host-root ran from $(tail -1 "$HOST_ROOT_CALLS"), want the staging copy" +if compgen -G "$DATA_DIR/.install.*" >/dev/null; then + fail "the staging copy was left in $DATA_DIR" +fi + +# The host root does not exist before the first install. +install_binary "$replacement_binary" >/dev/null || fail "install into a missing host root failed" +[[ -x "$INSTALL_DIR/aks-flex-node" ]] || fail "binary missing under the new host root" +[[ "$(stat -c %a "$INSTALL_DIR")" == "755" ]] || fail "host root bin dir mode is $(stat -c %a "$INSTALL_DIR"), want 755" + +make_stub "$WORK_DIR/relative-agent" "relative/root" +if resolve_install_dir "$WORK_DIR/relative-agent" >/dev/null 2>&1; then + fail "a relative host root was accepted" +fi + +make_stub "$WORK_DIR/legacy-agent" +resolve_install_dir "$WORK_DIR/legacy-agent" >"$WORK_DIR/legacy.log" 2>&1 || fail "resolving an older release failed" +[[ "$INSTALL_DIR" == "/usr/local/bin" && "$MANAGED_BINARY_DIR" == "/usr/local/lib/aks-flex-node" ]] || + fail "an older release was not installed under /usr/local: $INSTALL_DIR" +[[ "$LEGACY_AGENT" == true ]] || fail "an older release was not recorded as one" +grep -q "predates /opt/unbounded" "$WORK_DIR/legacy.log" || fail "an older release was not reported" + +# Where that directory is read-only, as on Azure Container Linux, the operator is told why. +mkdir -p "$WORK_DIR/ro" +mount -t tmpfs -o ro tmpfs "$WORK_DIR/ro" || fail "could not mount a read-only directory" +INSTALL_DIR="$WORK_DIR/ro" +if install_binary "$replacement_binary" >"$WORK_DIR/ro.log" 2>&1; then + fail "installed into a read-only directory" +fi +umount "$WORK_DIR/ro" +grep -q "predates /opt/unbounded and needs a writable" "$WORK_DIR/ro.log" || fail "a read-only install directory was not explained" + printf 'install_test: ok\n' diff --git a/scripts/uninstall.sh b/scripts/uninstall.sh index 34368060..78c1300b 100755 --- a/scripts/uninstall.sh +++ b/scripts/uninstall.sh @@ -12,12 +12,21 @@ BLUE='\033[0;34m' NC='\033[0m' # No Color # Configuration (should match install.sh) -INSTALL_DIR="/usr/local/bin" +# The binaries are under the host root, or under the legacy root on a host installed by a release +# before the host root. There the host root may be a link to the legacy root. Both are swept. +HOST_ROOT="/opt/unbounded" +LEGACY_ROOT="/usr/local" +# The directory reset runs the binary from; see find_install_dir. +INSTALL_DIR="$HOST_ROOT/bin" CONFIG_DIR="/etc/aks-flex-node" DATA_DIR="/var/lib/aks-flex-node" LOG_DIR="/var/log/aks-flex-node" SERVICE_UNIT="aks-flex-node-agent.service" SERVICE_UNIT_PATH="/etc/systemd/system/$SERVICE_UNIT" +RECOVERY_UNIT_PATH="/etc/systemd/system/aks-flex-node-agent-recovery.service" +# Installed by `aks-flex-node ignition` to run bootstrap.sh on first boot. +FIRST_BOOT_UNIT="aks-flex-node-bootstrap.service" +FIRST_BOOT_UNIT_PATH="/etc/systemd/system/$FIRST_BOOT_UNIT" # Functions log_info() { @@ -36,12 +45,24 @@ log_error() { echo -e "${RED}ERROR:${NC} $1" } +# find_install_dir picks the root that holds the binary, so reset can run it. +find_install_dir() { + local root + + for root in "$HOST_ROOT" "$LEGACY_ROOT"; do + if [[ -x "$root/bin/aks-flex-node" ]]; then + INSTALL_DIR="$root/bin" + return 0 + fi + done +} + confirm_uninstall() { echo -e "${YELLOW}AKS Flex Node Uninstaller${NC}" echo -e "${YELLOW}===========================${NC}" echo "" echo "This will remove the following components:" - echo "• AKS Flex Node binary ($INSTALL_DIR/aks-flex-node)" + echo "• AKS Flex Node binaries (under $HOST_ROOT and $LEGACY_ROOT)" echo "• Systemd service (aks-flex-node-agent.service)" echo "• Configuration directory ($CONFIG_DIR)" echo "• Data directory ($DATA_DIR)" @@ -75,13 +96,18 @@ run_reset() { systemctl stop "$SERVICE_UNIT" 2>/dev/null || true systemctl disable "$SERVICE_UNIT" 2>/dev/null || true - - if [[ -e "$SERVICE_UNIT_PATH" ]]; then - rm -f "$SERVICE_UNIT_PATH" - log_success "Removed systemd unit: $SERVICE_UNIT_PATH" - else - log_info "Systemd unit not found: $SERVICE_UNIT_PATH" - fi + # --now as well: the first-boot unit stays active after it runs, and a later provisioning + # could not start it again. + systemctl disable --now "$FIRST_BOOT_UNIT" 2>/dev/null || true + + for unit_path in "$SERVICE_UNIT_PATH" "$RECOVERY_UNIT_PATH" "$FIRST_BOOT_UNIT_PATH"; do + if [[ -e "$unit_path" ]]; then + rm -f "$unit_path" + log_success "Removed systemd unit: $unit_path" + else + log_info "Systemd unit not found: $unit_path" + fi + done systemctl daemon-reload 2>/dev/null || true return 0 @@ -110,13 +136,49 @@ remove_directories() { } remove_binary() { - log_info "Removing binary..." + local root + + log_info "Removing binaries..." - if [[ -f "$INSTALL_DIR/aks-flex-node" ]]; then - rm -f "$INSTALL_DIR/aks-flex-node" - log_success "Removed binary: $INSTALL_DIR/aks-flex-node" + for root in "$HOST_ROOT" "$LEGACY_ROOT"; do + # -L as well as -e: once the agent has run this is a symlink into the managed layout, and a + # dangling one still has to go. + if [[ -e "$root/bin/aks-flex-node" || -L "$root/bin/aks-flex-node" ]]; then + rm -f "$root/bin/aks-flex-node" + log_success "Removed binary: $root/bin/aks-flex-node" + fi + + # Reset keeps the managed blue/green layout so a reinstall can reuse it; uninstall removes it. + if [[ -d "$root/lib/aks-flex-node" ]]; then + rm -rf "$root/lib/aks-flex-node" + log_success "Removed managed binaries: $root/lib/aks-flex-node" + fi + done + + remove_host_root +} + +# remove_host_root removes the host root once nothing is left in it, or the link to the legacy root +# on a host installed by an older release. A link somewhere else is not the agent's. +remove_host_root() { + local dir + + if [[ -L "$HOST_ROOT" ]]; then + if [[ "$(readlink -- "$HOST_ROOT")" == "$LEGACY_ROOT" ]]; then + rm -f -- "$HOST_ROOT" + log_success "Removed link: $HOST_ROOT" + fi + return 0 + fi + + [[ -d "$HOST_ROOT" ]] || return 0 + for dir in "$HOST_ROOT/lib" "$HOST_ROOT/bin" "$HOST_ROOT/libexec" "$HOST_ROOT"; do + rmdir -- "$dir" 2>/dev/null || true + done + if [[ -e "$HOST_ROOT" ]]; then + log_info "Kept $HOST_ROOT: it still holds files" else - log_info "Binary not found: $INSTALL_DIR/aks-flex-node" + log_success "Removed directory: $HOST_ROOT" fi } @@ -143,6 +205,8 @@ main() { exit 1 fi + find_install_dir + # Confirm uninstall confirm_uninstall "${1:-}" @@ -158,5 +222,6 @@ main() { show_completion_message } -# Run main function -main "$@" +if [[ "${BASH_SOURCE[0]:-$0}" == "$0" ]]; then + main "$@" +fi diff --git a/scripts/uninstall_test.sh b/scripts/uninstall_test.sh new file mode 100755 index 00000000..8fe213fc --- /dev/null +++ b/scripts/uninstall_test.sh @@ -0,0 +1,100 @@ +#!/bin/bash + +set -euo pipefail + +REPO_ROOT=$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd) +SCRIPT="$REPO_ROOT/scripts/uninstall.sh" +WORK_DIR=$(mktemp -d) +trap 'rm -rf "$WORK_DIR"' EXIT + +fail() { + printf 'uninstall_test: %s\n' "$*" >&2 + exit 1 +} + +bash -n "$SCRIPT" + +# shellcheck source=scripts/uninstall.sh +source "$SCRIPT" +HOST_ROOT="$WORK_DIR/opt/unbounded" +LEGACY_ROOT="$WORK_DIR/usr/local" + +# populate lays out a release's binaries under a root: the compatibility link, dangling as reset +# can leave it, and the managed layout reset keeps. +populate() { + local root="$1" + mkdir -p "$root/bin" "$root/lib/aks-flex-node" + ln -s "$root/lib/aks-flex-node/aks-flex-node-current" "$root/bin/aks-flex-node" + touch "$root/lib/aks-flex-node/aks-flex-node-blue" +} + +assert_removed() { + local root="$1" + [[ ! -e "$root/bin/aks-flex-node" && ! -L "$root/bin/aks-flex-node" ]] || fail "binary link was left under $root" + [[ ! -e "$root/lib/aks-flex-node" ]] || fail "managed binary layout was left under $root" +} + +# reset runs the binary from whichever root holds it. +populate "$LEGACY_ROOT" +printf '#!/bin/sh\n' > "$LEGACY_ROOT/lib/aks-flex-node/aks-flex-node-current" +chmod 0755 "$LEGACY_ROOT/lib/aks-flex-node/aks-flex-node-current" +find_install_dir +[[ "$INSTALL_DIR" == "$LEGACY_ROOT/bin" ]] || fail "reset would not run the binary under the legacy root: $INSTALL_DIR" +rm -rf "$LEGACY_ROOT" + +# A host installed under the host root: its binaries go, and the host root with them once empty. +populate "$HOST_ROOT" +mkdir -p "$HOST_ROOT/libexec" +remove_binary >/dev/null +assert_removed "$HOST_ROOT" +[[ ! -e "$HOST_ROOT" ]] || fail "empty host root was left: $(find "$HOST_ROOT" | tr '\n' ' ')" + +# The legacy root is swept as well, because uninstalling does not depend on the host root having +# been migrated, and a host root that still holds something else is kept. +populate "$LEGACY_ROOT" +mkdir -p "$HOST_ROOT/bin" +touch "$HOST_ROOT/bin/someone-else" +remove_binary >/dev/null +assert_removed "$LEGACY_ROOT" +[[ -e "$HOST_ROOT/bin/someone-else" ]] || fail "a host root holding other files was removed" +rm -rf "$HOST_ROOT" + +# On a migrated host the host root is a link to the legacy root. The files are removed through it, +# and then the link. +populate "$LEGACY_ROOT" +mkdir -p "$(dirname "$HOST_ROOT")" +ln -s "$LEGACY_ROOT" "$HOST_ROOT" +remove_binary >/dev/null +assert_removed "$LEGACY_ROOT" +[[ ! -e "$HOST_ROOT" && ! -L "$HOST_ROOT" ]] || fail "the host root link was left" + +# A link to anywhere else is not the agent's. +mkdir -p "$WORK_DIR/elsewhere" +ln -s "$WORK_DIR/elsewhere" "$HOST_ROOT" +remove_binary >/dev/null +[[ -L "$HOST_ROOT" ]] || fail "a host root link to somewhere else was removed" +rm "$HOST_ROOT" + +# A second run over a clean host succeeds. +remove_binary >/dev/null || fail "remove_binary is not idempotent" + +# Without a binary to run reset, run_reset removes the units itself, including the first-boot unit +# from an Ignition install. +SYSTEMCTL_CALLS="$WORK_DIR/systemctl-calls" +systemctl() { printf '%s\n' "$*" >>"$SYSTEMCTL_CALLS"; } +INSTALL_DIR="$WORK_DIR/no-binary/bin" +SERVICE_UNIT_PATH="$WORK_DIR/systemd/$SERVICE_UNIT" +RECOVERY_UNIT_PATH="$WORK_DIR/systemd/aks-flex-node-agent-recovery.service" +FIRST_BOOT_UNIT_PATH="$WORK_DIR/systemd/$FIRST_BOOT_UNIT" +mkdir -p "$WORK_DIR/systemd" +touch "$SERVICE_UNIT_PATH" "$RECOVERY_UNIT_PATH" "$FIRST_BOOT_UNIT_PATH" + +run_reset >/dev/null +for unit_path in "$SERVICE_UNIT_PATH" "$RECOVERY_UNIT_PATH" "$FIRST_BOOT_UNIT_PATH"; do + [[ ! -e "$unit_path" ]] || fail "unit was left: $unit_path" +done +grep -Fxq "disable --now aks-flex-node-bootstrap.service" "$SYSTEMCTL_CALLS" || + fail "first-boot unit was not stopped: $(tr '\n' ';' <"$SYSTEMCTL_CALLS")" +unset -f systemctl + +printf 'uninstall_test: ok\n'