From de41cea78c0996e5a824f587e12232a6c5e7326b Mon Sep 17 00:00:00 2001 From: Alice Frosi Date: Wed, 23 Sep 2026 06:19:58 +0000 Subject: [PATCH 1/4] make: add upgrade testing support Add variables and targets to support operator upgrade e2e testing. RELEASED_OPERATOR_TAG and RELEASED_OPERATOR_IMG control which released version to test against. The push-released-operator-image target pulls the released image and pushes it to the bink registry. deploy-bink runs this automatically when RELEASED_OPERATOR_IMG is set. Example usage: make deploy-bink RELEASED_OPERATOR_IMG=ghcr.io/bootc-dev/bootc-operator:v0.1.0 make e2e RUN=TestOperatorUpgrade RELEASED_OPERATOR_TAG=v0.1.0 \ RELEASED_OPERATOR_IMG=ghcr.io/bootc-dev/bootc-operator:v0.1.0 V=1 Assisted-by: AI Signed-off-by: Alice Frosi --- Makefile | 21 ++++++++++++++++++++- 1 file changed, 20 insertions(+), 1 deletion(-) diff --git a/Makefile b/Makefile index 1f4e984..26b298c 100644 --- a/Makefile +++ b/Makefile @@ -17,6 +17,14 @@ BINK_NODE_DISK_IMAGE ?= ghcr.io/bootc-dev/bink/node:v$(DEFAULT_KUBE_MINOR)-fedor BINK_LOCAL_REGISTRY_NODE_IMAGE ?= registry.cluster.local:5000/node E2E_REGISTRY_USER ?= e2e-user E2E_REGISTRY_PASSWORD ?= e2e-password +# Released operator for upgrade testing (set to run TestOperatorUpgrade). +# RELEASED_OPERATOR_TAG: GitHub release tag to download install.yaml from. +# RELEASED_OPERATOR_IMG: container image to use (defaults to upstream, override for downstream). +# Example: RELEASED_OPERATOR_TAG=v0.1.0 +# RELEASED_OPERATOR_IMG=ghcr.io/bootc-dev/bootc-operator:v0.1.0 +RELEASED_OPERATOR_TAG ?= +RELEASED_OPERATOR_IMG ?= +IMG_BINK_RELEASED ?= registry.cluster.local:5000/bootc-operator-released:latest # YEAR defines the year value used for substituting the YEAR placeholder in the boilerplate header. YEAR ?= $(shell date +%Y) @@ -106,6 +114,8 @@ e2e: ## Run e2e tests (requires: make deploy-bink). V=1 for verbose. RUN= E2E_NODE_IMAGE_UPDATE_DIGEST=$$(skopeo inspect --tls-verify=false docker://localhost:5000/node:update | jq -r '.Digest') \ E2E_NODE_IMAGE_UPDATE2_DIGEST=$$(skopeo inspect --tls-verify=false docker://localhost:5000/node:update2 | jq -r '.Digest') \ E2E_REGISTRY_USER=$(E2E_REGISTRY_USER) E2E_REGISTRY_PASSWORD=$(E2E_REGISTRY_PASSWORD) \ + $(if $(RELEASED_OPERATOR_TAG),E2E_OPERATOR_RELEASE_TAG=$(RELEASED_OPERATOR_TAG)) \ + $(if $(RELEASED_OPERATOR_IMG),E2E_OPERATOR_RELEASED_IMG=$(IMG_BINK_RELEASED)) \ go test -timeout 40m -count=1 $(if $(V),-v) $(if $(RUN),-run $(RUN)) . # EKS e2e settings @@ -167,6 +177,15 @@ build-update-image: ## Build derived node images for update testing and push to podman build -t localhost:5000/node:update2 -f - . podman push --tls-verify=false localhost:5000/node:update2 +.PHONY: push-released-operator-image +push-released-operator-image: ## Pull released operator image and push to bink registry for upgrade testing. + @if [ -z "$(RELEASED_OPERATOR_IMG)" ]; then \ + echo "Error: RELEASED_OPERATOR_IMG must be set (e.g. ghcr.io/bootc-dev/bootc-operator:v0.1.0)"; \ + exit 1; \ + fi + podman pull $(RELEASED_OPERATOR_IMG) + podman push --tls-verify=false $(RELEASED_OPERATOR_IMG) localhost:5000/bootc-operator-released:latest + ##@ Deployment ifndef ignore-not-found @@ -219,7 +238,7 @@ start-bink: seed-node-image ## Start a bink cluster (idempotent). kubectl --kubeconfig $(KUBECONFIG_BINK) wait --for=condition=Ready node/controller --timeout=5m .PHONY: deploy-bink -deploy-bink: start-bink build-update-image kustomize ## Deploy to a bink cluster (requires: buildimg). +deploy-bink: start-bink build-update-image $(if $(RELEASED_OPERATOR_IMG),push-released-operator-image) kustomize ## Deploy to a bink cluster (requires: buildimg). podman push --tls-verify=false $(IMG) localhost:5000/bootc-operator-e2e:latest # On re-deploy, restart the rollout to force a re-pull of the :latest tag. # On fresh deploy, skip the restart -- the pod is already pulling the correct image. From d36300194a607e2d869e1ce8c618ed8ced0f87b6 Mon Sep 17 00:00:00 2001 From: Alice Frosi Date: Wed, 23 Sep 2026 06:20:05 +0000 Subject: [PATCH 2/4] gha: run operator upgrade test in CI Set RELEASED_OPERATOR_TAG to v0.1.0 and pass the released operator image to deploy-bink and e2e targets. The upgrade test is skipped when the env vars are empty, so this is a no-op until the test file is added. Assisted-by: AI Signed-off-by: Alice Frosi --- .github/workflows/ci.yaml | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/.github/workflows/ci.yaml b/.github/workflows/ci.yaml index 86aa8a0..4b64722 100644 --- a/.github/workflows/ci.yaml +++ b/.github/workflows/ci.yaml @@ -9,6 +9,7 @@ on: env: BINK_COMMIT: e7d4574fdc610a9fd9bd067aff984516ec7c507d + RELEASED_OPERATOR_TAG: v0.1.1 permissions: {} @@ -197,6 +198,8 @@ jobs: - name: Deploy to bink cluster run: make deploy-bink + env: + RELEASED_OPERATOR_IMG: ghcr.io/${{ github.repository }}:${{ env.RELEASED_OPERATOR_TAG }} - name: Gather deploy logs if: failure() @@ -204,6 +207,8 @@ jobs: - name: Run e2e tests run: make e2e V=1 + env: + RELEASED_OPERATOR_IMG: ghcr.io/${{ github.repository }}:${{ env.RELEASED_OPERATOR_TAG }} - name: Upload logs if: always() From b2df6ec450b205827a9d09e21f618a94ef3c17a9 Mon Sep 17 00:00:00 2001 From: Alice Frosi Date: Wed, 23 Sep 2026 06:20:15 +0000 Subject: [PATCH 3/4] e2e: add operator upgrade test Test the upgrade path by downloading the released install.yaml, applying it to install the released operator, then re-applying the current manifests (including CRDs) on top without deleting. This verifies that CRD schema changes, RBAC updates, and deployment spec changes apply cleanly over a running operator. Register apiextensionsv1 in the e2e client scheme so the test can manage CRD resources directly. Assisted-by: AI Signed-off-by: Alice Frosi --- test/e2e/e2eutil/env.go | 4 + test/e2e/upgrade_test.go | 442 +++++++++++++++++++++++++++++++++++++++ 2 files changed, 446 insertions(+) create mode 100644 test/e2e/upgrade_test.go diff --git a/test/e2e/e2eutil/env.go b/test/e2e/e2eutil/env.go index f51143d..b235b19 100644 --- a/test/e2e/e2eutil/env.go +++ b/test/e2e/e2eutil/env.go @@ -20,6 +20,7 @@ import ( "github.com/google/go-containerregistry/pkg/v1/remote" . "github.com/onsi/gomega" //nolint:staticcheck corev1 "k8s.io/api/core/v1" + apiextensionsv1 "k8s.io/apiextensions-apiserver/pkg/apis/apiextensions/v1" "k8s.io/client-go/kubernetes/scheme" "k8s.io/client-go/tools/clientcmd" "sigs.k8s.io/controller-runtime/pkg/client" @@ -424,6 +425,9 @@ func buildClient(t *testing.T, kubeconfigPath string) client.Client { if err := bootcv1alpha1.AddToScheme(scheme.Scheme); err != nil { t.Fatalf("adding bootc scheme: %v", err) } + if err := apiextensionsv1.AddToScheme(scheme.Scheme); err != nil { + t.Fatalf("adding apiextensions scheme: %v", err) + } cfg, err := clientcmd.BuildConfigFromFlags("", kubeconfigPath) if err != nil { diff --git a/test/e2e/upgrade_test.go b/test/e2e/upgrade_test.go new file mode 100644 index 0000000..d4bf9e3 --- /dev/null +++ b/test/e2e/upgrade_test.go @@ -0,0 +1,442 @@ +// SPDX-License-Identifier: Apache-2.0 + +package e2e + +import ( + "bytes" + "context" + "fmt" + "io" + "net/http" + "os" + "os/exec" + "path/filepath" + "testing" + "time" + + . "github.com/onsi/gomega" + "github.com/onsi/gomega/types" + appsv1 "k8s.io/api/apps/v1" + corev1 "k8s.io/api/core/v1" + apiextensionsv1 "k8s.io/apiextensions-apiserver/pkg/apis/apiextensions/v1" + apierrors "k8s.io/apimachinery/pkg/api/errors" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + utilyaml "k8s.io/apimachinery/pkg/util/yaml" + "sigs.k8s.io/controller-runtime/pkg/client" + sigsyaml "sigs.k8s.io/yaml" + + bootcv1alpha1 "github.com/bootc-dev/bootc-operator/api/v1alpha1" + "github.com/bootc-dev/bootc-operator/test/e2e/e2eutil" + testutil "github.com/bootc-dev/bootc-operator/test/util" +) + +const ( + releaseManifestURL = "https://github.com/bootc-dev/bootc-operator/releases/download/%s/install.yaml" +) + +var operatorCRDNames = []string{ + "bootcnodepools.node.bootc.dev", + "bootcnodes.node.bootc.dev", +} + +// TestOperatorUpgrade installs the released version of the operator +// from its published manifest, verifies basic functionality, then +// upgrades by re-applying the current manifests (including CRDs) on +// top and verifies the operator still works. +func TestOperatorUpgrade(t *testing.T) { + e2eutil.Providers(t, "bink") + + releaseTag := os.Getenv("E2E_OPERATOR_RELEASE_TAG") + if releaseTag == "" { + t.Skip("E2E_OPERATOR_RELEASE_TAG not set") + } + releasedImg := os.Getenv("E2E_OPERATOR_RELEASED_IMG") + if releasedImg == "" { + t.Skip("E2E_OPERATOR_RELEASED_IMG not set") + } + + g := NewWithT(t) + g.SetDefaultEventuallyTimeout(pollTimeout) + g.SetDefaultEventuallyPollingInterval(pollInterval) + + env := e2eutil.New(t) + ctx := context.Background() + + // Capture the current operator image before we replace it. + var origDeploy appsv1.Deployment + g.Expect(env.Client.Get(ctx, operatorDeployKey(), &origDeploy)).To(Succeed()) + currentImg := containerImage(t, origDeploy.Spec.Template.Spec.Containers, "manager") + t.Logf("Current operator image: %s", currentImg) + t.Logf("Released operator tag: %s, image: %s", releaseTag, releasedImg) + + // Phase 1: Delete the current operator and install the released + // version from its published manifest. + installReleasedOperator(t, g, ctx, env, releaseTag, releasedImg) + + t.Cleanup(func() { + t.Logf("Restoring operator to current version...") + applyCurrentManifests(t, currentImg) + waitForOperatorReady(t, g, ctx, env.Client) + t.Logf("Operator restored") + }) + + // Phase 2: Verify the released version works. + nodeName := env.AddNode(t) + + pool := env.NewPool("upgrade", env.NodeImageDigestedPullSpec()) + g.Expect(env.Client.Create(ctx, pool)).To(Succeed()) + + waitForNodeIdle(t, g, ctx, env.Client, nodeName, 3*time.Minute) + + t.Logf("Node %q is Idle with released operator", nodeName) + + // Phase 3: Upgrade by applying the current manifests on top. + applyCurrentManifests(t, currentImg) + waitForOperatorReady(t, g, ctx, env.Client) + + t.Logf("Upgraded operator to current version via manifest apply") + + // Verify the pre-existing pool and node survived the upgrade. + waitForNodeIdle(t, g, ctx, env.Client, nodeName, 3*time.Minute) + g.Eventually(fetchPoolStatus(ctx, env.Client, pool)). + Should(poolAllUpdated(1, env.NodeImageDigest())) + + t.Logf("Pool and node survived the upgrade in expected state") + + // Phase 4: Verify the operator still works after upgrade by + // triggering a node image update and verifying the full update + // lifecycle completes. + updateRef := env.NodeImageUpdateDigestedPullSpec() + + modified := pool.DeepCopy() + modified.Spec.Image.Ref = updateRef + g.Expect(env.Client.Patch(ctx, modified, client.MergeFrom(pool))).To(Succeed()) + *pool = *modified + + t.Logf("Patched pool to update image %s", updateRef) + + waitForNodeIdle(t, g, ctx, env.Client, nodeName, 5*time.Minute, + HaveField("Booted", HaveField("ImageDigest", env.NodeImageUpdateDigest())), + ) + + t.Logf("Node %q completed update with upgraded operator", nodeName) + + g.Eventually(fetchPoolStatus(ctx, env.Client, pool)). + Should(poolAllUpdated(1, env.NodeImageUpdateDigest())) +} + +// installReleasedOperator deletes the current operator, downloads the +// released install manifest, patches the image reference, and applies +// it. +func installReleasedOperator( + t *testing.T, + g Gomega, + ctx context.Context, + env *e2eutil.Env, + releaseTag, releasedImg string, +) { + t.Helper() + + deleteCurrentOperator(t, g, ctx, env) + + manifest := downloadReleaseManifest(t, releaseTag) + patchedManifest := patchManifestImage(t, manifest, releasedImg) + + kubectlApply(t, patchedManifest) + waitForOperatorReady(t, g, ctx, env.Client) + + t.Logf("Released operator %s installed from manifest", releaseTag) +} + +// deleteCurrentOperator deletes custom resources (while the operator +// is still running so it can handle finalizer removal), then removes +// the operator Deployment, DaemonSet, and CRDs. +func deleteCurrentOperator( + t *testing.T, + g Gomega, + ctx context.Context, + env *e2eutil.Env, +) { + t.Helper() + + // Delete CRs first while the operator is running so it can + // remove finalizers. If CRs with finalizers remain when the + // controller is gone, CRD deletion hangs indefinitely. + t.Logf("Deleting custom resources...") + g.Expect(env.Client.DeleteAllOf(ctx, &bootcv1alpha1.BootcNodePool{})).To(Succeed()) + g.Expect(env.Client.DeleteAllOf(ctx, &bootcv1alpha1.BootcNode{})).To(Succeed()) + + g.Eventually(func(g Gomega) { + var pools bootcv1alpha1.BootcNodePoolList + g.Expect(env.Client.List(ctx, &pools)).To(Succeed()) + g.Expect(pools.Items).To(BeEmpty()) + var nodes bootcv1alpha1.BootcNodeList + g.Expect(env.Client.List(ctx, &nodes)).To(Succeed()) + g.Expect(nodes.Items).To(BeEmpty()) + }).WithTimeout(2*time.Minute).Should(Succeed(), + "expected all custom resources to be deleted") + + var deploy appsv1.Deployment + g.Expect(env.Client.Get(ctx, operatorDeployKey(), &deploy)).To(Succeed()) + var ds appsv1.DaemonSet + g.Expect(env.Client.Get(ctx, operatorDaemonSetKey(), &ds)).To(Succeed()) + + t.Logf("Deleting operator deployment, daemonset, and CRDs...") + g.Expect(env.Client.Delete(ctx, &deploy)).To(Succeed()) + g.Expect(env.Client.Delete(ctx, &ds)).To(Succeed()) + + for _, crdName := range operatorCRDNames { + var crd apiextensionsv1.CustomResourceDefinition + if err := env.Client.Get(ctx, client.ObjectKey{Name: crdName}, &crd); err == nil { + g.Expect(env.Client.Delete(ctx, &crd)).To(Succeed()) + } + } + + g.Eventually(func(g Gomega) { + var d appsv1.Deployment + err := env.Client.Get(ctx, operatorDeployKey(), &d) + g.Expect(apierrors.IsNotFound(err)).To(BeTrue()) + var dset appsv1.DaemonSet + err = env.Client.Get(ctx, operatorDaemonSetKey(), &dset) + g.Expect(apierrors.IsNotFound(err)).To(BeTrue()) + for _, crdName := range operatorCRDNames { + var crd apiextensionsv1.CustomResourceDefinition + err = env.Client.Get(ctx, client.ObjectKey{Name: crdName}, &crd) + g.Expect(apierrors.IsNotFound(err)).To(BeTrue()) + } + }).WithTimeout(2*time.Minute).Should(Succeed(), + "expected operator resources to be deleted") + + t.Logf("Operator deployment, daemonset, and CRDs deleted") +} + +// downloadReleaseManifest fetches install.yaml from a GitHub release. +func downloadReleaseManifest(t *testing.T, tag string) []byte { + t.Helper() + + url := fmt.Sprintf(releaseManifestURL, tag) + t.Logf("Downloading release manifest from %s", url) + + resp, err := http.Get(url) //nolint:gosec + if err != nil { + t.Fatalf("downloading release manifest: %v", err) + } + defer func() { _ = resp.Body.Close() }() + + if resp.StatusCode != http.StatusOK { + t.Fatalf("downloading release manifest: HTTP %d", resp.StatusCode) + } + + data, err := io.ReadAll(resp.Body) + if err != nil { + t.Fatalf("reading release manifest body: %v", err) + } + + return data +} + +// patchManifestImage replaces container image references in +// Deployment (manager) and DaemonSet (daemon) documents. +func patchManifestImage(t *testing.T, manifest []byte, img string) []byte { + t.Helper() + + var out bytes.Buffer + reader := utilyaml.NewDocumentDecoder(io.NopCloser(bytes.NewReader(manifest))) + defer func() { _ = reader.Close() }() + + first := true + for { + buf := make([]byte, len(manifest)+256) + n, err := reader.Read(buf) + if err != nil { + if err == io.EOF { + break + } + t.Fatalf("reading YAML document: %v", err) + } + doc := buf[:n] + + var obj map[string]interface{} + if err := sigsyaml.Unmarshal(doc, &obj); err != nil { + t.Fatalf("unmarshaling YAML document: %v", err) + } + + kind, _ := obj["kind"].(string) + switch kind { + case "Deployment": + setContainerImage(t, obj, "manager", img) + case "DaemonSet": + setContainerImage(t, obj, "daemon", img) + } + + patched, err := sigsyaml.Marshal(obj) + if err != nil { + t.Fatalf("marshaling patched %s: %v", kind, err) + } + + if !first { + out.WriteString("---\n") + } + out.Write(patched) + first = false + } + + return out.Bytes() +} + +// setContainerImage patches the image field of a named container in a +// Deployment or DaemonSet manifest map. +func setContainerImage(t *testing.T, obj map[string]interface{}, containerName, img string) { + t.Helper() + + spec, _ := obj["spec"].(map[string]interface{}) + template, _ := spec["template"].(map[string]interface{}) + podSpec, _ := template["spec"].(map[string]interface{}) + containers, _ := podSpec["containers"].([]interface{}) + + for _, c := range containers { + container, _ := c.(map[string]interface{}) + if name, _ := container["name"].(string); name == containerName { + container["image"] = img + return + } + } + t.Fatalf("container %q not found in %s", containerName, obj["kind"]) +} + +// kubectlApply applies the given manifest bytes via kubectl. +func kubectlApply(t *testing.T, manifest []byte) { + t.Helper() + + kubeconfigPath := os.Getenv("KUBECONFIG") + args := []string{"apply", "--server-side", "-f", "-"} + if kubeconfigPath != "" { + args = append([]string{"--kubeconfig", kubeconfigPath}, args...) + } + + cmd := exec.Command("kubectl", args...) + cmd.Stdin = bytes.NewReader(manifest) + + out, err := cmd.CombinedOutput() + if err != nil { + t.Fatalf("kubectl apply failed: %v\n%s", err, out) + } + + t.Logf("kubectl apply:\n%s", out) +} + +// applyCurrentManifests runs the equivalent of "make deploy" for the +// current version: kustomize build + image patch + kubectl apply. +func applyCurrentManifests(t *testing.T, img string) { + t.Helper() + + repoRoot := findRepoRoot(t) + + kustomize := filepath.Join(repoRoot, "bin", "kustomize") + if _, err := os.Stat(kustomize); err != nil { + kustomize = "kustomize" + } + + cmd := exec.Command(kustomize, "build", filepath.Join(repoRoot, "config", "default")) + manifest, err := cmd.CombinedOutput() + if err != nil { + t.Fatalf("kustomize build failed: %v\n%s", err, manifest) + } + + manifest = patchManifestImage(t, manifest, img) + kubectlApply(t, manifest) +} + +// findRepoRoot walks up from the test directory to find the repository +// root (containing go.mod). +func findRepoRoot(t *testing.T) string { + t.Helper() + + dir, err := os.Getwd() + if err != nil { + t.Fatalf("getting working directory: %v", err) + } + + for { + if _, err := os.Stat(filepath.Join(dir, "go.mod")); err == nil { + return dir + } + parent := filepath.Dir(dir) + if parent == dir { + t.Fatal("could not find repository root (go.mod)") + } + dir = parent + } +} + +// waitForOperatorReady polls until the operator deployment has one available and up-to-date replica. +func waitForOperatorReady( + t *testing.T, + g Gomega, + ctx context.Context, + c client.Client, +) { + t.Helper() + g.Eventually(func(g Gomega) { + var d appsv1.Deployment + g.Expect(c.Get(ctx, operatorDeployKey(), &d)).To(Succeed()) + g.Expect(d.Status.Replicas).To(Equal(int32(1))) + g.Expect(d.Status.UpdatedReplicas).To(Equal(int32(1))) + g.Expect(d.Status.AvailableReplicas).To(Equal(int32(1))) + }).WithTimeout(3*time.Minute).Should(Succeed(), + "expected operator deployment to be ready") +} + +func waitForNodeIdle( + t *testing.T, + g Gomega, + ctx context.Context, + c client.Client, + nodeName string, + timeout time.Duration, + extraMatchers ...types.GomegaMatcher, +) { + t.Helper() + matchers := []types.GomegaMatcher{ + HaveField("Booted", Not(BeNil())), + HaveField("Conditions", ContainElement(And( + HaveField("Type", bootcv1alpha1.NodeIdle), + HaveField("Status", metav1.ConditionTrue), + HaveField("Reason", bootcv1alpha1.NodeReasonIdle), + ))), + } + matchers = append(matchers, extraMatchers...) + g.Eventually(func() (bootcv1alpha1.BootcNodeStatus, error) { + var bn bootcv1alpha1.BootcNode + err := c.Get(ctx, client.ObjectKey{Name: nodeName}, &bn) + return bn.Status, err + }).WithTimeout(timeout).Should(And(matchers...)) +} + +// operatorDeployKey returns the namespaced name of the operator Deployment. +func operatorDeployKey() client.ObjectKey { + return client.ObjectKey{ + Namespace: testutil.OperatorNamespaceName, + Name: "bootc-operator-controller-manager", + } +} + +// operatorDaemonSetKey returns the namespaced name of the operator DaemonSet. +func operatorDaemonSetKey() client.ObjectKey { + return client.ObjectKey{ + Namespace: testutil.OperatorNamespaceName, + Name: "bootc-operator-daemon", + } +} + +// containerImage returns the image of the named container, or fails the test if not found. +func containerImage(t *testing.T, containers []corev1.Container, name string) string { + t.Helper() + for _, c := range containers { + if c.Name == name { + return c.Image + } + } + t.Fatalf("container %q not found in pod spec", name) + return "" +} From eb4a64ac80004151bf7c0eaeadcb4d907b8b079c Mon Sep 17 00:00:00 2001 From: Alice Frosi Date: Thu, 24 Sep 2026 06:01:02 +0000 Subject: [PATCH 4/4] e2e: extract WaitForNodeIdle into testutil Move the idle-wait polling pattern into a shared testutil.WaitForNodeIdle helper and replace all 14 occurrences across both test files. Assisted-by: AI Signed-off-by: Alice Frosi --- test/e2e/bootcnode_test.go | 222 ++++++------------------------------- test/e2e/upgrade_test.go | 34 +----- test/util/wait.go | 45 ++++++++ 3 files changed, 83 insertions(+), 218 deletions(-) create mode 100644 test/util/wait.go diff --git a/test/e2e/bootcnode_test.go b/test/e2e/bootcnode_test.go index e6e765d..d49a6fc 100644 --- a/test/e2e/bootcnode_test.go +++ b/test/e2e/bootcnode_test.go @@ -86,22 +86,12 @@ func TestControllerMembership(t *testing.T) { HaveField("Status.Phase", corev1.PodRunning), )), "expected exactly one running daemon pod on %s", nodeName) - g.Eventually(func() (bootcv1alpha1.BootcNodeStatus, error) { - var bn bootcv1alpha1.BootcNode - err := env.Client.Get(ctx, client.ObjectKey{Name: nodeName}, &bn) - return bn.Status, err - }).WithTimeout(3 * time.Minute).Should(And( + testutil.WaitForNodeIdle(t, g, ctx, env.Client, nodeName, 3*time.Minute, HaveField("Booted", And( - Not(BeNil()), HaveField("Image", env.NodeImageDigestedPullSpec()), HaveField("ImageDigest", env.NodeImageDigest()), )), - HaveField("Conditions", ContainElement(And( - HaveField("Type", bootcv1alpha1.NodeIdle), - HaveField("Status", metav1.ConditionTrue), - HaveField("Reason", bootcv1alpha1.NodeReasonIdle), - ))), - )) + ) // Verify pool status reflects steady state. g.Eventually(fetchPoolStatus(ctx, env.Client, pool)). @@ -126,18 +116,7 @@ func TestUpdateReboot(t *testing.T) { pool := env.NewPool("workers", env.NodeImageDigestedPullSpec()) g.Expect(env.Client.Create(ctx, pool)).To(Succeed()) - g.Eventually(func() (bootcv1alpha1.BootcNodeStatus, error) { - var bn bootcv1alpha1.BootcNode - err := env.Client.Get(ctx, client.ObjectKey{Name: nodeName}, &bn) - return bn.Status, err - }).WithTimeout(3 * time.Minute).Should(And( - HaveField("Booted", Not(BeNil())), - HaveField("Conditions", ContainElement(And( - HaveField("Type", bootcv1alpha1.NodeIdle), - HaveField("Status", metav1.ConditionTrue), - HaveField("Reason", bootcv1alpha1.NodeReasonIdle), - ))), - )) + testutil.WaitForNodeIdle(t, g, ctx, env.Client, nodeName, 3*time.Minute) t.Logf("Node %q is Idle with original image", nodeName) @@ -224,21 +203,9 @@ func TestUpdateReboot(t *testing.T) { // Phase 4: Wait for Idle with the update digest — proves the full // update lifecycle completed (staging, reboot, boot into new image). - g.Eventually(func() (bootcv1alpha1.BootcNodeStatus, error) { - var bn bootcv1alpha1.BootcNode - err := env.Client.Get(ctx, client.ObjectKey{Name: nodeName}, &bn) - return bn.Status, err - }).WithTimeout(5*time.Minute).Should(And( - HaveField("Booted", And( - Not(BeNil()), - HaveField("ImageDigest", env.NodeImageUpdateDigest()), - )), - HaveField("Conditions", ContainElement(And( - HaveField("Type", bootcv1alpha1.NodeIdle), - HaveField("Status", metav1.ConditionTrue), - HaveField("Reason", bootcv1alpha1.NodeReasonIdle), - ))), - ), "expected node to reach Idle with update image after reboot") + testutil.WaitForNodeIdle(t, g, ctx, env.Client, nodeName, 5*time.Minute, + HaveField("Booted", HaveField("ImageDigest", env.NodeImageUpdateDigest())), + ) t.Logf("Node %q is Idle with update image", nodeName) @@ -280,21 +247,9 @@ func TestUpdateReboot(t *testing.T) { t.Logf("Patched pool to rollback to original image %s", originalRef) // Phase 8: Wait for Idle with the original digest — proves rollback succeeded. - g.Eventually(func() (bootcv1alpha1.BootcNodeStatus, error) { - var bn2 bootcv1alpha1.BootcNode - err := env.Client.Get(ctx, client.ObjectKey{Name: nodeName}, &bn2) - return bn2.Status, err - }).WithTimeout(5*time.Minute).Should(And( - HaveField("Booted", And( - Not(BeNil()), - HaveField("ImageDigest", Equal(env.NodeImageDigest())), - )), - HaveField("Conditions", ContainElement(And( - HaveField("Type", bootcv1alpha1.NodeIdle), - HaveField("Status", metav1.ConditionTrue), - HaveField("Reason", bootcv1alpha1.NodeReasonIdle), - ))), - ), "expected node to reach Idle with original image after rollback") + testutil.WaitForNodeIdle(t, g, ctx, env.Client, nodeName, 5*time.Minute, + HaveField("Booted", HaveField("ImageDigest", Equal(env.NodeImageDigest()))), + ) t.Logf("Node %q successfully rolled back to original image", nodeName) @@ -333,20 +288,9 @@ func TestTagResolution(t *testing.T) { t.Logf("Tag resolved to original digest %s", env.NodeImageDigest()) // Wait for node to reach Idle with the original image. - g.Eventually(func() (bootcv1alpha1.BootcNodeStatus, error) { - var bn bootcv1alpha1.BootcNode - err := env.Client.Get(ctx, client.ObjectKey{Name: nodeName}, &bn) - return bn.Status, err - }).WithTimeout(3 * time.Minute).Should(And( - HaveField("Booted", And( - Not(BeNil()), - HaveField("ImageDigest", Equal(env.NodeImageDigest())), - )), - HaveField("Conditions", ContainElement(And( - HaveField("Type", bootcv1alpha1.NodeIdle), - HaveField("Status", metav1.ConditionTrue), - ))), - )) + testutil.WaitForNodeIdle(t, g, ctx, env.Client, nodeName, 3*time.Minute, + HaveField("Booted", HaveField("ImageDigest", Equal(env.NodeImageDigest()))), + ) t.Logf("Node %q is Idle with original image", nodeName) @@ -391,20 +335,9 @@ func TestTagResolution(t *testing.T) { ))) // Wait for node to reach Idle with the update image. - g.Eventually(func() (bootcv1alpha1.BootcNodeStatus, error) { - var bn bootcv1alpha1.BootcNode - err := env.Client.Get(ctx, client.ObjectKey{Name: nodeName}, &bn) - return bn.Status, err - }).WithTimeout(5 * time.Minute).Should(And( - HaveField("Booted", And( - Not(BeNil()), - HaveField("ImageDigest", Equal(env.NodeImageUpdateDigest())), - )), - HaveField("Conditions", ContainElement(And( - HaveField("Type", bootcv1alpha1.NodeIdle), - HaveField("Status", metav1.ConditionTrue), - ))), - )) + testutil.WaitForNodeIdle(t, g, ctx, env.Client, nodeName, 5*time.Minute, + HaveField("Booted", HaveField("ImageDigest", Equal(env.NodeImageUpdateDigest()))), + ) t.Logf("Node %q is Idle with update image", nodeName) } @@ -431,18 +364,7 @@ func TestMidRolloutImageChange(t *testing.T) { g.Expect(env.Client.Create(ctx, pool)).To(Succeed()) for _, nodeName := range []string{nodeA, nodeB} { - g.Eventually(func() (bootcv1alpha1.BootcNode, error) { - var bn bootcv1alpha1.BootcNode - err := env.Client.Get(ctx, client.ObjectKey{Name: nodeName}, &bn) - return bn, err - }).WithTimeout(3 * time.Minute).Should(SatisfyAll( - HaveField("Status.Booted", Not(BeNil())), - HaveField("Status.Conditions", ContainElement(And( - HaveField("Type", bootcv1alpha1.NodeIdle), - HaveField("Status", metav1.ConditionTrue), - HaveField("Reason", bootcv1alpha1.NodeReasonIdle), - ))), - )) + testutil.WaitForNodeIdle(t, g, ctx, env.Client, nodeName, 3*time.Minute) } t.Logf("Both nodes are Idle with original image") @@ -503,19 +425,9 @@ func TestMidRolloutImageChange(t *testing.T) { // Phase 6: Wait for both nodes to be Idle with the second update image. for _, nodeName := range []string{nodeA, nodeB} { - g.Eventually(func() (bootcv1alpha1.BootcNode, error) { - var bn bootcv1alpha1.BootcNode - err := env.Client.Get(ctx, client.ObjectKey{Name: nodeName}, &bn) - return bn, err - }).WithTimeout(8*time.Minute).Should(SatisfyAll( - HaveField("Status.Booted", Not(BeNil())), - HaveField("Status.Booted.ImageDigest", Equal(env.NodeImageUpdate2Digest())), - HaveField("Status.Conditions", ContainElement(And( - HaveField("Type", bootcv1alpha1.NodeIdle), - HaveField("Status", metav1.ConditionTrue), - HaveField("Reason", bootcv1alpha1.NodeReasonIdle), - ))), - ), "expected node %s to reach Idle with second update image", nodeName) + testutil.WaitForNodeIdle(t, g, ctx, env.Client, nodeName, 8*time.Minute, + HaveField("Booted", HaveField("ImageDigest", Equal(env.NodeImageUpdate2Digest()))), + ) } t.Logf("Both nodes are Idle with second update image") @@ -627,21 +539,12 @@ func TestPauseResume(t *testing.T) { pool := env.NewPool("bnp-pause", env.NodeImageDigestedPullSpec()) g.Expect(env.Client.Create(ctx, pool)).To(Succeed()) - var bn bootcv1alpha1.BootcNode - g.Eventually(func() (bootcv1alpha1.BootcNodeStatus, error) { - err := env.Client.Get(ctx, client.ObjectKey{Name: nodeName}, &bn) - return bn.Status, err - }).WithTimeout(3 * time.Minute).Should(And( - HaveField("Booted", Not(BeNil())), - HaveField("Conditions", ContainElement(And( - HaveField("Type", bootcv1alpha1.NodeIdle), - HaveField("Status", metav1.ConditionTrue), - HaveField("Reason", bootcv1alpha1.NodeReasonIdle), - ))), - )) + testutil.WaitForNodeIdle(t, g, ctx, env.Client, nodeName, 3*time.Minute) t.Logf("Node %q is Idle with original image", nodeName) + var bn bootcv1alpha1.BootcNode + // Phase 2: Patch pool to update image with paused=true. updateRef := env.NodeImageUpdateDigestedPullSpec() @@ -712,20 +615,9 @@ func TestPauseResume(t *testing.T) { // Phase 5: Wait for node to complete the update — proves the full // update lifecycle completed after resume (reboot, boot into new image). - g.Eventually(func() (bootcv1alpha1.BootcNodeStatus, error) { - err := env.Client.Get(ctx, client.ObjectKey{Name: nodeName}, &bn) - return bn.Status, err - }).WithTimeout(5*time.Minute).Should(And( - HaveField("Booted", And( - Not(BeNil()), - HaveField("ImageDigest", Equal(env.NodeImageUpdateDigest())), - )), - HaveField("Conditions", ContainElement(And( - HaveField("Type", bootcv1alpha1.NodeIdle), - HaveField("Status", metav1.ConditionTrue), - HaveField("Reason", bootcv1alpha1.NodeReasonIdle), - ))), - ), "expected node to reach Idle with update image after resume") + testutil.WaitForNodeIdle(t, g, ctx, env.Client, nodeName, 5*time.Minute, + HaveField("Booted", HaveField("ImageDigest", Equal(env.NodeImageUpdateDigest()))), + ) t.Logf("Node %q completed update after resume", nodeName) @@ -752,21 +644,12 @@ func TestNonExistingImage(t *testing.T) { pool := env.NewPool("bnp-noimg", env.NodeImageDigestedPullSpec()) g.Expect(env.Client.Create(ctx, pool)).To(Succeed()) - var bn bootcv1alpha1.BootcNode - g.Eventually(func() (bootcv1alpha1.BootcNodeStatus, error) { - err := env.Client.Get(ctx, client.ObjectKey{Name: nodeName}, &bn) - return bn.Status, err - }).WithTimeout(3 * time.Minute).Should(And( - HaveField("Booted", Not(BeNil())), - HaveField("Conditions", ContainElement(And( - HaveField("Type", bootcv1alpha1.NodeIdle), - HaveField("Status", metav1.ConditionTrue), - HaveField("Reason", bootcv1alpha1.NodeReasonIdle), - ))), - )) + testutil.WaitForNodeIdle(t, g, ctx, env.Client, nodeName, 3*time.Minute) t.Logf("Node %q is Idle with original image", nodeName) + var bn bootcv1alpha1.BootcNode + // Phase 2: Patch pool to update to a non-existing image. nonExistingRef := "localhost:5000/node@sha256:0000000000000000000000000000000000000000000000000000000000000000" @@ -886,21 +769,9 @@ func TestPullSecretAuth(t *testing.T) { // Check Booted.Image (the full ref with manifest digest) rather // than Booted.ImageDigest (the content digest) because they can // differ with remote registries. - g.Eventually(func() (bootcv1alpha1.BootcNodeStatus, error) { - var bn bootcv1alpha1.BootcNode - err := env.Client.Get(ctx, client.ObjectKey{Name: nodeName}, &bn) - return bn.Status, err - }).WithTimeout(5 * time.Minute).Should(And( - HaveField("Booted", And( - Not(BeNil()), - HaveField("Image", Equal(authImageRef)), - )), - HaveField("Conditions", ContainElement(And( - HaveField("Type", bootcv1alpha1.NodeIdle), - HaveField("Status", metav1.ConditionTrue), - HaveField("Reason", bootcv1alpha1.NodeReasonIdle), - ))), - )) + testutil.WaitForNodeIdle(t, g, ctx, env.Client, nodeName, 5*time.Minute, + HaveField("Booted", HaveField("Image", Equal(authImageRef))), + ) t.Logf("Node %q booted into auth-registry image", nodeName) @@ -930,18 +801,7 @@ func TestControllerRecovery(t *testing.T) { pool := env.NewPool("bnp-recovery", env.NodeImageDigestedPullSpec()) g.Expect(env.Client.Create(ctx, pool)).To(Succeed()) - g.Eventually(func() (bootcv1alpha1.BootcNodeStatus, error) { - var bn bootcv1alpha1.BootcNode - err := env.Client.Get(ctx, client.ObjectKey{Name: nodeName}, &bn) - return bn.Status, err - }).WithTimeout(3 * time.Minute).Should(And( - HaveField("Booted", Not(BeNil())), - HaveField("Conditions", ContainElement(And( - HaveField("Type", bootcv1alpha1.NodeIdle), - HaveField("Status", metav1.ConditionTrue), - HaveField("Reason", bootcv1alpha1.NodeReasonIdle), - ))), - )) + testutil.WaitForNodeIdle(t, g, ctx, env.Client, nodeName, 3*time.Minute) t.Logf("Node %q is Idle with original image", nodeName) @@ -1024,21 +884,9 @@ func TestControllerRecovery(t *testing.T) { scaleController(t, env, ctx, 1) t.Logf("Controller restored; waiting for the interrupted rollout to finish") - g.Eventually(func() (bootcv1alpha1.BootcNodeStatus, error) { - var bn bootcv1alpha1.BootcNode - err := env.Client.Get(ctx, client.ObjectKey{Name: nodeName}, &bn) - return bn.Status, err - }).WithTimeout(5*time.Minute).Should(And( - HaveField("Booted", And( - Not(BeNil()), - HaveField("ImageDigest", Equal(env.NodeImageUpdateDigest())), - )), - HaveField("Conditions", ContainElement(And( - HaveField("Type", bootcv1alpha1.NodeIdle), - HaveField("Status", metav1.ConditionTrue), - HaveField("Reason", bootcv1alpha1.NodeReasonIdle), - ))), - ), "expected node to reach Idle with update image after controller recovery") + testutil.WaitForNodeIdle(t, g, ctx, env.Client, nodeName, 5*time.Minute, + HaveField("Booted", HaveField("ImageDigest", Equal(env.NodeImageUpdateDigest()))), + ) t.Logf("Node %q completed the interrupted rollout after controller recovery", nodeName) diff --git a/test/e2e/upgrade_test.go b/test/e2e/upgrade_test.go index d4bf9e3..ec24ea1 100644 --- a/test/e2e/upgrade_test.go +++ b/test/e2e/upgrade_test.go @@ -15,12 +15,10 @@ import ( "time" . "github.com/onsi/gomega" - "github.com/onsi/gomega/types" appsv1 "k8s.io/api/apps/v1" corev1 "k8s.io/api/core/v1" apiextensionsv1 "k8s.io/apiextensions-apiserver/pkg/apis/apiextensions/v1" apierrors "k8s.io/apimachinery/pkg/api/errors" - metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" utilyaml "k8s.io/apimachinery/pkg/util/yaml" "sigs.k8s.io/controller-runtime/pkg/client" sigsyaml "sigs.k8s.io/yaml" @@ -86,7 +84,7 @@ func TestOperatorUpgrade(t *testing.T) { pool := env.NewPool("upgrade", env.NodeImageDigestedPullSpec()) g.Expect(env.Client.Create(ctx, pool)).To(Succeed()) - waitForNodeIdle(t, g, ctx, env.Client, nodeName, 3*time.Minute) + testutil.WaitForNodeIdle(t, g, ctx, env.Client, nodeName, 3*time.Minute) t.Logf("Node %q is Idle with released operator", nodeName) @@ -97,7 +95,7 @@ func TestOperatorUpgrade(t *testing.T) { t.Logf("Upgraded operator to current version via manifest apply") // Verify the pre-existing pool and node survived the upgrade. - waitForNodeIdle(t, g, ctx, env.Client, nodeName, 3*time.Minute) + testutil.WaitForNodeIdle(t, g, ctx, env.Client, nodeName, 3*time.Minute) g.Eventually(fetchPoolStatus(ctx, env.Client, pool)). Should(poolAllUpdated(1, env.NodeImageDigest())) @@ -115,7 +113,7 @@ func TestOperatorUpgrade(t *testing.T) { t.Logf("Patched pool to update image %s", updateRef) - waitForNodeIdle(t, g, ctx, env.Client, nodeName, 5*time.Minute, + testutil.WaitForNodeIdle(t, g, ctx, env.Client, nodeName, 5*time.Minute, HaveField("Booted", HaveField("ImageDigest", env.NodeImageUpdateDigest())), ) @@ -387,32 +385,6 @@ func waitForOperatorReady( "expected operator deployment to be ready") } -func waitForNodeIdle( - t *testing.T, - g Gomega, - ctx context.Context, - c client.Client, - nodeName string, - timeout time.Duration, - extraMatchers ...types.GomegaMatcher, -) { - t.Helper() - matchers := []types.GomegaMatcher{ - HaveField("Booted", Not(BeNil())), - HaveField("Conditions", ContainElement(And( - HaveField("Type", bootcv1alpha1.NodeIdle), - HaveField("Status", metav1.ConditionTrue), - HaveField("Reason", bootcv1alpha1.NodeReasonIdle), - ))), - } - matchers = append(matchers, extraMatchers...) - g.Eventually(func() (bootcv1alpha1.BootcNodeStatus, error) { - var bn bootcv1alpha1.BootcNode - err := c.Get(ctx, client.ObjectKey{Name: nodeName}, &bn) - return bn.Status, err - }).WithTimeout(timeout).Should(And(matchers...)) -} - // operatorDeployKey returns the namespaced name of the operator Deployment. func operatorDeployKey() client.ObjectKey { return client.ObjectKey{ diff --git a/test/util/wait.go b/test/util/wait.go new file mode 100644 index 0000000..a981123 --- /dev/null +++ b/test/util/wait.go @@ -0,0 +1,45 @@ +// SPDX-License-Identifier: Apache-2.0 + +package testutil + +import ( + "context" + "testing" + "time" + + "github.com/onsi/gomega" + "github.com/onsi/gomega/types" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "sigs.k8s.io/controller-runtime/pkg/client" + + bootcv1alpha1 "github.com/bootc-dev/bootc-operator/api/v1alpha1" +) + +// WaitForNodeIdle polls a BootcNode until it reaches Idle state with +// Booted not nil. Extra matchers are ANDed with the base assertions, +// allowing callers to add checks like ImageDigest or Image. +func WaitForNodeIdle( + t *testing.T, + g gomega.Gomega, + ctx context.Context, + c client.Client, + nodeName string, + timeout time.Duration, + extraMatchers ...types.GomegaMatcher, +) { + t.Helper() + matchers := []types.GomegaMatcher{ + gomega.HaveField("Booted", gomega.Not(gomega.BeNil())), + gomega.HaveField("Conditions", gomega.ContainElement(gomega.And( + gomega.HaveField("Type", bootcv1alpha1.NodeIdle), + gomega.HaveField("Status", metav1.ConditionTrue), + gomega.HaveField("Reason", bootcv1alpha1.NodeReasonIdle), + ))), + } + matchers = append(matchers, extraMatchers...) + g.Eventually(func() (bootcv1alpha1.BootcNodeStatus, error) { + var bn bootcv1alpha1.BootcNode + err := c.Get(ctx, client.ObjectKey{Name: nodeName}, &bn) + return bn.Status, err + }).WithTimeout(timeout).Should(gomega.And(matchers...)) +}