diff --git a/cloud/fake/client.go b/cloud/fake/client.go index 0642e49..05fc443 100644 --- a/cloud/fake/client.go +++ b/cloud/fake/client.go @@ -250,11 +250,12 @@ func (c *Client) EnsureBastion(_ context.Context, input cloud.BastionInput) (*cl publicIP := c.publicIPByTags(input.Tags) securityGroup := c.securityGroupByTags(input.Tags) return &cloud.Bastion{ - ServerID: serverEntry.server.ID, - ServerState: serverEntry.server.State, - PublicIPID: idOfPublicIP(publicIP), - PublicIP: ipOfPublicIP(publicIP), - SecurityGroupID: idOfSecurityGroup(securityGroup), + ServerID: serverEntry.server.ID, + ServerState: serverEntry.server.State, + ServerPowerStatus: serverEntry.server.PowerStatus, + PublicIPID: idOfPublicIP(publicIP), + PublicIP: ipOfPublicIP(publicIP), + SecurityGroupID: idOfSecurityGroup(securityGroup), }, nil } } @@ -569,6 +570,16 @@ func (c *Client) ServerHasSecurityGroup(serverID, securityGroupID string) bool { return ok } +// SetServerState sets the state and power status of a fake server (test helper). +func (c *Client) SetServerState(serverID, state, powerStatus string) { + c.mu.Lock() + defer c.mu.Unlock() + if entry, ok := c.servers[serverID]; ok { + entry.server.State = state + entry.server.PowerStatus = powerStatus + } +} + // ServerUserData returns the user-data stored for a fake server. func (c *Client) ServerUserData(serverID string) []byte { c.mu.Lock() diff --git a/cloud/sdk_client.go b/cloud/sdk_client.go index fc02cb2..14755a8 100644 --- a/cloud/sdk_client.go +++ b/cloud/sdk_client.go @@ -290,11 +290,12 @@ func (c *SDKClient) EnsureBastion(ctx context.Context, input BastionInput) (*Bas } return &Bastion{ - ServerID: server.ID, - ServerState: server.State, - PublicIPID: publicIP.ID, - PublicIP: publicIP.IP, - SecurityGroupID: securityGroup.ID, + ServerID: server.ID, + ServerState: server.State, + ServerPowerStatus: server.PowerStatus, + PublicIPID: publicIP.ID, + PublicIP: publicIP.IP, + SecurityGroupID: securityGroup.ID, }, nil } @@ -962,9 +963,10 @@ func (c *SDKClient) serverFromSDK(ctx context.Context, server *iaas.Server) *Ser return nil } out := &Server{ - ID: server.GetId(), - Name: server.GetName(), - State: server.GetStatus(), + ID: server.GetId(), + Name: server.GetName(), + State: server.GetStatus(), + PowerStatus: server.GetPowerStatus(), } nics, err := c.iaasClient.DefaultAPI.ListServerNICs(ctx, c.projectID, c.region, out.ID).Execute() if err == nil { diff --git a/cloud/types.go b/cloud/types.go index ddc5d6d..74a7235 100644 --- a/cloud/types.go +++ b/cloud/types.go @@ -15,11 +15,12 @@ package cloud // Server describes a STACKIT compute instance in provider-neutral terms. type Server struct { - ID string - Name string - State string - ProviderID string - Addresses []Address + ID string + Name string + State string + PowerStatus string + ProviderID string + Addresses []Address } // Address is an IP or DNS endpoint of a Server. @@ -59,11 +60,12 @@ type SecurityGroup struct { // Bastion describes the provider-managed SSH bastion resources. type Bastion struct { - ServerID string - ServerState string - PublicIPID string - PublicIP string - SecurityGroupID string + ServerID string + ServerState string + ServerPowerStatus string + PublicIPID string + PublicIP string + SecurityGroupID string } // CreateServerInput holds all parameters required to create a new VM. diff --git a/controller/controller_test_helpers_test.go b/controller/controller_test_helpers_test.go index d103137..20c70cb 100644 --- a/controller/controller_test_helpers_test.go +++ b/controller/controller_test_helpers_test.go @@ -17,6 +17,7 @@ import ( corev1 "k8s.io/api/core/v1" apierrors "k8s.io/apimachinery/pkg/api/errors" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/client-go/tools/events" "sigs.k8s.io/controller-runtime/pkg/client" "sigs.k8s.io/controller-runtime/pkg/reconcile" @@ -216,3 +217,16 @@ func deleteIfExists(ctx context.Context, obj client.Object) { Expect(apierrors.IsNotFound(err)).To(BeTrue()) } } + +// drainEvents returns all events recorded so far. +func drainEvents(recorder *events.FakeRecorder) []string { + var out []string + for { + select { + case event := <-recorder.Events: + out = append(out, event) + default: + return out + } + } +} diff --git a/controller/server_state.go b/controller/server_state.go new file mode 100644 index 0000000..f3c43b6 --- /dev/null +++ b/controller/server_state.go @@ -0,0 +1,74 @@ +/* +Copyright 2026. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 +*/ + +package controller + +import "fmt" + +type serverStateInfo struct { + reason string + description string + // warn marks states that do not resolve on their own. Their reason must + // differ from the transitional state leading there, because warnings are + // deduplicated by reason. + warn bool +} + +// serverStates maps the documented STACKIT server states other than ACTIVE. +var serverStates = map[string]serverStateInfo{ + "CREATING": {"InstanceStarting", "server is starting", false}, + "STARTING": {"InstanceStarting", "server is starting", false}, + "REBOOT": {"InstanceBusy", "server is temporarily unavailable", false}, + "REBOOTING": {"InstanceBusy", "server is temporarily unavailable", false}, + "REBUILD": {"InstanceBusy", "server is temporarily unavailable", false}, + "REBUILDING": {"InstanceBusy", "server is temporarily unavailable", false}, + "RESIZING": {"InstanceBusy", "server is temporarily unavailable", false}, + "MIGRATING": {"InstanceBusy", "server is temporarily unavailable", false}, + "UPDATING": {"InstanceBusy", "server is temporarily unavailable", false}, + "RESTORING": {"InstanceBusy", "server is temporarily unavailable", false}, + "SNAPSHOTTING": {"InstanceBusy", "server is temporarily unavailable", false}, + "BACKING-UP": {"InstanceBusy", "server is temporarily unavailable", false}, + "STOPPING": {"InstanceStopping", "server is stopping", false}, + "INACTIVE": {"InstanceStopped", "server is stopped", true}, + "DEALLOCATING": {"InstanceDeallocating", "server is being deallocated", false}, + "DEALLOCATED": {"InstanceDeallocated", "server is deallocated", true}, + "PAUSED": {"InstancePaused", "server is paused", true}, + "RESCUING": {"InstanceEnteringRescue", "server is entering rescue mode", false}, + "RESCUE": {"InstanceInRescue", "server is in rescue mode", true}, + "UNRESCUING": {"InstanceLeavingRescue", "server is leaving rescue mode", false}, + "DELETING": {"InstanceDeleting", "server is being deleted", false}, + "DELETED": {"InstanceDeleting", "server is deleted", false}, + "ERROR": {"InstanceFailed", "server failed", true}, +} + +// serverStateCondition maps a server state and power status to a condition +// reason and message. ready is true when the server can be used. +func serverStateCondition(state, powerStatus string) (ready bool, reason, message string, warn bool) { + details := "state " + state + if powerStatus != "" { + details += ", power status " + powerStatus + } + + if state == "" || state == "ACTIVE" { + switch powerStatus { + case "CRASHED": + return false, "InstanceCrashed", fmt.Sprintf("server crashed (%s)", details), true + case "ERROR": + return false, "InstancePowerError", fmt.Sprintf("server power error (%s)", details), true + } + return true, "", "", false + } + + info, ok := serverStates[state] + if !ok { + info = serverStateInfo{"InstanceNotActive", "server is not active", false} + } + return false, info.reason, fmt.Sprintf("%s (%s)", info.description, details), info.warn +} diff --git a/controller/server_state_test.go b/controller/server_state_test.go new file mode 100644 index 0000000..5565f68 --- /dev/null +++ b/controller/server_state_test.go @@ -0,0 +1,56 @@ +/* +Copyright 2026. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 +*/ + +package controller + +import ( + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("serverStateCondition", func() { + DescribeTable("maps server state and power status", + func(state, powerStatus string, wantReady bool, wantReason, wantMessage string, wantWarn bool) { + ready, reason, message, warn := serverStateCondition(state, powerStatus) + Expect(ready).To(Equal(wantReady)) + Expect(reason).To(Equal(wantReason)) + Expect(message).To(Equal(wantMessage)) + Expect(warn).To(Equal(wantWarn)) + }, + Entry("active and running", "ACTIVE", "RUNNING", true, "", "", false), + Entry("unknown state", "", "", true, "", "", false), + Entry("active but crashed", "ACTIVE", "CRASHED", false, "InstanceCrashed", + "server crashed (state ACTIVE, power status CRASHED)", true), + Entry("active with power error", "ACTIVE", "ERROR", false, "InstancePowerError", + "server power error (state ACTIVE, power status ERROR)", true), + Entry("creating", "CREATING", "", false, "InstanceStarting", "server is starting (state CREATING)", false), + Entry("rebooting", "REBOOTING", "RUNNING", false, "InstanceBusy", + "server is temporarily unavailable (state REBOOTING, power status RUNNING)", false), + Entry("stopping", "STOPPING", "RUNNING", false, "InstanceStopping", + "server is stopping (state STOPPING, power status RUNNING)", false), + Entry("stopped", "INACTIVE", "STOPPED", false, "InstanceStopped", + "server is stopped (state INACTIVE, power status STOPPED)", true), + Entry("deallocating", "DEALLOCATING", "STOPPED", false, "InstanceDeallocating", + "server is being deallocated (state DEALLOCATING, power status STOPPED)", false), + Entry("deallocated", "DEALLOCATED", "STOPPED", false, "InstanceDeallocated", + "server is deallocated (state DEALLOCATED, power status STOPPED)", true), + Entry("paused", "PAUSED", "", false, "InstancePaused", "server is paused (state PAUSED)", true), + Entry("entering rescue", "RESCUING", "", false, "InstanceEnteringRescue", + "server is entering rescue mode (state RESCUING)", false), + Entry("leaving rescue", "UNRESCUING", "", false, "InstanceLeavingRescue", + "server is leaving rescue mode (state UNRESCUING)", false), + Entry("rescue", "RESCUE", "RUNNING", false, "InstanceInRescue", + "server is in rescue mode (state RESCUE, power status RUNNING)", true), + Entry("deleting", "DELETING", "", false, "InstanceDeleting", "server is being deleted (state DELETING)", false), + Entry("error", "ERROR", "ERROR", false, "InstanceFailed", "server failed (state ERROR, power status ERROR)", true), + Entry("undocumented state", "SOMETHING", "", false, "InstanceNotActive", + "server is not active (state SOMETHING)", false), + ) +}) diff --git a/controller/stackitcluster_bastion.go b/controller/stackitcluster_bastion.go index 3a3927d..a4268d0 100644 --- a/controller/stackitcluster_bastion.go +++ b/controller/stackitcluster_bastion.go @@ -130,13 +130,16 @@ func (r *StackitClusterReconciler) reconcileBastion( stackitCluster, nil, corev1.EventTypeNormal, "BastionCreated", "Create", "Created bastion %s", bastion.ServerID, ) } - if bastion.ServerState != "" && bastion.ServerState != "ACTIVE" { - clusterScope.SetNotReady( - "Provisioning", - fmt.Sprintf("bastion server state is %s", bastion.ServerState), - infrav1.ClusterBastionReadyCondition, - infrav1.ClusterReadyCondition, - ) + if ready, reason, message, warn := serverStateCondition(bastion.ServerState, bastion.ServerPowerStatus); !ready { + previousReason := "" + if previous := meta.FindStatusCondition(stackitCluster.Status.Conditions, infrav1.ClusterBastionReadyCondition); previous != nil { + previousReason = previous.Reason + } + clusterScope.SetNotReady(reason, "bastion "+message, infrav1.ClusterBastionReadyCondition, infrav1.ClusterReadyCondition) + // Only warn on entering the state, not on every requeue. + if warn && r.Recorder != nil && previousReason != reason { + r.Recorder.Eventf(stackitCluster, nil, corev1.EventTypeWarning, reason, "Reconcile", "Bastion server %s: %s", bastion.ServerID, message) + } return ctrl.Result{RequeueAfter: 15 * time.Second}, false, nil } if bastion.PublicIP == "" { diff --git a/controller/stackitcluster_controller_test.go b/controller/stackitcluster_controller_test.go index 1ca1de0..bd35fec 100644 --- a/controller/stackitcluster_controller_test.go +++ b/controller/stackitcluster_controller_test.go @@ -19,8 +19,10 @@ import ( . "github.com/onsi/gomega" corev1 "k8s.io/api/core/v1" apierrors "k8s.io/apimachinery/pkg/api/errors" + "k8s.io/apimachinery/pkg/api/meta" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/types" + "k8s.io/client-go/tools/events" clusterv1 "sigs.k8s.io/cluster-api/api/core/v1beta2" "sigs.k8s.io/controller-runtime/pkg/reconcile" @@ -138,6 +140,30 @@ var _ = Describe("StackitCluster Controller", func() { expectCondition(got.Status.Conditions, infrav1.ClusterBastionReadyCondition, metav1.ConditionTrue, "Available") }) + It("reports a failed bastion server with its own reason and warns", func() { + got := &infrav1.StackitCluster{} + Expect(k8sClient.Get(ctx, stackitKey, got)).To(Succeed()) + got.Spec.Bastion = validBastionSpec() + Expect(k8sClient.Update(ctx, got)).To(Succeed()) + _, err := reconciler.Reconcile(ctx, request) + Expect(err).NotTo(HaveOccurred()) + + Expect(k8sClient.Get(ctx, stackitKey, got)).To(Succeed()) + fakeCloud.SetServerState(got.Status.Bastion.ServerID, "ERROR", "") + recorder := events.NewFakeRecorder(10) + reconciler.Recorder = recorder + + result, err := reconciler.Reconcile(ctx, request) + Expect(err).NotTo(HaveOccurred()) + Expect(result.RequeueAfter).To(Equal(15 * time.Second)) + + Expect(k8sClient.Get(ctx, stackitKey, got)).To(Succeed()) + expectCondition(got.Status.Conditions, infrav1.ClusterBastionReadyCondition, metav1.ConditionFalse, "InstanceFailed") + condition := meta.FindStatusCondition(got.Status.Conditions, infrav1.ClusterBastionReadyCondition) + Expect(condition.Message).To(Equal("bastion server failed (state ERROR)")) + Expect(drainEvents(recorder)).To(ContainElement(HavePrefix("Warning InstanceFailed"))) + }) + It("deletes existing bastion resources when bastion is disabled", func() { got := &infrav1.StackitCluster{} Expect(k8sClient.Get(ctx, stackitKey, got)).To(Succeed()) diff --git a/controller/stackitmachine_controller_test.go b/controller/stackitmachine_controller_test.go index 968f5f9..16c6096 100644 --- a/controller/stackitmachine_controller_test.go +++ b/controller/stackitmachine_controller_test.go @@ -19,8 +19,10 @@ import ( . "github.com/onsi/gomega" corev1 "k8s.io/api/core/v1" apierrors "k8s.io/apimachinery/pkg/api/errors" + "k8s.io/apimachinery/pkg/api/meta" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/types" + "k8s.io/client-go/tools/events" clusterv1 "sigs.k8s.io/cluster-api/api/core/v1beta2" "sigs.k8s.io/controller-runtime/pkg/reconcile" @@ -178,6 +180,59 @@ var _ = Describe("StackitMachine Controller", func() { "legacy status.ready must follow the Ready condition, not contradict it") }) + It("reports a stopped server with its own reason and warns once", func() { + recorder := events.NewFakeRecorder(10) + reconciler.Recorder = recorder + _, err := reconciler.Reconcile(ctx, request) + Expect(err).NotTo(HaveOccurred()) + Expect(drainEvents(recorder)).To(ContainElement(HavePrefix("Normal InstanceCreated"))) + + got := &infrav1.StackitMachine{} + Expect(k8sClient.Get(ctx, stackitKey, got)).To(Succeed()) + + By("passing through the transitional STOPPING state without a warning") + fakeCloud.SetServerState(got.Status.InstanceID, "STOPPING", "RUNNING") + _, err = reconciler.Reconcile(ctx, request) + Expect(err).NotTo(HaveOccurred()) + Expect(drainEvents(recorder)).NotTo(ContainElement(HavePrefix("Warning"))) + + By("warning once the server is stopped") + fakeCloud.SetServerState(got.Status.InstanceID, "INACTIVE", "STOPPED") + result, err := reconciler.Reconcile(ctx, request) + Expect(err).NotTo(HaveOccurred()) + Expect(result.RequeueAfter).To(Equal(15 * time.Second)) + + Expect(k8sClient.Get(ctx, stackitKey, got)).To(Succeed()) + Expect(got.Status.Ready).To(BeFalse()) + Expect(got.Status.InstanceState).To(Equal("INACTIVE")) + expectCondition(got.Status.Conditions, infrav1.MachineInstanceReadyCondition, metav1.ConditionFalse, "InstanceStopped") + condition := meta.FindStatusCondition(got.Status.Conditions, infrav1.MachineInstanceReadyCondition) + Expect(condition.Message).To(Equal("server is stopped (state INACTIVE, power status STOPPED)")) + Expect(drainEvents(recorder)).To(ContainElement(HavePrefix("Warning InstanceStopped"))) + + By("not warning again while the server stays stopped") + _, err = reconciler.Reconcile(ctx, request) + Expect(err).NotTo(HaveOccurred()) + Expect(drainEvents(recorder)).NotTo(ContainElement(HavePrefix("Warning"))) + }) + + It("reports an active but crashed server as not ready", func() { + _, err := reconciler.Reconcile(ctx, request) + Expect(err).NotTo(HaveOccurred()) + + got := &infrav1.StackitMachine{} + Expect(k8sClient.Get(ctx, stackitKey, got)).To(Succeed()) + fakeCloud.SetServerState(got.Status.InstanceID, "ACTIVE", "CRASHED") + + _, err = reconciler.Reconcile(ctx, request) + Expect(err).NotTo(HaveOccurred()) + + Expect(k8sClient.Get(ctx, stackitKey, got)).To(Succeed()) + Expect(got.Status.Ready).To(BeFalse()) + expectCondition(got.Status.Conditions, infrav1.MachineInstanceReadyCondition, metav1.ConditionFalse, "InstanceCrashed") + expectCondition(got.Status.Conditions, infrav1.MachineReadyCondition, metav1.ConditionFalse, "InstanceCrashed") + }) + It("attaches provider-managed node SSH access when bastion is enabled", func() { stackitCluster := &infrav1.StackitCluster{} Expect(k8sClient.Get(ctx, types.NamespacedName{Name: clusterName, Namespace: namespace}, stackitCluster)).To(Succeed()) diff --git a/controller/stackitmachine_infrastructure.go b/controller/stackitmachine_infrastructure.go index 114ff9d..d86002d 100644 --- a/controller/stackitmachine_infrastructure.go +++ b/controller/stackitmachine_infrastructure.go @@ -18,6 +18,7 @@ import ( corev1 "k8s.io/api/core/v1" apierrors "k8s.io/apimachinery/pkg/api/errors" + "k8s.io/apimachinery/pkg/api/meta" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/types" clusterv1 "sigs.k8s.io/cluster-api/api/core/v1beta2" @@ -129,13 +130,16 @@ func (r *StackitMachineReconciler) reconcileNormal(ctx context.Context, machineS stackitMachine.Status.Addresses = machineAddressesFromCloud(server.Addresses) providerID := machineScope.SetInstance(server) - if server.State != "" && server.State != "ACTIVE" { - machineScope.SetNotReady( - "Provisioning", - fmt.Sprintf("server state is %s", server.State), - infrav1.MachineInstanceReadyCondition, - infrav1.MachineReadyCondition, - ) + if ready, reason, message, warn := serverStateCondition(server.State, server.PowerStatus); !ready { + previousReason := "" + if previous := meta.FindStatusCondition(stackitMachine.Status.Conditions, infrav1.MachineInstanceReadyCondition); previous != nil { + previousReason = previous.Reason + } + machineScope.SetNotReady(reason, message, infrav1.MachineInstanceReadyCondition, infrav1.MachineReadyCondition) + // Only warn on entering the state, not on every requeue. + if warn && r.Recorder != nil && previousReason != reason { + r.Recorder.Eventf(stackitMachine, nil, corev1.EventTypeWarning, reason, "Reconcile", "Server %s: %s", server.ID, message) + } return ctrl.Result{RequeueAfter: 15 * time.Second}, nil } diff --git a/docs/src/usage/cluster-template.md b/docs/src/usage/cluster-template.md index 788c86d..5c1b24f 100644 --- a/docs/src/usage/cluster-template.md +++ b/docs/src/usage/cluster-template.md @@ -10,6 +10,7 @@ non-topology workload cluster. It renders: - `MachineDeployment` - worker `StackitMachineTemplate` - worker `KubeadmConfigTemplate` +- `MachineHealthCheck` for control-plane and worker Machines - `ClusterResourceSet` and resource-set `Secret` for `cloud-provider-stackit` Render a cluster: @@ -51,3 +52,20 @@ workload API is reachable, install a CNI that matches the configured pod/service CIDRs and network policy expectations before expecting Nodes to become Ready. For a reproducible development path, use the helper documented in [Workload CNI](cni.md). + +The MachineHealthChecks remediate Machines whose Node does not register within +10 minutes or reports `Ready=False` or `Ready=Unknown` for 10 minutes. The +`MachineDeployment` limits worker remediation to 2 Machines at a time via +`spec.remediation.maxInFlight`. + +Worker remediation only runs while at most 40% of the workers are unhealthy +(`triggerIf.unhealthyLessThanOrEqualTo`, rounded down). A MachineDeployment +with fewer than 3 workers is therefore never remediated; use +`WORKER_MACHINE_COUNT` ≥ 3 for automatic replacement. A network outage longer +than 10 minutes can still replace healthy workers once the first of them +recovers and the unhealthy share drops below the limit. + +Control-plane and worker Machines use `nodeDrainTimeoutSeconds: 900`, so a +Node whose Pods cannot terminate does not block deletion forever. This applies +to every Machine deletion, including upgrades and scale-down: Pods still +protected by a PodDisruptionBudget after 15 minutes are not waited for. diff --git a/docs/src/usage/clusterclass.md b/docs/src/usage/clusterclass.md index 96cc0a3..d964762 100644 --- a/docs/src/usage/clusterclass.md +++ b/docs/src/usage/clusterclass.md @@ -64,6 +64,13 @@ the generated `StackitCluster` and `StackitMachine` objects. This is useful for policy, ownership, automation, and observability labels that should be present on generated infrastructure objects. +The class also defines `healthCheck` and `deletion` for the control plane and +the `default-worker` class with the same checks, remediation limits and drain +timeouts as the classic template's MachineHealthChecks (see +[Classic Cluster Template](cluster-template.md)). Disable the health checks per +cluster with `spec.topology.controlPlane.healthCheck.enabled: false` or the +equivalent field on a MachineDeployment topology. + ## Prerequisites The management cluster must run CAPI core and kubeadm-control-plane with diff --git a/docs/src/usage/cni.md b/docs/src/usage/cni.md index 403539b..f154201 100644 --- a/docs/src/usage/cni.md +++ b/docs/src/usage/cni.md @@ -5,9 +5,11 @@ The default cluster template provisions STACKIT infrastructure and installs keeps CNI choice with the cluster operator, where network policy, routing, MTU, IPAM, and upgrade strategy belong. -Nodes will not become fully Ready until a CNI is installed. For local validation -and simple development clusters, this repository provides a repeatable helper -for Cilium or Calico. +Nodes will not become fully Ready until a CNI is installed. The templates ship +MachineHealthChecks that replace Machines whose Node stays `Ready=False` for 10 +minutes, so install the CNI within that window after the Nodes join. For local +validation and simple development clusters, this repository provides a +repeatable helper for Cilium or Calico. For the complete tested workload addon flow, including the embedded `cloud-provider-stackit` `ClusterResourceSet` and verification commands, see diff --git a/templates/cluster-template-bastion.yaml b/templates/cluster-template-bastion.yaml index 50488f5..6c34176 100644 --- a/templates/cluster-template-bastion.yaml +++ b/templates/cluster-template-bastion.yaml @@ -86,6 +86,8 @@ spec: version: ${KUBERNETES_VERSION} machineTemplate: spec: + deletion: + nodeDrainTimeoutSeconds: 900 infrastructureRef: apiGroup: infrastructure.cluster.x-k8s.io kind: StackitMachineTemplate @@ -158,6 +160,8 @@ metadata: spec: clusterName: ${CLUSTER_NAME} replicas: ${WORKER_MACHINE_COUNT} + remediation: + maxInFlight: 2 selector: matchLabels: cluster.x-k8s.io/cluster-name: ${CLUSTER_NAME} @@ -170,6 +174,8 @@ spec: spec: clusterName: ${CLUSTER_NAME} version: ${KUBERNETES_VERSION} + deletion: + nodeDrainTimeoutSeconds: 900 bootstrap: configRef: apiGroup: bootstrap.cluster.x-k8s.io @@ -180,6 +186,49 @@ spec: kind: StackitMachineTemplate name: ${CLUSTER_NAME}-md-0 --- +apiVersion: cluster.x-k8s.io/v1beta2 +kind: MachineHealthCheck +metadata: + name: ${CLUSTER_NAME}-control-plane + namespace: ${NAMESPACE} +spec: + clusterName: ${CLUSTER_NAME} + selector: + matchLabels: + cluster.x-k8s.io/control-plane: "" + checks: + nodeStartupTimeoutSeconds: 600 + unhealthyNodeConditions: + - type: Ready + status: "False" + timeoutSeconds: 600 + - type: Ready + status: "Unknown" + timeoutSeconds: 600 +--- +apiVersion: cluster.x-k8s.io/v1beta2 +kind: MachineHealthCheck +metadata: + name: ${CLUSTER_NAME}-md-0 + namespace: ${NAMESPACE} +spec: + clusterName: ${CLUSTER_NAME} + selector: + matchLabels: + cluster.x-k8s.io/deployment-name: ${CLUSTER_NAME}-md-0 + checks: + nodeStartupTimeoutSeconds: 600 + unhealthyNodeConditions: + - type: Ready + status: "False" + timeoutSeconds: 600 + - type: Ready + status: "Unknown" + timeoutSeconds: 600 + remediation: + triggerIf: + unhealthyLessThanOrEqualTo: 40% +--- apiVersion: infrastructure.cluster.x-k8s.io/v1alpha1 kind: StackitMachineTemplate metadata: diff --git a/templates/cluster-template-development.yaml b/templates/cluster-template-development.yaml index e8a06d3..38487bd 100644 --- a/templates/cluster-template-development.yaml +++ b/templates/cluster-template-development.yaml @@ -50,6 +50,8 @@ spec: version: ${KUBERNETES_VERSION} machineTemplate: spec: + deletion: + nodeDrainTimeoutSeconds: 900 infrastructureRef: apiGroup: infrastructure.cluster.x-k8s.io kind: StackitMachineTemplate @@ -103,6 +105,8 @@ metadata: spec: clusterName: ${CLUSTER_NAME} replicas: ${WORKER_MACHINE_COUNT} + remediation: + maxInFlight: 2 selector: matchLabels: cluster.x-k8s.io/cluster-name: ${CLUSTER_NAME} @@ -115,6 +119,8 @@ spec: spec: clusterName: ${CLUSTER_NAME} version: ${KUBERNETES_VERSION} + deletion: + nodeDrainTimeoutSeconds: 900 bootstrap: configRef: apiGroup: bootstrap.cluster.x-k8s.io @@ -125,6 +131,49 @@ spec: kind: StackitMachineTemplate name: ${CLUSTER_NAME}-md-0 --- +apiVersion: cluster.x-k8s.io/v1beta2 +kind: MachineHealthCheck +metadata: + name: ${CLUSTER_NAME}-control-plane + namespace: ${NAMESPACE} +spec: + clusterName: ${CLUSTER_NAME} + selector: + matchLabels: + cluster.x-k8s.io/control-plane: "" + checks: + nodeStartupTimeoutSeconds: 600 + unhealthyNodeConditions: + - type: Ready + status: "False" + timeoutSeconds: 600 + - type: Ready + status: "Unknown" + timeoutSeconds: 600 +--- +apiVersion: cluster.x-k8s.io/v1beta2 +kind: MachineHealthCheck +metadata: + name: ${CLUSTER_NAME}-md-0 + namespace: ${NAMESPACE} +spec: + clusterName: ${CLUSTER_NAME} + selector: + matchLabels: + cluster.x-k8s.io/deployment-name: ${CLUSTER_NAME}-md-0 + checks: + nodeStartupTimeoutSeconds: 600 + unhealthyNodeConditions: + - type: Ready + status: "False" + timeoutSeconds: 600 + - type: Ready + status: "Unknown" + timeoutSeconds: 600 + remediation: + triggerIf: + unhealthyLessThanOrEqualTo: 40% +--- apiVersion: infrastructure.cluster.x-k8s.io/v1alpha1 kind: StackitMachineTemplate metadata: diff --git a/templates/cluster-template-flatcar-workers.yaml b/templates/cluster-template-flatcar-workers.yaml index 57974ce..a3087ef 100644 --- a/templates/cluster-template-flatcar-workers.yaml +++ b/templates/cluster-template-flatcar-workers.yaml @@ -50,6 +50,8 @@ spec: version: ${KUBERNETES_VERSION} machineTemplate: spec: + deletion: + nodeDrainTimeoutSeconds: 900 infrastructureRef: apiGroup: infrastructure.cluster.x-k8s.io kind: StackitMachineTemplate @@ -121,6 +123,8 @@ metadata: spec: clusterName: ${CLUSTER_NAME} replicas: ${WORKER_MACHINE_COUNT} + remediation: + maxInFlight: 2 selector: matchLabels: cluster.x-k8s.io/cluster-name: ${CLUSTER_NAME} @@ -133,6 +137,8 @@ spec: spec: clusterName: ${CLUSTER_NAME} version: ${KUBERNETES_VERSION} + deletion: + nodeDrainTimeoutSeconds: 900 bootstrap: configRef: apiGroup: bootstrap.cluster.x-k8s.io @@ -143,6 +149,49 @@ spec: kind: StackitMachineTemplate name: ${CLUSTER_NAME}-md-0 --- +apiVersion: cluster.x-k8s.io/v1beta2 +kind: MachineHealthCheck +metadata: + name: ${CLUSTER_NAME}-control-plane + namespace: ${NAMESPACE} +spec: + clusterName: ${CLUSTER_NAME} + selector: + matchLabels: + cluster.x-k8s.io/control-plane: "" + checks: + nodeStartupTimeoutSeconds: 600 + unhealthyNodeConditions: + - type: Ready + status: "False" + timeoutSeconds: 600 + - type: Ready + status: "Unknown" + timeoutSeconds: 600 +--- +apiVersion: cluster.x-k8s.io/v1beta2 +kind: MachineHealthCheck +metadata: + name: ${CLUSTER_NAME}-md-0 + namespace: ${NAMESPACE} +spec: + clusterName: ${CLUSTER_NAME} + selector: + matchLabels: + cluster.x-k8s.io/deployment-name: ${CLUSTER_NAME}-md-0 + checks: + nodeStartupTimeoutSeconds: 600 + unhealthyNodeConditions: + - type: Ready + status: "False" + timeoutSeconds: 600 + - type: Ready + status: "Unknown" + timeoutSeconds: 600 + remediation: + triggerIf: + unhealthyLessThanOrEqualTo: 40% +--- apiVersion: infrastructure.cluster.x-k8s.io/v1alpha1 kind: StackitMachineTemplate metadata: diff --git a/templates/cluster-template.yaml b/templates/cluster-template.yaml index f2e66af..14f2034 100644 --- a/templates/cluster-template.yaml +++ b/templates/cluster-template.yaml @@ -50,6 +50,8 @@ spec: version: ${KUBERNETES_VERSION} machineTemplate: spec: + deletion: + nodeDrainTimeoutSeconds: 900 infrastructureRef: apiGroup: infrastructure.cluster.x-k8s.io kind: StackitMachineTemplate @@ -121,6 +123,8 @@ metadata: spec: clusterName: ${CLUSTER_NAME} replicas: ${WORKER_MACHINE_COUNT} + remediation: + maxInFlight: 2 selector: matchLabels: cluster.x-k8s.io/cluster-name: ${CLUSTER_NAME} @@ -133,6 +137,8 @@ spec: spec: clusterName: ${CLUSTER_NAME} version: ${KUBERNETES_VERSION} + deletion: + nodeDrainTimeoutSeconds: 900 bootstrap: configRef: apiGroup: bootstrap.cluster.x-k8s.io @@ -143,6 +149,49 @@ spec: kind: StackitMachineTemplate name: ${CLUSTER_NAME}-md-0 --- +apiVersion: cluster.x-k8s.io/v1beta2 +kind: MachineHealthCheck +metadata: + name: ${CLUSTER_NAME}-control-plane + namespace: ${NAMESPACE} +spec: + clusterName: ${CLUSTER_NAME} + selector: + matchLabels: + cluster.x-k8s.io/control-plane: "" + checks: + nodeStartupTimeoutSeconds: 600 + unhealthyNodeConditions: + - type: Ready + status: "False" + timeoutSeconds: 600 + - type: Ready + status: "Unknown" + timeoutSeconds: 600 +--- +apiVersion: cluster.x-k8s.io/v1beta2 +kind: MachineHealthCheck +metadata: + name: ${CLUSTER_NAME}-md-0 + namespace: ${NAMESPACE} +spec: + clusterName: ${CLUSTER_NAME} + selector: + matchLabels: + cluster.x-k8s.io/deployment-name: ${CLUSTER_NAME}-md-0 + checks: + nodeStartupTimeoutSeconds: 600 + unhealthyNodeConditions: + - type: Ready + status: "False" + timeoutSeconds: 600 + - type: Ready + status: "Unknown" + timeoutSeconds: 600 + remediation: + triggerIf: + unhealthyLessThanOrEqualTo: 40% +--- apiVersion: infrastructure.cluster.x-k8s.io/v1alpha1 kind: StackitMachineTemplate metadata: diff --git a/templates/clusterclass.yaml b/templates/clusterclass.yaml index 04f7431..57b5c34 100644 --- a/templates/clusterclass.yaml +++ b/templates/clusterclass.yaml @@ -18,6 +18,18 @@ spec: apiVersion: infrastructure.cluster.x-k8s.io/v1alpha1 kind: StackitMachineTemplate name: stackit-control-plane + deletion: + nodeDrainTimeoutSeconds: 900 + healthCheck: + checks: + nodeStartupTimeoutSeconds: 600 + unhealthyNodeConditions: + - type: Ready + status: "False" + timeoutSeconds: 600 + - type: Ready + status: "Unknown" + timeoutSeconds: 600 workers: machineDeployments: - class: default-worker @@ -31,6 +43,22 @@ spec: apiVersion: infrastructure.cluster.x-k8s.io/v1alpha1 kind: StackitMachineTemplate name: stackit-default-worker + deletion: + nodeDrainTimeoutSeconds: 900 + healthCheck: + checks: + nodeStartupTimeoutSeconds: 600 + unhealthyNodeConditions: + - type: Ready + status: "False" + timeoutSeconds: 600 + - type: Ready + status: "Unknown" + timeoutSeconds: 600 + remediation: + maxInFlight: 2 + triggerIf: + unhealthyLessThanOrEqualTo: 40% variables: - name: stackitProjectID required: true