Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions docs/manual/operations.md
Original file line number Diff line number Diff line change
Expand Up @@ -272,8 +272,8 @@ If the cluster is a multi-region cluster, perform this step for all running regi
- Now you can set `spec.Skip = false` to let the operator take over again.
- Depending on the state of the multi-region cluster, you probably want to change the desired database configuration to drop ha.

The [kubectl-fdb plugin](../../kubectl-fdb/Readme.md) provides a `recover-multi-region-cluster` command that can be used to automatically recover a cluster with the above steps.
The command has some additional safety checks, to ensure the steps are only performed on a cluster that is unhealthy and the majority of coordinators are unreachable.
The [kubectl-fdb plugin](../../kubectl-fdb/Readme.md) provides `recover multi-region` and `recover single-dc` commands that can be used to automatically recover a cluster with the above steps.
The commands have some additional safety checks, to ensure the steps are only performed on a cluster that is unhealthy and the majority of coordinators are unreachable.

## Next

Expand Down
69 changes: 61 additions & 8 deletions e2e/fixtures/fdb_cluster.go
Original file line number Diff line number Diff line change
Expand Up @@ -42,7 +42,7 @@ import (
"github.com/onsi/gomega"
corev1 "k8s.io/api/core/v1"
"k8s.io/apimachinery/pkg/api/equality"
kubeErrors "k8s.io/apimachinery/pkg/api/errors"
k8serrors "k8s.io/apimachinery/pkg/api/errors"
"k8s.io/apimachinery/pkg/api/resource"
"k8s.io/apimachinery/pkg/util/wait"
"sigs.k8s.io/controller-runtime/pkg/client"
Expand Down Expand Up @@ -829,6 +829,59 @@ func (fdbCluster *FdbCluster) SetPodAsUnschedulable(ctx context.Context, pod cor
}).WithTimeout(5*time.Minute).WithPolling(2*time.Second).MustPassRepeatedly(5).Should(gomega.BeEmpty(), "Not able to set pod as unschedulable")
}

// SetPodsAsUnschedulable sets the provided slice of Pods on the NoSchedule list of the current FoundationDBCluster. This will make
// sure that the Pods are stuck in Pending.
func (fdbCluster *FdbCluster) SetPodsAsUnschedulable(ctx context.Context, pods []corev1.Pod) {
unschedulableProcessGroups := make([]fdbv1beta2.ProcessGroupID, 0, len(pods))

for _, pod := range pods {
unschedulableProcessGroups = append(unschedulableProcessGroups, GetProcessGroupID(pod))
}

fdbCluster.SetProcessGroupsAsUnschedulable(
ctx,
unschedulableProcessGroups,
)

select {
case <-ctx.Done():
return
case <-time.After(5 * time.Second):
}

for _, pod := range pods {
fetchedPod := &corev1.Pod{}
err := fdbCluster.getClient().
Get(ctx, client.ObjectKeyFromObject(&pod), fetchedPod)
if err != nil {
continue
}

// Try deleting the Pod as a workaround until the operator handles all cases.
if fetchedPod.Spec.NodeName != "" && fetchedPod.DeletionTimestamp.IsZero() {
Comment thread
johscheuer marked this conversation as resolved.
gomega.Expect(fdbCluster.getClient().Delete(ctx, &pod)).
NotTo(gomega.HaveOccurred())
}
}

gomega.Eventually(func(g gomega.Gomega) {
for _, pod := range pods {
fetchedPod := &corev1.Pod{}
err := fdbCluster.getClient().
Get(ctx, client.ObjectKeyFromObject(&pod), fetchedPod)
g.Expect(err).NotTo(gomega.HaveOccurred())

// Try deleting the Pod as a workaround until the operator handles all cases.
if fetchedPod.Spec.NodeName != "" && fetchedPod.DeletionTimestamp.IsZero() {
g.Expect(fdbCluster.getClient().Delete(ctx, &pod)).
NotTo(gomega.HaveOccurred())
}

g.Expect(fetchedPod.Spec.NodeName).To(gomega.BeEmpty())
}
}).WithTimeout(5*time.Minute).WithPolling(2*time.Second).MustPassRepeatedly(5).Should(gomega.Succeed(), "Not able to set pods as unschedulable")
}

// SetProcessGroupsAsUnschedulable sets the provided process groups on the NoSchedule list of the current FoundationDBCluster. This will make
// sure that the Pod is stuck in Pending.
func (fdbCluster *FdbCluster) SetProcessGroupsAsUnschedulable(
Expand Down Expand Up @@ -980,7 +1033,7 @@ func (fdbCluster *FdbCluster) WaitForPodRemoval(ctx context.Context, pod *corev1
gomega.Eventually(func() bool {
err := fdbCluster.getClient().
Get(ctx, client.ObjectKeyFromObject(pod), fetchedPod)
if err != nil && kubeErrors.IsNotFound(err) {
if err != nil && k8serrors.IsNotFound(err) {
return true
}

Expand Down Expand Up @@ -1216,7 +1269,7 @@ func (fdbCluster *FdbCluster) CheckPodIsDeleted(ctx context.Context, podName str
Get(ctx, client.ObjectKey{Namespace: fdbCluster.Namespace(), Name: podName}, pod)

if err != nil {
if kubeErrors.IsNotFound(err) {
if k8serrors.IsNotFound(err) {
return true
}
}
Expand Down Expand Up @@ -1260,13 +1313,13 @@ func (fdbCluster *FdbCluster) SetUseDNSInClusterFile(
return fdbCluster.WaitForReconciliation(ctx)
}

// Destroy will remove the underlying cluster.
func (fdbCluster *FdbCluster) Destroy(ctx context.Context) error {
return fdbCluster.DestroyWithWaitForTearDown(ctx, false)
// Delete will remove the underlying cluster.
func (fdbCluster *FdbCluster) Delete(ctx context.Context) error {
return fdbCluster.DeleteWithWaitForTearDown(ctx, false)
}

// DestroyWithWaitForTearDown will remove the underlying cluster and wait for the resources to be removed.
func (fdbCluster *FdbCluster) DestroyWithWaitForTearDown(
// DeleteWithWaitForTearDown will remove the underlying cluster and wait for the resources to be removed.
func (fdbCluster *FdbCluster) DeleteWithWaitForTearDown(
ctx context.Context,
waitForTearDown bool,
) error {
Expand Down
2 changes: 1 addition & 1 deletion e2e/fixtures/ha_fdb_cluster.go
Original file line number Diff line number Diff line change
Expand Up @@ -219,7 +219,7 @@ func (factory *Factory) createHaFdbClusterSpec(
// Delete removes all Clusters associated FoundationDBClusters.
func (haFDBCluster *HaFdbCluster) Delete(ctx context.Context) {
for _, cluster := range haFDBCluster.GetAllClusters() {
gomega.Expect(cluster.Destroy(ctx)).NotTo(gomega.HaveOccurred())
gomega.Expect(cluster.Delete(ctx)).NotTo(gomega.HaveOccurred())
}
}

Expand Down
2 changes: 1 addition & 1 deletion e2e/test_operator_backups/operator_backup_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -103,7 +103,7 @@ var _ = Describe("Operator Backup", Label("e2e", "pr", "foundationdb-pr"), func(

namespace := fdbCluster.Namespace()
// Delete the FDB cluster to have a clean start.
Expect(fdbCluster.DestroyWithWaitForTearDown(ctx, true)).To(Succeed())
Expect(fdbCluster.DeleteWithWaitForTearDown(ctx, true)).To(Succeed())
// Restart the operator pods.
factory.RecreateOperatorPods(ctx, namespace)
})
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -85,7 +85,7 @@ var _ = Describe("Test Operator Velocity", Label("e2e"), func() {

runTime := time.Since(startTime)
log.Println("Single-DC cluster creation took: ", runTime.String())
Expect(fdbCluster.Destroy(ctx)).ToNot(HaveOccurred())
Expect(fdbCluster.Delete(ctx)).ToNot(HaveOccurred())
})
})

Expand Down
4 changes: 2 additions & 2 deletions e2e/test_operator_ha_failure/operator_ha_failure_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -134,7 +134,7 @@ var _ = Describe("Operator HA Failure tests", Label("e2e"), func() {

keyValues = primary.GenerateRandomValues(10, prefix)
primary.WriteKeyValuesWithTimeout(ctx, keyValues, 120)
// Destroy primary and primary satellite (should have mutations that are not present in the remote side).
// Delete primary and primary satellite (should have mutations that are not present in the remote side).
primary.SetSkipReconciliation(ctx, true)
primarySatellite.SetSkipReconciliation(ctx, true)
// We also destroy the remote satellite, it shouldn't matter in this case as the remote satellite
Expand Down Expand Up @@ -199,7 +199,7 @@ var _ = Describe("Operator HA Failure tests", Label("e2e"), func() {
&operatorPod,
"manager",
fmt.Sprintf(
"kubectl-fdb -n %s recover-multi-region-cluster --version-check=false --wait=false %s",
"kubectl-fdb -n %s recover multi-region --version-check=false --wait=false %s",
remote.Namespace(),
remote.Name(),
),
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -82,7 +82,7 @@ var _ = PDescribe("Operator Migrate Image Type", Label("e2e"), func() {
})

AfterEach(func(ctx SpecContext) {
Expect(fdbCluster.Destroy(ctx)).NotTo(HaveOccurred())
Expect(fdbCluster.Delete(ctx)).NotTo(HaveOccurred())
})

It("should convert the cluster", func(ctx SpecContext) {
Expand Down Expand Up @@ -136,7 +136,7 @@ var _ = PDescribe("Operator Migrate Image Type", Label("e2e"), func() {
})

AfterEach(func(ctx SpecContext) {
Expect(fdbCluster.Destroy(ctx)).NotTo(HaveOccurred())
Expect(fdbCluster.Delete(ctx)).NotTo(HaveOccurred())
})

It("should convert the cluster", func(ctx SpecContext) {
Expand Down Expand Up @@ -184,7 +184,7 @@ var _ = PDescribe("Operator Migrate Image Type", Label("e2e"), func() {
})

AfterEach(func(ctx SpecContext) {
Expect(fdbCluster.Destroy(ctx)).NotTo(HaveOccurred())
Expect(fdbCluster.Delete(ctx)).NotTo(HaveOccurred())
})

It("should convert the cluster", func(ctx SpecContext) {
Expand Down
Loading
Loading