mirror of
https://github.com/zalando/postgres-operator.git
synced 2026-09-30 21:30:28 +02:00
Retry moveMasterPodsOffNode on failure instead of aborting (#3179)
attemptToMoveMasterPodsOffNode's transient errors (e.g. no synced standby available yet) were treated as terminal, so the retry loop gave up after a single ~4-5s attempt instead of retrying over master_pod_move_timeout. This left masters stuck on draining nodes indefinitely once a PodDisruptionBudget started rejecting evictions. Log the failure and return (false, nil) so the retry loop keeps polling every minute until the timeout expires. Fixes #3176
This commit is contained in:
@@ -156,7 +156,8 @@ func (c *Controller) moveMasterPodsOffNode(node *v1.Node) {
|
||||
func() (bool, error) {
|
||||
err := c.attemptToMoveMasterPodsOffNode(node)
|
||||
if err != nil {
|
||||
return false, err
|
||||
c.logger.Warningf("attempt to move master pods off node %q failed, will retry: %v", node.Name, err)
|
||||
return false, nil
|
||||
}
|
||||
return true, nil
|
||||
},
|
||||
|
||||
Reference in New Issue
Block a user