Skip to content

Commit

Permalink
Restart unhealthy API servers when provisioning/upgrading clusters
Browse files Browse the repository at this point in the history
Signed-off-by: Marko Mudrinić <mudrinic.mare@gmail.com>
  • Loading branch information
xmudrii committed Feb 8, 2021
1 parent 222201d commit 8c87b06
Show file tree
Hide file tree
Showing 3 changed files with 29 additions and 0 deletions.
15 changes: 15 additions & 0 deletions pkg/scripts/node.go
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,17 @@ var (
sudo KUBECONFIG=/etc/kubernetes/admin.conf \
kubectl drain {{ .NODE_NAME }} --ignore-daemonsets --delete-local-data
`)

restartKubeAPIServerTemplate = heredoc.Doc(`
apiserver_id=$(sudo crictl ps --name=kube-apiserver -q)
[ -z "$apiserver_id" ] && exit 1
sudo crictl logs "$apiserver_id" > /tmp/kube-apiserver.log 2>&1
if grep -q "etcdserver: no leader" /tmp/kube-apiserver.log; then
sudo crictl rm "$apiserver_id"
sleep 10
fi
`)
)

func DrainNode(nodeName string) (string, error) {
Expand All @@ -40,3 +51,7 @@ func DrainNode(nodeName string) (string, error) {
func Hostname() string {
return hostnameScript
}

func RestartKubeAPIServer() string {
return restartKubeAPIServerTemplate
}
12 changes: 12 additions & 0 deletions pkg/tasks/nodes.go
Original file line number Diff line number Diff line change
Expand Up @@ -56,3 +56,15 @@ func uncordonNode(s *state.State, host kubeoneapi.HostConfig) error {

return errors.WithStack(updateErr)
}

func restartKubeAPIServer(s *state.State) error {
s.Logger.Infoln("Restarting unhealthy API servers if needed...")
return s.RunTaskOnControlPlane(func(s *state.State, node *kubeoneapi.HostConfig, conn ssh.Connection) error {
_, _, err := s.Runner.Run(scripts.RestartKubeAPIServer(), nil)
if err != nil {
return err
}

return nil
}, state.RunSequentially)
}
2 changes: 2 additions & 0 deletions pkg/tasks/tasks.go
Original file line number Diff line number Diff line change
Expand Up @@ -114,6 +114,7 @@ func WithFullInstall(t Tasks) Tasks {
{Fn: repairClusterIfNeeded, ErrMsg: "failed to repair cluster"},
{Fn: joinControlplaneNode, ErrMsg: "failed to join other masters a cluster"},
{Fn: saveKubeconfig, ErrMsg: "failed to save kubeconfig to the local machine"},
{Fn: restartKubeAPIServer, ErrMsg: "failed to restart unhealthy kube-apiserver"},
}...).
append(kubernetesResources()...).
append(
Expand Down Expand Up @@ -188,6 +189,7 @@ func WithUpgrade(t Tasks) Tasks {
}...).
append(kubernetesResources()...).
append(
Task{Fn: restartKubeAPIServer, ErrMsg: "failed to restart unhealthy kube-apiserver"},
Task{Fn: upgradeStaticWorkers, ErrMsg: "unable to upgrade static worker nodes"},
Task{
Fn: upgradeMachineDeployments,
Expand Down

0 comments on commit 8c87b06

Please sign in to comment.