Files
actions-runner-controller/controllers/actions.github.com/ephemeralrunner_controller.go
T

1360 lines
55 KiB
Go

/*
Copyright 2020 The actions-runner-controller authors.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
*/
package actionsgithubcom
import (
"context"
"errors"
"fmt"
"strconv"
"strings"
"sync"
"time"
"github.com/actions/actions-runner-controller/apis/actions.github.com/v1alpha1"
"github.com/actions/actions-runner-controller/controllers/actions.github.com/metrics"
"github.com/actions/actions-runner-controller/controllers/actions.github.com/multiclient"
"github.com/actions/actions-runner-controller/github/actions"
"github.com/actions/scaleset"
"github.com/go-logr/logr"
corev1 "k8s.io/api/core/v1"
kerrors "k8s.io/apimachinery/pkg/api/errors"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/runtime"
"k8s.io/apimachinery/pkg/types"
ctrl "sigs.k8s.io/controller-runtime"
"sigs.k8s.io/controller-runtime/pkg/builder"
"sigs.k8s.io/controller-runtime/pkg/client"
"sigs.k8s.io/controller-runtime/pkg/controller/controllerutil"
"sigs.k8s.io/controller-runtime/pkg/predicate"
)
const (
ephemeralRunnerFinalizerName = "ephemeralrunner.actions.github.com/finalizer"
ephemeralRunnerActionsFinalizerName = "ephemeralrunner.actions.github.com/runner-registration-finalizer"
// busyRunnerRequeueInterval is how long a runner being deleted while its
// pod is still executing a job waits before the service is asked again.
busyRunnerRequeueInterval = 30 * time.Second
)
// EphemeralRunnerReconciler reconciles a EphemeralRunner object
type EphemeralRunnerReconciler struct {
client.Client
APIReader client.Reader
Log logr.Logger
Scheme *runtime.Scheme
PublishMetrics bool
// UnregistrationQueue takes the removal of runner registrations from the
// Actions service off the reconcile path. When it is left unset,
// SetupWithManager creates one and registers it with the manager.
UnregistrationQueue *RunnerUnregistrationQueue
// TerminatedPodGracePeriodSeconds is the grace period used when deleting a
// runner pod whose containers have all exited. It is zero by default, so
// the pod leaves the API as soon as the delete is issued instead of sitting
// in Terminating while the kubelet cleans up locally.
//
// Raise it to keep those pods around for longer, for example to give a log
// collector time to read them. A negative value asks for no override at
// all, leaving the deletion to the pod's own terminationGracePeriodSeconds.
//
// It is only ever applied to a pod with nothing left running in it. Pods
// that are still alive are always deleted gracefully.
TerminatedPodGracePeriodSeconds int64
ResourceBuilder
}
var ephemeralRunnerPhaseMetrics = struct {
sync.Mutex
phases map[types.NamespacedName]v1alpha1.EphemeralRunnerPhase
}{
phases: map[types.NamespacedName]v1alpha1.EphemeralRunnerPhase{},
}
// precompute backoff durations for failed ephemeral runners
// the len(failedRunnerBackoff) must be equal to maxFailures + 1
var failedRunnerBackoff = []time.Duration{
0,
5 * time.Second,
10 * time.Second,
20 * time.Second,
40 * time.Second,
80 * time.Second,
}
const maxFailures = 5
// +kubebuilder:rbac:groups=actions.github.com,resources=ephemeralrunners,verbs=get;list;watch;create;update;patch;delete
// +kubebuilder:rbac:groups=actions.github.com,resources=ephemeralrunners/status,verbs=get;update;patch
// +kubebuilder:rbac:groups=actions.github.com,resources=ephemeralrunners/finalizers,verbs=get;list;watch;create;update;patch;delete
// +kubebuilder:rbac:groups=core,resources=pods,verbs=get;list;watch;create;update;patch;delete
// +kubebuilder:rbac:groups=core,resources=pods/status,verbs=get
// +kubebuilder:rbac:groups=core,resources=secrets,verbs=create;get;list;watch;delete
// Reconcile is part of the main kubernetes reconciliation loop which aims to
// move the current state of the cluster closer to the desired state.
//
// For more details, check Reconcile and its Result here:
// - https://pkg.go.dev/sigs.k8s.io/controller-runtime@v0.6.4/pkg/reconcile
func (r *EphemeralRunnerReconciler) Reconcile(ctx context.Context, req ctrl.Request) (ctrl.Result, error) {
log := r.Log.WithValues("ephemeralrunner", req.NamespacedName)
var ephemeralRunner v1alpha1.EphemeralRunner
if err := r.Get(ctx, req.NamespacedName, &ephemeralRunner); err != nil {
return ctrl.Result{}, client.IgnoreNotFound(err)
}
runner := newLazyCopy(&ephemeralRunner)
if !ephemeralRunner.DeletionTimestamp.IsZero() {
r.publishEphemeralRunnerPhaseMetric(&ephemeralRunner, "", log)
if !controllerutil.ContainsFinalizer(&ephemeralRunner, ephemeralRunnerFinalizerName) {
return ctrl.Result{}, nil
}
deferredActionsFinalizer := false
if controllerutil.ContainsFinalizer(&ephemeralRunner, ephemeralRunnerActionsFinalizerName) {
// This finalizer exists to release the runner's registration with the
// Actions service. There are two ways that happens.
//
// A runner that exited with code 0 already removed its own
// registration on the way out. Runners are ephemeral, so a clean exit
// means the agent deregistered itself before it stopped, and there is
// nothing left to ask the service to remove. That is the path every
// completed job takes, and it costs no API call at all.
//
// Every other runner may still hold a registration. While its pod is
// alive, the runner may be executing a job this controller has not
// heard about: the status records the runner ID only after the pod
// exists, and a job only once the listener reports it. So the service
// is asked to remove the runner before a live pod is deleted, and a
// runner that is still executing a job keeps its pod and is checked
// again later. That covers a runner the EphemeralRunnerSet deleted
// before its ID was recorded as well as one deleted by hand.
//
// A pod with nothing left running cannot be executing a job, so for it,
// and for a runner without a pod, the removal is handed to background
// workers instead, so that deleting the pod and the secret below is
// never held up by an external API. See RunnerUnregistrationQueue for
// what that costs.
var runnerID int
if runnerSelfDeregistered(&ephemeralRunner) {
log.Info("Runner exited successfully and deregistered itself, skipping its removal from the service")
} else {
getActionsClient := sync.OnceValues(func() (multiclient.Client, error) {
return r.GetActionsService(ctx, &ephemeralRunner)
})
// Resolved before the finalizer goes, because recovering an ID the
// status never recorded reads the jitconfig secret, which the
// cleanup below deletes.
id, err := r.registeredRunnerID(ctx, &ephemeralRunner, getActionsClient, log)
if err != nil {
log.Error(err, "Failed to resolve the registration of an ephemeral runner being deleted")
return ctrl.Result{}, err
}
runnerID = id
if runnerID != 0 {
removed, err := r.removeRunnerOfLivePod(ctx, &ephemeralRunner, runnerID, getActionsClient, log)
switch {
case errors.Is(err, scaleset.JobStillRunningError):
log.Info("Runner is still running a job, keeping its pod", "runnerId", runnerID, "requeueAfter", busyRunnerRequeueInterval)
return ctrl.Result{RequeueAfter: busyRunnerRequeueInterval}, nil
case err != nil:
log.Error(err, "Failed to remove the runner of a live pod from the service", "runnerId", runnerID)
return ctrl.Result{}, err
case removed:
runnerID = 0
}
}
}
log.Info(
"Removing the runner registration finalizer",
"unregisterFromService", runnerID != 0,
"phase", ephemeralRunner.Status.Phase,
)
removedActionsFinalizer := controllerutil.RemoveFinalizer(runner.Mutate(), ephemeralRunnerActionsFinalizerName)
// The patch has to land before the runner is queued: queueing first
// would ask the service to remove the same runner twice when the patch
// fails and the reconcile comes back through this branch. A runner with
// nothing to queue has nothing to order the patch against, so its
// removal rides along with the finalizer patch made after cleanup
// below rather than paying for a round trip of its own.
deferredActionsFinalizer = removedActionsFinalizer && runnerID == 0
if removedActionsFinalizer && !deferredActionsFinalizer {
if err := r.Patch(ctx, &ephemeralRunner, runner.MergeFrom()); err != nil {
log.Error(err, "Failed to update ephemeral runner after removing finalizer")
return ctrl.Result{}, err
}
}
if runnerID != 0 {
r.UnregistrationQueue.Push(&ephemeralRunner, runnerID)
}
log.Info("Removed the runner registration finalizer from ephemeral runner")
}
log.Info("Finalizing ephemeral runner")
err := r.cleanupResources(ctx, &ephemeralRunner, log)
if err != nil {
log.Error(err, "Failed to clean up ephemeral runner owned resources")
return ctrl.Result{}, err
}
if ephemeralRunner.HasContainerHookConfigured() {
log.Info("Runner has container hook configured, cleaning up container hook resources")
err = r.cleanupContainerHooksResources(ctx, &ephemeralRunner, log)
if err != nil {
log.Error(err, "Failed to clean up container hooks resources")
return ctrl.Result{}, err
}
}
log.Info("Removing finalizer")
if controllerutil.RemoveFinalizer(runner.Mutate(), ephemeralRunnerFinalizerName) || deferredActionsFinalizer {
log.Info("Removed finalizer from ephemeral runner")
if err := r.Patch(ctx, &ephemeralRunner, runner.MergeFrom()); client.IgnoreNotFound(err) != nil {
log.Error(err, "Failed to update ephemeral runner after removing finalizer")
return ctrl.Result{}, err
}
}
r.ResourceCache.Delete(&ephemeralRunner)
return ctrl.Result{}, nil
}
r.publishEphemeralRunnerPhaseMetric(&ephemeralRunner, ephemeralRunner.Status.Phase, log)
if ephemeralRunner.IsDone() {
log.Info("Cleaning up resources after after ephemeral runner termination", "phase", ephemeralRunner.Status.Phase)
// markAsFailed and markAsOutdated release the registration as they record
// the terminal phase, but the patch that does it can fail after the phase
// is already recorded, and a retry lands here rather than back in them.
// Repeated here so that error costs a reconcile instead of leaving the
// registration held until the set gets around to deleting the runner.
// Does nothing once the registration is released.
//
// A runner that deregistered itself holds nothing to release, so there is
// no error to recover from and nothing to queue. Releasing it here would
// only drop the finalizer, which the deletion below does anyway in a patch
// it already makes, at the cost of an extra write and the reconcile that
// write wakes.
if !runnerSelfDeregistered(&ephemeralRunner) {
if err := r.queueUnregistration(ctx, &ephemeralRunner, log); err != nil {
log.Error(err, "Failed to release the registration of a terminated ephemeral runner")
return ctrl.Result{}, err
}
}
err := r.cleanupResources(ctx, &ephemeralRunner, log)
if err != nil {
log.Error(err, "Failed to clean up ephemeral runner owned resources")
return ctrl.Result{}, err
}
// Stop reconciling on this object.
// The EphemeralRunnerSet is responsible for cleaning it up.
log.Info("EphemeralRunner has already finished. Stopping reconciliation and waiting for EphemeralRunnerSet to clean it up", "phase", ephemeralRunner.Status.Phase)
return ctrl.Result{}, nil
}
missingFinalizers := !controllerutil.ContainsFinalizer(&ephemeralRunner, ephemeralRunnerFinalizerName) ||
!controllerutil.ContainsFinalizer(&ephemeralRunner, ephemeralRunnerActionsFinalizerName)
if missingFinalizers {
log.Info("Adding finalizers")
controllerutil.AddFinalizer(runner.Mutate(), ephemeralRunnerFinalizerName)
controllerutil.AddFinalizer(runner.Mutate(), ephemeralRunnerActionsFinalizerName)
if err := r.Patch(ctx, &ephemeralRunner, runner.MergeFrom()); err != nil {
log.Error(err, "Failed to update with finalizer set")
return ctrl.Result{}, err
}
log.Info("Successfully added finalizers")
}
secret := new(corev1.Secret)
if err := r.Get(ctx, req.NamespacedName, secret); err != nil {
if !kerrors.IsNotFound(err) {
log.Error(err, "Failed to fetch secret")
return ctrl.Result{}, err
}
jitConfig, err := r.createRunnerJitConfig(ctx, &ephemeralRunner, log)
switch {
case err == nil:
// create secret if not created
log.Info("Creating new ephemeral runner secret for jitconfig.")
jitSecret, err := r.createSecret(ctx, &ephemeralRunner, jitConfig, log)
if err != nil {
return ctrl.Result{}, fmt.Errorf("failed to create secret: %w", err)
}
log.Info("Created new ephemeral runner secret for jitconfig.")
secret = jitSecret
case errors.Is(err, retryableError):
log.Info("Encountered retryable error, requeueing", "error", err.Error())
return ctrl.Result{RequeueAfter: 500 * time.Millisecond}, nil
case errors.Is(err, fatalError):
log.Info("JIT config cannot be created for this ephemeral runner, issuing delete", "error", err.Error())
if err := r.Delete(ctx, &ephemeralRunner); err != nil {
return ctrl.Result{}, fmt.Errorf("failed to delete the ephemeral runner: %w", err)
}
log.Info("Request to delete ephemeral runner has been issued")
return ctrl.Result{}, nil
default:
log.Error(err, "Failed to create ephemeral runners secret", "error", err.Error())
return ctrl.Result{}, err
}
}
var (
initialRunnerID int
initialRunnerName string
)
if ephemeralRunner.Status.RunnerID == 0 {
runnerID, err := runnerIDFromJITSecret(secret)
if err != nil {
log.Error(err, "Runner config secret contains an invalid runner ID")
// Replacing a secret already used by a pod could associate a new
// registration with a live runner. Only regenerate before it starts.
if r.APIReader == nil {
return ctrl.Result{}, fmt.Errorf("cannot safely replace jitconfig secret without APIReader: %w", err)
}
podErr := r.APIReader.Get(ctx, req.NamespacedName, new(corev1.Pod))
if podErr == nil {
return ctrl.Result{}, err
}
if !kerrors.IsNotFound(podErr) {
return ctrl.Result{}, fmt.Errorf("failed to check runner pod before replacing invalid jitconfig secret: %w", podErr)
}
log.Info("Deleting corrupted runner config secret")
if err := r.Delete(ctx, secret); err != nil {
return ctrl.Result{}, fmt.Errorf("failed to delete the corrupted runner config secret: %w", err)
}
log.Info("Corrupted runner config secret has been deleted")
return ctrl.Result{RequeueAfter: 500 * time.Millisecond}, nil
}
initialRunnerID = runnerID
initialRunnerName = string(secret.Data["runnerName"])
}
if len(ephemeralRunner.Status.Failures) > maxFailures {
log.Info(fmt.Sprintf("EphemeralRunner has failed more than %d times. Deleting ephemeral runner so it can be re-created", maxFailures))
if err := r.Delete(ctx, &ephemeralRunner); err != nil {
log.Error(fmt.Errorf("failed to delete ephemeral runner after %d failures: %w", maxFailures, err), "Failed to delete ephemeral runner")
return ctrl.Result{}, err
}
return ctrl.Result{}, nil
}
now := metav1.Now()
lastFailure := ephemeralRunner.Status.LastFailure()
backoffDuration := failedRunnerBackoff[len(ephemeralRunner.Status.Failures)]
nextReconciliation := lastFailure.Add(backoffDuration)
if !lastFailure.IsZero() && now.Before(&metav1.Time{Time: nextReconciliation}) {
requeueAfter := nextReconciliation.Sub(now.Time)
log.Info(
"Backing off the next reconciliation due to failure",
"lastFailure", lastFailure,
"nextReconciliation", nextReconciliation,
"requeueAfter", requeueAfter,
)
if requeueAfter <= 0 {
requeueAfter = time.Millisecond
}
return ctrl.Result{
RequeueAfter: requeueAfter,
}, nil
}
pod := new(corev1.Pod)
if err := r.Get(ctx, req.NamespacedName, pod); err != nil {
if !kerrors.IsNotFound(err) {
log.Error(err, "Failed to fetch the pod")
return ctrl.Result{}, err
}
log.Info("Ephemeral runner pod does not exist. Creating new ephemeral runner")
result, err := r.createPod(ctx, &ephemeralRunner, secret, log)
switch {
case err == nil:
return result, nil
case kerrors.IsAlreadyExists(err):
log.Info("Runner pod already exists. Waiting for the pod event to be received")
return ctrl.Result{RequeueAfter: 5 * time.Second}, nil
case kerrors.IsInvalid(err):
log.Error(err, "Failed to create a pod due to unrecoverable failure")
errMessage := fmt.Sprintf("Failed to create the pod: %v", err)
if err := r.markAsFailed(ctx, &ephemeralRunner, errMessage, ReasonInvalidPodFailure, log); err != nil {
log.Error(err, "Failed to set ephemeral runner to phase Failed")
return ctrl.Result{}, err
}
return ctrl.Result{}, nil
case kerrors.IsForbidden(err):
if status, ok := err.(kerrors.APIStatus); ok || errors.As(err, &status) {
isResourceQuotaExceeded := strings.Contains(status.Status().Message, "exceeded quota:")
isAboutToExpire := ephemeralRunner.CreationTimestamp.Time.Add(10 * time.Minute).Before(time.Now())
switch {
case isResourceQuotaExceeded && isAboutToExpire:
log.Error(err, "Failed to create a pod due to resource quota exceeded and the ephemeral runner is about to expire; re-creating the ephemeral runner")
if err := r.Delete(ctx, &ephemeralRunner); err != nil {
log.Error(err, "Failed to delete the ephemeral runner")
return ctrl.Result{}, err
}
return ctrl.Result{}, nil
case isResourceQuotaExceeded:
log.Error(err, "Resource quota is exceeded; requeue in 30s to retry pod creation")
return ctrl.Result{RequeueAfter: 30 * time.Second}, nil
default:
// other forbidden errors
// fallthrough to the default handling below
}
}
log.Error(err, "Failed to create a pod due to unrecoverable failure")
errMessage := fmt.Sprintf("Failed to create the pod: %v", err)
if err := r.markAsFailed(ctx, &ephemeralRunner, errMessage, ReasonInvalidPodFailure, log); err != nil {
log.Error(err, "Failed to set ephemeral runner to phase Failed")
return ctrl.Result{}, err
}
return ctrl.Result{}, nil
default:
log.Error(err, "Failed to create the pod")
return ctrl.Result{}, err
}
}
cs := runnerContainerStatus(pod)
switch {
case pod.Status.Phase == corev1.PodFailed: // All containers are stopped
log.Info(
"Pod is in failed phase, inspecting runner container status",
"podReason", pod.Status.Reason,
"podMessage", pod.Status.Message,
"podConditions", pod.Status.Conditions,
)
// If the runner pod did not have chance to start, terminated state may not be set.
// Therefore, we should try to restart it.
if cs == nil || cs.State.Terminated == nil {
log.Info("Runner container does not have state set, deleting pod as failed so it can be restarted")
return ctrl.Result{}, r.deleteEphemeralRunnerOrPod(ctx, &ephemeralRunner, pod, log)
}
switch cs.State.Terminated.ExitCode {
case 0:
log.Info("Runner container has succeeded but pod is in failed phase; Assume successful exit")
// If the pod is in a failed state, that means that at least one container exited with non-zero exit code.
// If the runner container exits with 0, we assume that the runner has finished successfully.
// If side-car container exits with non-zero, it shouldn't affect the runner. Runner exit code
// drives the controller's inference of whether the job has succeeded or failed.
if err := r.markAsSucceeded(ctx, &ephemeralRunner, pod, log); err != nil {
log.Error(err, "Failed to set ephemeral runner to phase Succeeded")
return ctrl.Result{}, err
}
if err := r.Delete(ctx, &ephemeralRunner); err != nil {
log.Error(err, "Failed to delete ephemeral runner after successful completion")
return ctrl.Result{}, err
}
return ctrl.Result{}, nil
case 7:
if err := r.markAsOutdated(ctx, &ephemeralRunner, log); err != nil {
log.Error(err, "Failed to set ephemeral runner to phase Outdated")
return ctrl.Result{}, err
}
return ctrl.Result{}, nil
}
log.Error(
errors.New("ephemeral runner container has failed, with runner container exit code non-zero"),
"Ephemeral runner container has failed, and runner container termination exit code is non-zero",
"containerTerminatedState", cs.State.Terminated,
)
return ctrl.Result{}, r.deleteEphemeralRunnerOrPod(ctx, &ephemeralRunner, pod, log)
case initContainerFailed(pod):
log.Info(
"Pod has a failed init container, deleting pod as failed so it can be restarted",
"initContainerStatuses", pod.Status.InitContainerStatuses,
)
return ctrl.Result{}, r.deleteEphemeralRunnerOrPod(ctx, &ephemeralRunner, pod, log)
case cs == nil:
// starting, no container state yet
log.Info("Waiting for runner container status to be available")
return ctrl.Result{}, nil
case cs.State.Terminated == nil: // container is not terminated and pod phase is not failed, so runner is still running
log.Info("Runner container is still running; updating ephemeral runner status")
if err := r.updateRunStatusFromPod(ctx, &ephemeralRunner, pod, initialRunnerID, initialRunnerName, log); err != nil {
log.Info("Failed to update ephemeral runner status. Requeue to not miss this event")
return ctrl.Result{}, err
}
return ctrl.Result{}, nil
case cs.State.Terminated.ExitCode == 7: // outdated
if err := r.markAsOutdated(ctx, &ephemeralRunner, log); err != nil {
log.Error(err, "Failed to set ephemeral runner to phase Outdated")
return ctrl.Result{}, err
}
return ctrl.Result{}, nil
case cs.State.Terminated.ExitCode != 0: // failed
log.Info("Ephemeral runner container failed", "exitCode", cs.State.Terminated.ExitCode)
return ctrl.Result{}, r.deleteEphemeralRunnerOrPod(ctx, &ephemeralRunner, pod, log)
default: // succeeded
log.Info("Ephemeral runner has finished successfully, deleting ephemeral runner", "exitCode", cs.State.Terminated.ExitCode)
if err := r.markAsSucceeded(ctx, &ephemeralRunner, pod, log); err != nil {
log.Error(err, "Failed to set ephemeral runner to phase Succeeded")
return ctrl.Result{}, err
}
if err := r.Delete(ctx, &ephemeralRunner); err != nil {
log.Error(err, "Failed to delete ephemeral runner after successful completion")
return ctrl.Result{}, err
}
return ctrl.Result{}, nil
}
}
func (r *EphemeralRunnerReconciler) deleteEphemeralRunnerOrPod(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, pod *corev1.Pod, log logr.Logger) error {
if ephemeralRunner.HasJob() {
log.Error(
errors.New("ephemeral runner has a job assigned, but the pod has failed"),
"Ephemeral runner either has faulty entrypoint or something external killing the runner",
)
log.Info("Deleting the ephemeral runner that has a job assigned but the pod has failed")
if err := r.Delete(ctx, ephemeralRunner); err != nil {
log.Error(err, "Failed to delete the ephemeral runner that has a job assigned but the pod has failed")
return err
}
// The runner is gone, and its pod failed with a job assigned, so the
// registration is still held. The delete above runs the finalizer, which
// queues its removal.
log.Info("Deleted the ephemeral runner that has a job assigned but the pod has failed")
return nil
}
if err := r.deletePodAsFailed(ctx, ephemeralRunner, pod, log); err != nil {
log.Error(err, "Failed to delete runner pod on failure")
return err
}
return nil
}
func (r *EphemeralRunnerReconciler) cleanupResources(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, log logr.Logger) error {
log.Info("Cleaning up the runner pod")
pod := new(corev1.Pod)
err := r.Get(ctx, types.NamespacedName{Namespace: ephemeralRunner.Namespace, Name: ephemeralRunner.Name}, pod)
switch {
case err == nil:
if pod.DeletionTimestamp.IsZero() {
log.Info("Deleting the runner pod")
if err := r.Delete(ctx, pod, r.deletePodOptions(pod)...); err != nil && !kerrors.IsNotFound(err) {
return fmt.Errorf("failed to delete pod: %w", err)
}
log.Info("Deleted the runner pod")
} else {
log.Info("Pod contains deletion timestamp")
}
case kerrors.IsNotFound(err):
log.Info("Runner pod is deleted")
default:
return err
}
log.Info("Cleaning up the runner jitconfig secret")
secret := new(corev1.Secret)
err = r.Get(ctx, types.NamespacedName{Namespace: ephemeralRunner.Namespace, Name: ephemeralRunner.Name}, secret)
switch {
case err == nil:
if secret.DeletionTimestamp.IsZero() {
log.Info("Deleting the jitconfig secret")
if err := r.Delete(ctx, secret); err != nil && !kerrors.IsNotFound(err) {
return fmt.Errorf("failed to delete secret: %w", err)
}
log.Info("Deleted jitconfig secret")
} else {
log.Info("Secret contains deletion timestamp")
}
case kerrors.IsNotFound(err):
log.Info("Runner jitconfig secret is deleted")
default:
return err
}
return nil
}
func (r *EphemeralRunnerReconciler) cleanupContainerHooksResources(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, log logr.Logger) error {
log.Info("Cleaning up runner linked pods")
var errs []error
if err := r.cleanupRunnerLinkedPods(ctx, ephemeralRunner, log); err != nil {
errs = append(errs, err)
}
log.Info("Cleaning up runner linked secrets")
if err := r.cleanupRunnerLinkedSecrets(ctx, ephemeralRunner, log); err != nil {
errs = append(errs, err)
}
return errors.Join(errs...)
}
func (r *EphemeralRunnerReconciler) cleanupRunnerLinkedPods(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, log logr.Logger) error {
runnerLinedLabels := client.MatchingLabels(
map[string]string{
"runner-pod": ephemeralRunner.Name,
},
)
var runnerLinkedPodList corev1.PodList
if err := r.List(ctx, &runnerLinkedPodList, client.InNamespace(ephemeralRunner.Namespace), runnerLinedLabels); err != nil {
return fmt.Errorf("failed to list runner-linked pods: %w", err)
}
if len(runnerLinkedPodList.Items) == 0 {
log.Info("Runner-linked pods are deleted")
return nil
}
log.Info("Deleting container hooks runner-linked pods", "count", len(runnerLinkedPodList.Items))
var errs []error
for i := range runnerLinkedPodList.Items {
linkedPod := &runnerLinkedPodList.Items[i]
if !linkedPod.DeletionTimestamp.IsZero() {
continue
}
log.Info("Deleting container hooks runner-linked pod", "name", linkedPod.Name)
if err := r.Delete(ctx, linkedPod, r.deletePodOptions(linkedPod)...); err != nil && !kerrors.IsNotFound(err) {
errs = append(errs, fmt.Errorf("failed to delete runner linked pod %q: %w", linkedPod.Name, err))
}
}
return errors.Join(errs...)
}
func (r *EphemeralRunnerReconciler) cleanupRunnerLinkedSecrets(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, log logr.Logger) error {
runnerLinkedLabels := client.MatchingLabels(
map[string]string{
"runner-pod": ephemeralRunner.Name,
},
)
var runnerLinkedSecretList corev1.SecretList
if err := r.List(ctx, &runnerLinkedSecretList, client.InNamespace(ephemeralRunner.Namespace), runnerLinkedLabels); err != nil {
return fmt.Errorf("failed to list runner-linked secrets: %w", err)
}
if len(runnerLinkedSecretList.Items) == 0 {
log.Info("Runner-linked secrets are deleted")
return nil
}
log.Info("Deleting container hooks runner-linked secrets", "count", len(runnerLinkedSecretList.Items))
var errs []error
for i := range runnerLinkedSecretList.Items {
s := &runnerLinkedSecretList.Items[i]
if !s.DeletionTimestamp.IsZero() {
continue
}
log.Info("Deleting container hooks runner-linked secret", "name", s.Name)
if err := r.Delete(ctx, s); err != nil && !kerrors.IsNotFound(err) {
errs = append(errs, fmt.Errorf("failed to delete runner linked secret %q: %w", s.Name, err))
}
}
return errors.Join(errs...)
}
func (r *EphemeralRunnerReconciler) markAsFailed(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, errMessage string, reason string, log logr.Logger) error {
log.Info("Updating ephemeral runner status to Failed")
original := ephemeralRunner.DeepCopy()
ephemeralRunner.Status.Phase = v1alpha1.EphemeralRunnerPhaseFailed
ephemeralRunner.Status.Reason = reason
ephemeralRunner.Status.Message = errMessage
if err := r.Status().Patch(ctx, ephemeralRunner, client.MergeFrom(original)); err != nil {
return fmt.Errorf("failed to update ephemeral runner status Phase/Message: %w", err)
}
r.publishEphemeralRunnerPhaseMetric(ephemeralRunner, ephemeralRunner.Status.Phase, log)
// A failed runner is not deleted here; it stays until the EphemeralRunnerSet
// cleans it up, which can be a long time, so the registration is released now
// rather than waiting for the finalizer.
if err := r.queueUnregistration(ctx, ephemeralRunner, log); err != nil {
return err
}
log.Info("EphemeralRunner is marked as Failed and queued for removal from the service")
return nil
}
// queueUnregistration releases the runner's registration with the Actions
// service: it hands the removal to the background workers and drops the
// finalizer that exists to make it happen.
//
// A runner that exited with code 0 deregistered itself, so it has nothing to
// hand over and only the finalizer goes.
//
// Dropping the finalizer is also what keeps this to a single removal. Without
// it the deletion that eventually follows would queue the same runner again.
// It doubles as the guard that makes this safe to call repeatedly: a runner
// whose registration is already released is left alone.
func (r *EphemeralRunnerReconciler) queueUnregistration(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, log logr.Logger) error {
if !controllerutil.ContainsFinalizer(ephemeralRunner, ephemeralRunnerActionsFinalizerName) {
return nil
}
var runnerID int
if runnerSelfDeregistered(ephemeralRunner) {
log.Info("Runner exited successfully and deregistered itself, skipping its removal from the service")
} else {
id, err := r.registeredRunnerID(ctx, ephemeralRunner, func() (multiclient.Client, error) {
return r.GetActionsService(ctx, ephemeralRunner)
}, log)
if err != nil {
return err
}
runnerID = id
}
original := ephemeralRunner.DeepCopy()
controllerutil.RemoveFinalizer(ephemeralRunner, ephemeralRunnerActionsFinalizerName)
if err := r.Patch(ctx, ephemeralRunner, client.MergeFrom(original)); err != nil && !kerrors.IsNotFound(err) {
return fmt.Errorf("failed to remove the runner registration finalizer: %w", err)
}
// Queued only once the finalizer is gone. A NotFound patch means another
// actor already removed it and the runner finished deletion, while any other
// failed patch leaves the removal to the retry rather than queueing it twice.
if runnerID != 0 {
r.UnregistrationQueue.Push(ephemeralRunner, runnerID)
}
return nil
}
func (r *EphemeralRunnerReconciler) markAsOutdated(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, log logr.Logger) error {
log.Info("Updating ephemeral runner status to Outdated")
original := ephemeralRunner.DeepCopy()
ephemeralRunner.Status.Phase = v1alpha1.EphemeralRunnerPhaseOutdated
ephemeralRunner.Status.Reason = "Outdated"
ephemeralRunner.Status.Message = "Runner is deprecated"
if err := r.Status().Patch(ctx, ephemeralRunner, client.MergeFrom(original)); err != nil {
return fmt.Errorf("failed to update ephemeral runner status Phase/Message: %w", err)
}
r.publishEphemeralRunnerPhaseMetric(ephemeralRunner, ephemeralRunner.Status.Phase, log)
// Queued rather than removed here, for the same reason as markAsFailed: an
// outdated runner waits on the EphemeralRunnerSet to delete it, and the
// phase transition has no reason to wait on the service.
if err := r.queueUnregistration(ctx, ephemeralRunner, log); err != nil {
return err
}
log.Info("EphemeralRunner is marked as Outdated and queued for removal from the service")
return nil
}
func (r *EphemeralRunnerReconciler) markAsSucceeded(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, pod *corev1.Pod, log logr.Logger) error {
log.Info("Updating ephemeral runner status to Succeeded")
original := ephemeralRunner.DeepCopy()
ephemeralRunner.Status.Phase = v1alpha1.EphemeralRunnerPhaseSucceeded
ephemeralRunner.Status.Ready = false
ephemeralRunner.Status.Reason = pod.Status.Reason
ephemeralRunner.Status.Message = pod.Status.Message
if err := r.Status().Patch(ctx, ephemeralRunner, client.MergeFrom(original)); err != nil {
return fmt.Errorf("failed to update ephemeral runner status Phase/Message: %w", err)
}
r.publishEphemeralRunnerPhaseMetric(ephemeralRunner, ephemeralRunner.Status.Phase, log)
log.Info("EphemeralRunner is marked as Succeeded")
return nil
}
// deletePodAsFailed is responsible for deleting the pod and updating the .Status.Failures for tracking failure count.
// It should not be responsible for setting the status to Failed.
//
// It should be called by deleteEphemeralRunnerOrPod which is responsible for deciding whether to delete the EphemeralRunner or just the Pod.
func (r *EphemeralRunnerReconciler) deletePodAsFailed(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, pod *corev1.Pod, log logr.Logger) error {
if pod.DeletionTimestamp.IsZero() {
log.Info("Deleting the ephemeral runner pod", "podId", pod.UID)
if err := r.Delete(ctx, pod, r.deletePodOptions(pod)...); err != nil && !kerrors.IsNotFound(err) {
return fmt.Errorf("failed to delete pod with status failed: %w", err)
}
}
log.Info("Updating ephemeral runner status to track the failure count")
original := ephemeralRunner.DeepCopy()
if ephemeralRunner.Status.Failures == nil {
ephemeralRunner.Status.Failures = make(map[string]metav1.Time)
}
ephemeralRunner.Status.Failures[string(pod.UID)] = metav1.Now()
ephemeralRunner.Status.Ready = false
ephemeralRunner.Status.Reason = pod.Status.Reason
ephemeralRunner.Status.Message = pod.Status.Message
if err := r.Status().Patch(ctx, ephemeralRunner, client.MergeFrom(original)); err != nil {
return fmt.Errorf("failed to update ephemeral runner status with failure count: %w", err)
}
log.Info("EphemeralRunner pod is deleted and status is updated with failure count")
return nil
}
func (r *EphemeralRunnerReconciler) createRunnerJitConfig(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, log logr.Logger) (*scaleset.RunnerScaleSetJitRunnerConfig, error) {
// Runner is not registered with the service. We need to register it first
log.Info("Creating ephemeral runner JIT config")
actionsClient, err := r.GetActionsService(ctx, ephemeralRunner)
if err != nil {
return nil, fmt.Errorf("failed to get actions client for generating JIT config: %w", err)
}
jitSettings := &scaleset.RunnerScaleSetJitRunnerSetting{
Name: ephemeralRunner.Name,
}
for i := range ephemeralRunner.Spec.Spec.Containers {
if ephemeralRunner.Spec.Spec.Containers[i].Name == v1alpha1.EphemeralRunnerContainerName &&
ephemeralRunner.Spec.Spec.Containers[i].WorkingDir != "" {
jitSettings.WorkFolder = ephemeralRunner.Spec.Spec.Containers[i].WorkingDir
}
}
jitConfig, err := actionsClient.GenerateJitRunnerConfig(ctx, jitSettings, ephemeralRunner.Spec.RunnerScaleSetID)
if err == nil { // if NO error
log.Info("Created ephemeral runner JIT config", "runnerId", jitConfig.Runner.ID)
return jitConfig, nil
}
if !errors.Is(err, scaleset.RunnerExistsError) {
return nil, fmt.Errorf("failed to generate JIT config with generic error: %w", err)
}
// If the runner with the name we want already exists it means:
// - We might have a name collision.
// - Our previous reconciliation loop failed to update the
// status with the runnerId and runnerJITConfig after the `GenerateJitRunnerConfig`
// created the runner registration on the service.
// We will try to get the runner and see if it's belong to this AutoScalingRunnerSet,
// if so, we can simply delete the runner registration and create a new one.
log.Info("Getting runner jit config failed with conflict error, trying to get the runner by name", "runnerName", ephemeralRunner.Name)
existingRunner, err := actionsClient.GetRunnerByName(ctx, ephemeralRunner.Name)
if err != nil {
return nil, fmt.Errorf("failed to get runner by name: %w", err)
}
if existingRunner == nil {
log.Info("Runner with the same name does not exist anymore, re-queuing the reconciliation")
return nil, fmt.Errorf("%w: runner existed, retry configuration", retryableError)
}
log.Info("Found the runner with the same name", "runnerId", existingRunner.ID, "runnerScaleSetId", existingRunner.RunnerScaleSetID)
if existingRunner.RunnerScaleSetID == ephemeralRunner.Spec.RunnerScaleSetID {
log.Info("Removing the runner with the same name")
err := actionsClient.RemoveRunner(ctx, int64(existingRunner.ID))
if err != nil {
return nil, fmt.Errorf("failed to remove runner from the service: %w", err)
}
log.Info("Removed the runner with the same name, re-queuing the reconciliation")
return nil, fmt.Errorf("%w: runner existed belonging to the scale set, retry configuration", retryableError)
}
return nil, fmt.Errorf("%w: runner with the same name but doesn't belong to this RunnerScaleSet: %w", fatalError, err)
}
func (r *EphemeralRunnerReconciler) createPod(ctx context.Context, runner *v1alpha1.EphemeralRunner, secret *corev1.Secret, log logr.Logger) (ctrl.Result, error) {
var envs []corev1.EnvVar
if runner.Spec.ProxySecretRef != "" {
http := corev1.EnvVar{
Name: "http_proxy",
ValueFrom: &corev1.EnvVarSource{
SecretKeyRef: &corev1.SecretKeySelector{
LocalObjectReference: corev1.LocalObjectReference{
Name: runner.Spec.ProxySecretRef,
},
Key: "http_proxy",
},
},
}
if runner.Spec.Proxy.HTTP != nil {
envs = append(envs, http)
}
https := corev1.EnvVar{
Name: "https_proxy",
ValueFrom: &corev1.EnvVarSource{
SecretKeyRef: &corev1.SecretKeySelector{
LocalObjectReference: corev1.LocalObjectReference{
Name: runner.Spec.ProxySecretRef,
},
Key: "https_proxy",
},
},
}
if runner.Spec.Proxy.HTTPS != nil {
envs = append(envs, https)
}
noProxy := corev1.EnvVar{
Name: "no_proxy",
ValueFrom: &corev1.EnvVarSource{
SecretKeyRef: &corev1.SecretKeySelector{
LocalObjectReference: corev1.LocalObjectReference{
Name: runner.Spec.ProxySecretRef,
},
Key: "no_proxy",
},
},
}
if len(runner.Spec.Proxy.NoProxy) > 0 {
envs = append(envs, noProxy)
}
}
log.Info("Creating new pod for ephemeral runner")
newPod, err := r.newEphemeralRunnerPod(runner, secret, envs...)
if err != nil {
log.Error(err, "Failed to build new pod")
return ctrl.Result{}, err
}
log.Info("Created new pod spec for ephemeral runner")
if err := r.Create(ctx, newPod); err != nil {
log.Error(err, "Failed to create pod resource for ephemeral runner.")
return ctrl.Result{}, err
}
log.Info("Created ephemeral runner pod",
"runnerScaleSetId", runner.Spec.RunnerScaleSetID,
"runnerName", runner.Status.RunnerName,
"runnerId", runner.Status.RunnerID,
"configUrl", runner.Spec.GitHubConfigURL,
"podName", newPod.Name)
return ctrl.Result{}, nil
}
func (r *EphemeralRunnerReconciler) createSecret(ctx context.Context, runner *v1alpha1.EphemeralRunner, jitConfig *scaleset.RunnerScaleSetJitRunnerConfig, log logr.Logger) (*corev1.Secret, error) {
log.Info("Creating new secret for ephemeral runner")
jitSecret, err := r.newEphemeralRunnerJitSecret(runner, jitConfig)
if err != nil {
return nil, fmt.Errorf("failed to build jit secret: %w", err)
}
log.Info("Created new secret spec for ephemeral runner")
if err := r.Create(ctx, jitSecret); err != nil {
return nil, fmt.Errorf("failed to create jit secret: %w", err)
}
log.Info("Created ephemeral runner secret", "secretName", jitSecret.Name)
return jitSecret, nil
}
// updateRunStatusFromPod is responsible for updating non-terminal statuses.
// It should never update phase to Failed or Succeeded.
//
// The JIT config secret is the durable registration record until the Pod first
// reports a non-terminal status. Publishing identity with that status update
// avoids a separate status-only reconciliation after Pod creation.
func (r *EphemeralRunnerReconciler) updateRunStatusFromPod(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, pod *corev1.Pod, initialRunnerID int, initialRunnerName string, log logr.Logger) error {
if pod.Status.Phase == corev1.PodSucceeded || pod.Status.Phase == corev1.PodFailed {
return nil
}
ready := podReady(pod)
// Publish Pending as soon as the runner is observed non-terminal, regardless of
// the pod phase. The controller only reaches this point once the runner
// container status exists, and by then the pod has usually already advanced to
// Running, so keying the initial phase off PodPending would leave a runner
// phase-empty for its whole life -- omitted from the phase metrics, and in
// breach of the documented contract that Pending means "created, no job yet".
// Guarding on the empty phase alone is sufficient: every terminal phase, and
// Running itself, is non-empty, so this can never overwrite one.
phase := ephemeralRunner.Status.Phase
if phase == "" {
phase = v1alpha1.EphemeralRunnerPhasePending
}
// The listener writes only job metadata. The runner controller owns phase
// transitions and promotes an assigned Pending runner to Running without
// racing the listener's status patch.
if phase == v1alpha1.EphemeralRunnerPhasePending && ephemeralRunner.HasJob() {
phase = v1alpha1.EphemeralRunnerPhaseRunning
}
phaseChanged := phase != ephemeralRunner.Status.Phase
readyChanged := ready != ephemeralRunner.Status.Ready
identityChanged := ephemeralRunner.Status.RunnerID == 0
if !phaseChanged && !readyChanged && !identityChanged {
return nil
}
log.Info(
"Updating ephemeral runner status",
"statusPhase", pod.Status.Phase,
"statusReason", pod.Status.Reason,
"statusMessage", pod.Status.Message,
"ready", ready,
)
original := ephemeralRunner.DeepCopy()
ephemeralRunner.Status.Phase = phase
ephemeralRunner.Status.Ready = ready
ephemeralRunner.Status.Reason = pod.Status.Reason
ephemeralRunner.Status.Message = pod.Status.Message
if identityChanged {
ephemeralRunner.Status.RunnerID = initialRunnerID
ephemeralRunner.Status.RunnerName = initialRunnerName
}
if err := r.Status().Patch(ctx, ephemeralRunner, client.MergeFrom(original)); err != nil {
return fmt.Errorf("failed to update runner status for Phase/Reason/Message/Ready: %w", err)
}
r.publishEphemeralRunnerPhaseMetric(ephemeralRunner, ephemeralRunner.Status.Phase, log)
log.Info("Updated ephemeral runner status")
return nil
}
func (r *EphemeralRunnerReconciler) publishEphemeralRunnerPhaseMetric(ephemeralRunner *v1alpha1.EphemeralRunner, phase v1alpha1.EphemeralRunnerPhase, log logr.Logger) {
if !r.PublishMetrics {
return
}
commonLabels, err := ephemeralRunnerMetricLabels(ephemeralRunner)
if err != nil {
log.Error(err, "Failed to build ephemeral runner metric labels")
return
}
key := types.NamespacedName{Namespace: ephemeralRunner.Namespace, Name: ephemeralRunner.Name}
ephemeralRunnerPhaseMetrics.Lock()
defer ephemeralRunnerPhaseMetrics.Unlock()
previousPhase, ok := ephemeralRunnerPhaseMetrics.phases[key]
if ok && previousPhase == phase {
return
}
if ok {
metrics.SubEphemeralRunner(commonLabels, previousPhase)
}
if phase == "" {
delete(ephemeralRunnerPhaseMetrics.phases, key)
return
}
metrics.AddEphemeralRunner(commonLabels, phase)
ephemeralRunnerPhaseMetrics.phases[key] = phase
}
func ephemeralRunnerMetricLabels(ephemeralRunner *v1alpha1.EphemeralRunner) (metrics.CommonLabels, error) {
parsedURL, err := actions.ParseGitHubConfigFromURL(ephemeralRunner.Spec.GitHubConfigURL)
if err != nil {
return metrics.CommonLabels{}, fmt.Errorf("github config URL is invalid: %w", err)
}
return metrics.CommonLabels{
Name: ephemeralRunner.Labels[LabelKeyGitHubScaleSetName],
Namespace: ephemeralRunner.Labels[LabelKeyGitHubScaleSetNamespace],
Repository: parsedURL.Repository,
Organization: parsedURL.Organization,
Enterprise: parsedURL.Enterprise,
}, nil
}
// registeredRunnerID returns the ID of the registration the runner holds with
// the Actions service, or 0 when it never got one.
//
// Callers decide whether a removal is needed at all; this only names the
// registration to remove. See runnerSelfDeregistered for the runners that do
// not need one.
//
// The common answer comes from the runner's own status and costs nothing. The
// exception is a runner whose status never recorded an ID: the registration is
// created by GenerateJitRunnerConfig, and the ID it returns reaches the
// jitconfig secret before the status patch that publishes it. A runner deleted
// in that window holds a registration the status cannot name, so the secret is
// read to recover it. That read only happens for a runner that got that far and
// no further, never on the path a finishing job takes.
//
// A secret that cannot be read is an error rather than an answer. If it is
// absent, the service is checked by name before concluding the runner was
// never registered: GenerateJitRunnerConfig can register it just before
// createSecret persists the ID. Any other uncertainty leaves the question
// open, because answering 0 would drop the finalizer and lose the last record
// of a registration that does exist. Invalid IDs are errors, not evidence that
// the runner was never registered.
func (r *EphemeralRunnerReconciler) registeredRunnerID(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, getActionsClient func() (multiclient.Client, error), log logr.Logger) (int, error) {
if ephemeralRunner.Status.RunnerID < 0 {
return 0, fmt.Errorf("invalid runner ID in status: %d", ephemeralRunner.Status.RunnerID)
}
if ephemeralRunner.Status.RunnerID > 0 {
return ephemeralRunner.Status.RunnerID, nil
}
secret := new(corev1.Secret)
if err := r.Get(ctx, types.NamespacedName{Namespace: ephemeralRunner.Namespace, Name: ephemeralRunner.Name}, secret); err != nil {
if !kerrors.IsNotFound(err) {
return 0, fmt.Errorf("failed to read the jitconfig secret of a runner without a recorded ID: %w", err)
}
actionsClient, err := getActionsClient()
if err != nil {
return 0, fmt.Errorf("failed to get actions client for a runner without a recorded ID or jitconfig secret: %w", err)
}
existingRunner, err := actionsClient.GetRunnerByName(ctx, ephemeralRunner.Name)
if err != nil {
return 0, fmt.Errorf("failed to get runner by name for a runner without a recorded ID or jitconfig secret: %w", err)
}
if existingRunner == nil {
log.Info("No runner registration found for a runner without a recorded ID or jitconfig secret")
return 0, nil
}
if existingRunner.RunnerScaleSetID != ephemeralRunner.Spec.RunnerScaleSetID {
return 0, fmt.Errorf(
"runner registration %d found by name belongs to runner scale set %d, expected %d",
existingRunner.ID,
existingRunner.RunnerScaleSetID,
ephemeralRunner.Spec.RunnerScaleSetID,
)
}
if existingRunner.ID <= 0 {
return 0, fmt.Errorf("invalid runner ID returned by the Actions service: %d", existingRunner.ID)
}
log.Info("Recovered the runner ID from the Actions service", "runnerId", existingRunner.ID)
return existingRunner.ID, nil
}
runnerID, err := runnerIDFromJITSecret(secret)
if err != nil {
return 0, err
}
log.Info("Recovered the runner ID from the jitconfig secret", "runnerId", runnerID)
return runnerID, nil
}
func runnerIDFromJITSecret(secret *corev1.Secret) (int, error) {
runnerID, err := strconv.Atoi(string(secret.Data["runnerId"]))
if err != nil {
return 0, fmt.Errorf("invalid runner ID in jitconfig secret: %w", err)
}
if runnerID <= 0 {
return 0, fmt.Errorf("invalid runner ID in jitconfig secret: %d", runnerID)
}
return runnerID, nil
}
// removeRunnerOfLivePod removes the registration of a runner whose pod may
// still be executing a job, reporting whether it did. A runner without such a
// pod is left alone and reported as not removed.
//
// The service refuses to remove a runner that is executing a job, and that
// refusal is returned as scaleset.JobStillRunningError so the pod can be kept.
// Nothing here waits on the service when the pod is gone, being deleted, or
// has nothing left running.
func (r *EphemeralRunnerReconciler) removeRunnerOfLivePod(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, runnerID int, getActionsClient func() (multiclient.Client, error), log logr.Logger) (bool, error) {
if r.APIReader == nil {
return false, errors.New("APIReader is not configured, cannot confirm the runner pod state without reading through the cache")
}
// The cache can miss a newly created pod or still show its terminated
// predecessor. Neither is safe evidence for dropping finalizer protection.
pod := new(corev1.Pod)
if err := r.APIReader.Get(ctx, types.NamespacedName{Namespace: ephemeralRunner.Namespace, Name: ephemeralRunner.Name}, pod); err != nil {
if kerrors.IsNotFound(err) {
return false, nil
}
return false, fmt.Errorf("failed to get the runner pod: %w", err)
}
if !pod.DeletionTimestamp.IsZero() || podTerminated(pod) {
return false, nil
}
actionsClient, err := getActionsClient()
if err != nil {
return false, fmt.Errorf("failed to get actions client: %w", err)
}
if err := actionsClient.RemoveRunner(ctx, int64(runnerID)); err != nil {
if !errors.Is(err, scaleset.RunnerNotFoundError) && !errors.Is(err, scaleset.NotFoundError) {
return false, err
}
log.Info("Runner is already removed from the service", "runnerId", runnerID)
}
return true, nil
}
// SetupWithManager sets up the controller with the Manager.
func (r *EphemeralRunnerReconciler) SetupWithManager(mgr ctrl.Manager, opts ...Option) error {
r.ResourceBuilder.setSchemeIfUnset(r.Scheme)
if r.APIReader == nil {
r.APIReader = mgr.GetAPIReader()
}
if r.UnregistrationQueue == nil {
r.UnregistrationQueue = NewRunnerUnregistrationQueue(
r.Log.WithName("runner-unregistration"),
r.SecretResolver,
0,
)
if err := mgr.Add(r.UnregistrationQueue); err != nil {
return fmt.Errorf("failed to add the runner unregistration workers to the manager: %w", err)
}
}
return builderWithOptions(
ctrl.NewControllerManagedBy(mgr).
For(&v1alpha1.EphemeralRunner{}, builder.WithPredicates(ephemeralRunnerPredicate())).
Owns(&corev1.Pod{}, builder.WithPredicates(ephemeralRunnerOwnedPodPredicate())).
WithEventFilter(predicate.ResourceVersionChangedPredicate{}),
opts,
).Complete(r)
}
// podTerminated reports whether every container in the pod has stopped.
//
// No container the kubelet has reported on may still be running. That check is
// made whatever the pod phase says, because the phase is not always the
// kubelet's account of the containers: a pod is moved to Failed by the control
// plane when its node is lost or shut down, while the last status the kubelet
// managed to send still shows a container running on the other side of the
// partition. Believing the phase there would drop the pod out of the API while
// something is still alive under it.
//
// The phase is what says whether the containers that have not reported are
// still to come. A pod that has reached Succeeded or Failed is not going to
// start anything else, so a container missing from the status is one that never
// ran, which is how a pod whose init container failed is still terminated. Short
// of a terminal phase every container has to have reported, or the runner that
// is about to be reported as started would be missed.
//
// Native sidecars run as init containers that outlive the regular ones, so they
// are checked too. A pod still running one of those, or a legacy sidecar
// alongside the runner, is not terminated no matter what the runner container
// did.
func podTerminated(pod *corev1.Pod) bool {
for i := range pod.Status.ContainerStatuses {
if pod.Status.ContainerStatuses[i].State.Terminated == nil {
return false
}
}
for i := range pod.Status.InitContainerStatuses {
if pod.Status.InitContainerStatuses[i].State.Terminated == nil {
return false
}
}
switch pod.Status.Phase {
case corev1.PodSucceeded, corev1.PodFailed:
return true
}
return len(pod.Status.ContainerStatuses) == len(pod.Spec.Containers)
}
// deletePodOptions asks for an immediate deletion of a pod that has nothing
// left running in it.
//
// A graceful deletion exists to give containers their terminationGracePeriod to
// shut down, and the API object survives until the kubelet reports that they
// have. For a pod whose containers have all terminated there is nothing to
// shut down and nothing to protect: the grace period is spent waiting on the
// kubelet to finish unmounting volumes and tearing down the sandbox, which it
// does whether or not the object is still there.
//
// That wait is what fills a cluster with Terminating runner pods during a burst
// of jobs. They hold their name, their scheduling slot, and their share of any
// ResourceQuota, so the runners waiting to replace them cannot start. Dropping
// the object as the delete is issued hands those back immediately.
//
// How long to wait is TerminatedPodGracePeriodSeconds, zero by default. A
// negative value leaves the deletion alone, which restores whatever the pod
// asks for in its own spec.
//
// A pod that is still running is deleted normally. Skipping the grace period
// there would drop the object while its containers were still alive, leaving
// the kubelet to kill them with nothing in the API to account for the resources
// they hold in the meantime.
//
// The deletion is pinned to the pod the decision was made about. Pods are read
// through the informer cache and every generation of a runner's pod carries the
// same name, so a delete by name is a delete of whatever holds that name when
// the API server reads the request, not of the pod whose containers were
// observed to have stopped. A single controller cannot get that wrong, since it
// is the only thing creating that name and it only creates after a read says the
// name is free, but that argument is worth exactly as much as the single writer
// it assumes: during a leader election handover the outgoing leader's reconcile
// is still in flight while the new leader is already replacing pods. Naming the
// UID turns the delete into a conflict when it lands on a pod the controller
// never looked at. Without the grace period there is nothing to catch it
// afterwards: the pod would be gone the moment the request was accepted,
// killing a job instead of handing its container a SIGTERM to deregister with.
func (r *EphemeralRunnerReconciler) deletePodOptions(pod *corev1.Pod) []client.DeleteOption {
if !podTerminated(pod) || r.TerminatedPodGracePeriodSeconds < 0 {
return nil
}
opts := []client.DeleteOption{client.GracePeriodSeconds(r.TerminatedPodGracePeriodSeconds)}
if pod.UID != "" {
uid := pod.UID
opts = append(opts, client.Preconditions{UID: &uid})
}
return opts
}
func runnerContainerStatus(pod *corev1.Pod) *corev1.ContainerStatus {
for i := range pod.Status.ContainerStatuses {
cs := &pod.Status.ContainerStatuses[i]
if cs.Name == v1alpha1.EphemeralRunnerContainerName {
return cs
}
}
return nil
}
func initContainerFailed(pod *corev1.Pod) bool {
for i := range pod.Status.InitContainerStatuses {
cs := &pod.Status.InitContainerStatuses[i]
if cs.State.Terminated != nil && cs.State.Terminated.ExitCode != 0 {
return true
}
}
return false
}