/* Copyright 2020 The actions-runner-controller authors. Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file except in compliance with the License. You may obtain a copy of the License at http://www.apache.org/licenses/LICENSE-2.0 Unless required by applicable law or agreed to in writing, software distributed under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the License for the specific language governing permissions and limitations under the License. */ package actionsgithubcom import ( "context" "errors" "fmt" "strconv" "strings" "sync" "time" "github.com/actions/actions-runner-controller/apis/actions.github.com/v1alpha1" "github.com/actions/actions-runner-controller/controllers/actions.github.com/metrics" "github.com/actions/actions-runner-controller/github/actions" "github.com/actions/scaleset" "github.com/go-logr/logr" corev1 "k8s.io/api/core/v1" kerrors "k8s.io/apimachinery/pkg/api/errors" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/runtime" "k8s.io/apimachinery/pkg/types" ctrl "sigs.k8s.io/controller-runtime" "sigs.k8s.io/controller-runtime/pkg/builder" "sigs.k8s.io/controller-runtime/pkg/client" "sigs.k8s.io/controller-runtime/pkg/controller/controllerutil" "sigs.k8s.io/controller-runtime/pkg/predicate" ) const ( ephemeralRunnerFinalizerName = "ephemeralrunner.actions.github.com/finalizer" ephemeralRunnerActionsFinalizerName = "ephemeralrunner.actions.github.com/runner-registration-finalizer" ) // EphemeralRunnerReconciler reconciles a EphemeralRunner object type EphemeralRunnerReconciler struct { client.Client Log logr.Logger Scheme *runtime.Scheme PublishMetrics bool // UnregistrationQueue takes the removal of runner registrations from the // Actions service off the reconcile path. When it is left unset, // SetupWithManager creates one and registers it with the manager. UnregistrationQueue *RunnerUnregistrationQueue ResourceBuilder } var ephemeralRunnerPhaseMetrics = struct { sync.Mutex phases map[types.NamespacedName]v1alpha1.EphemeralRunnerPhase }{ phases: map[types.NamespacedName]v1alpha1.EphemeralRunnerPhase{}, } // precompute backoff durations for failed ephemeral runners // the len(failedRunnerBackoff) must be equal to maxFailures + 1 var failedRunnerBackoff = []time.Duration{ 0, 5 * time.Second, 10 * time.Second, 20 * time.Second, 40 * time.Second, 80 * time.Second, } const maxFailures = 5 // +kubebuilder:rbac:groups=actions.github.com,resources=ephemeralrunners,verbs=get;list;watch;create;update;patch;delete // +kubebuilder:rbac:groups=actions.github.com,resources=ephemeralrunners/status,verbs=get;update;patch // +kubebuilder:rbac:groups=actions.github.com,resources=ephemeralrunners/finalizers,verbs=get;list;watch;create;update;patch;delete // +kubebuilder:rbac:groups=core,resources=pods,verbs=get;list;watch;create;update;patch;delete // +kubebuilder:rbac:groups=core,resources=pods/status,verbs=get // +kubebuilder:rbac:groups=core,resources=secrets,verbs=create;get;list;watch;delete // Reconcile is part of the main kubernetes reconciliation loop which aims to // move the current state of the cluster closer to the desired state. // // For more details, check Reconcile and its Result here: // - https://pkg.go.dev/sigs.k8s.io/controller-runtime@v0.6.4/pkg/reconcile func (r *EphemeralRunnerReconciler) Reconcile(ctx context.Context, req ctrl.Request) (ctrl.Result, error) { log := r.Log.WithValues("ephemeralrunner", req.NamespacedName) var ephemeralRunner v1alpha1.EphemeralRunner if err := r.Get(ctx, req.NamespacedName, &ephemeralRunner); err != nil { return ctrl.Result{}, client.IgnoreNotFound(err) } runner := newLazyCopy(&ephemeralRunner) if !ephemeralRunner.DeletionTimestamp.IsZero() { r.publishEphemeralRunnerPhaseMetric(&ephemeralRunner, "", log) if !controllerutil.ContainsFinalizer(&ephemeralRunner, ephemeralRunnerFinalizerName) { return ctrl.Result{}, nil } deferredActionsFinalizer := false if controllerutil.ContainsFinalizer(&ephemeralRunner, ephemeralRunnerActionsFinalizerName) { // This finalizer exists to release the runner's registration with the // Actions service. There are two ways that happens. // // A runner that exited with code 0 already removed its own // registration on the way out. Runners are ephemeral, so a clean exit // means the agent deregistered itself before it stopped, and there is // nothing left to ask the service to remove. That is the path every // completed job takes, and it costs no API call at all. // // Every other runner may still hold a registration. Removing it is // handed to background workers rather than done here, so that deleting // the pod and the secret below is never held up by an external API. // See RunnerUnregistrationQueue for what that costs. // // Queueing is also what stops holding the pod alive when the service // reports that the runner is still executing a job. That used to keep // the runner pod of a job that is still running from being deleted out // from under it, but only for deletions that reach this branch // directly. The EphemeralRunnerSet does not rely on it: it refuses to // delete a runner that has a job assigned, and removes a runner from // the service before deleting it when it scales down. What is left is // an EphemeralRunner deleted by hand, and there the deletion is taken // at face value: the pod goes now, and the workers keep retrying the // removal until the service accepts it. var runnerID int if runnerSelfDeregistered(&ephemeralRunner) { log.Info("Runner exited successfully and deregistered itself, skipping its removal from the service") } else { // Resolved before the finalizer goes, because recovering an ID the // status never recorded reads the jitconfig secret, which the // cleanup below deletes. id, err := r.registeredRunnerID(ctx, &ephemeralRunner, log) if err != nil { log.Error(err, "Failed to resolve the registration of an ephemeral runner being deleted") return ctrl.Result{}, err } runnerID = id } log.Info( "Removing the runner registration finalizer", "unregisterFromService", runnerID != 0, "phase", ephemeralRunner.Status.Phase, ) removedActionsFinalizer := controllerutil.RemoveFinalizer(runner.Mutate(), ephemeralRunnerActionsFinalizerName) // The patch has to land before the runner is queued: queueing first // would ask the service to remove the same runner twice when the patch // fails and the reconcile comes back through this branch. A runner with // nothing to queue has nothing to order the patch against, so its // removal rides along with the finalizer patch made after cleanup // below rather than paying for a round trip of its own. deferredActionsFinalizer = removedActionsFinalizer && runnerID == 0 if removedActionsFinalizer && !deferredActionsFinalizer { if err := r.Patch(ctx, &ephemeralRunner, runner.MergeFrom()); err != nil { log.Error(err, "Failed to update ephemeral runner after removing finalizer") return ctrl.Result{}, err } } if runnerID != 0 { r.UnregistrationQueue.Push(&ephemeralRunner, runnerID) } log.Info("Removed the runner registration finalizer from ephemeral runner") } log.Info("Finalizing ephemeral runner") err := r.cleanupResources(ctx, &ephemeralRunner, log) if err != nil { log.Error(err, "Failed to clean up ephemeral runner owned resources") return ctrl.Result{}, err } if ephemeralRunner.HasContainerHookConfigured() { log.Info("Runner has container hook configured, cleaning up container hook resources") err = r.cleanupContainerHooksResources(ctx, &ephemeralRunner, log) if err != nil { log.Error(err, "Failed to clean up container hooks resources") return ctrl.Result{}, err } } log.Info("Removing finalizer") if controllerutil.RemoveFinalizer(runner.Mutate(), ephemeralRunnerFinalizerName) || deferredActionsFinalizer { log.Info("Removed finalizer from ephemeral runner") if err := r.Patch(ctx, &ephemeralRunner, runner.MergeFrom()); client.IgnoreNotFound(err) != nil { log.Error(err, "Failed to update ephemeral runner after removing finalizer") return ctrl.Result{}, err } } r.ResourceCache.Delete(&ephemeralRunner) return ctrl.Result{}, nil } r.publishEphemeralRunnerPhaseMetric(&ephemeralRunner, ephemeralRunner.Status.Phase, log) if ephemeralRunner.IsDone() { log.Info("Cleaning up resources after after ephemeral runner termination", "phase", ephemeralRunner.Status.Phase) // markAsFailed and markAsOutdated release the registration as they record // the terminal phase, but the patch that does it can fail after the phase // is already recorded, and a retry lands here rather than back in them. // Repeated here so that error costs a reconcile instead of leaving the // registration held until the set gets around to deleting the runner. // Does nothing once the registration is released. // // A runner that deregistered itself holds nothing to release, so there is // no error to recover from and nothing to queue. Releasing it here would // only drop the finalizer, which the deletion below does anyway in a patch // it already makes, at the cost of an extra write and the reconcile that // write wakes. if !runnerSelfDeregistered(&ephemeralRunner) { if err := r.queueUnregistration(ctx, &ephemeralRunner, log); err != nil { log.Error(err, "Failed to release the registration of a terminated ephemeral runner") return ctrl.Result{}, err } } err := r.cleanupResources(ctx, &ephemeralRunner, log) if err != nil { log.Error(err, "Failed to clean up ephemeral runner owned resources") return ctrl.Result{}, err } // Stop reconciling on this object. // The EphemeralRunnerSet is responsible for cleaning it up. log.Info("EphemeralRunner has already finished. Stopping reconciliation and waiting for EphemeralRunnerSet to clean it up", "phase", ephemeralRunner.Status.Phase) return ctrl.Result{}, nil } missingFinalizers := !controllerutil.ContainsFinalizer(&ephemeralRunner, ephemeralRunnerFinalizerName) || !controllerutil.ContainsFinalizer(&ephemeralRunner, ephemeralRunnerActionsFinalizerName) if missingFinalizers { log.Info("Adding finalizers") controllerutil.AddFinalizer(runner.Mutate(), ephemeralRunnerFinalizerName) controllerutil.AddFinalizer(runner.Mutate(), ephemeralRunnerActionsFinalizerName) if err := r.Patch(ctx, &ephemeralRunner, runner.MergeFrom()); err != nil { log.Error(err, "Failed to update with finalizer set") return ctrl.Result{}, err } log.Info("Successfully added finalizers") } secret := new(corev1.Secret) if err := r.Get(ctx, req.NamespacedName, secret); err != nil { if !kerrors.IsNotFound(err) { log.Error(err, "Failed to fetch secret") return ctrl.Result{}, err } jitConfig, err := r.createRunnerJitConfig(ctx, &ephemeralRunner, log) switch { case err == nil: // create secret if not created log.Info("Creating new ephemeral runner secret for jitconfig.") jitSecret, err := r.createSecret(ctx, &ephemeralRunner, jitConfig, log) if err != nil { return ctrl.Result{}, fmt.Errorf("failed to create secret: %w", err) } log.Info("Created new ephemeral runner secret for jitconfig.") secret = jitSecret case errors.Is(err, retryableError): log.Info("Encountered retryable error, requeueing", "error", err.Error()) return ctrl.Result{Requeue: true}, nil case errors.Is(err, fatalError): log.Info("JIT config cannot be created for this ephemeral runner, issuing delete", "error", err.Error()) if err := r.Delete(ctx, &ephemeralRunner); err != nil { return ctrl.Result{}, fmt.Errorf("failed to delete the ephemeral runner: %w", err) } log.Info("Request to delete ephemeral runner has been issued") return ctrl.Result{}, nil default: log.Error(err, "Failed to create ephemeral runners secret", "error", err.Error()) return ctrl.Result{}, err } } if ephemeralRunner.Status.RunnerID == 0 { log.Info("Updating ephemeral runner status with runnerId and runnerName") runnerID, err := strconv.Atoi(string(secret.Data["runnerId"])) if err != nil { log.Error(err, "Runner config secret is corrupted: missing runnerId") log.Info("Deleting corrupted runner config secret") if err := r.Delete(ctx, secret); err != nil { return ctrl.Result{}, fmt.Errorf("failed to delete the corrupted runner config secret") } log.Info("Corrupted runner config secret has been deleted") return ctrl.Result{Requeue: true}, nil } runnerName := string(secret.Data["runnerName"]) original := ephemeralRunner.DeepCopy() ephemeralRunner.Status.RunnerID = runnerID ephemeralRunner.Status.RunnerName = runnerName if err := r.Status().Patch(ctx, &ephemeralRunner, client.MergeFrom(original)); err != nil { return ctrl.Result{}, fmt.Errorf("failed to update runner status for RunnerId/RunnerName: %w", err) } log.Info("Updated ephemeral runner status with runnerId and runnerName") } if len(ephemeralRunner.Status.Failures) > maxFailures { log.Info(fmt.Sprintf("EphemeralRunner has failed more than %d times. Deleting ephemeral runner so it can be re-created", maxFailures)) if err := r.Delete(ctx, &ephemeralRunner); err != nil { log.Error(fmt.Errorf("failed to delete ephemeral runner after %d failures: %w", maxFailures, err), "Failed to delete ephemeral runner") return ctrl.Result{}, err } return ctrl.Result{}, nil } now := metav1.Now() lastFailure := ephemeralRunner.Status.LastFailure() backoffDuration := failedRunnerBackoff[len(ephemeralRunner.Status.Failures)] nextReconciliation := lastFailure.Add(backoffDuration) if !lastFailure.IsZero() && now.Before(&metav1.Time{Time: nextReconciliation}) { requeueAfter := nextReconciliation.Sub(now.Time) log.Info( "Backing off the next reconciliation due to failure", "lastFailure", lastFailure, "nextReconciliation", nextReconciliation, "requeueAfter", requeueAfter, ) return ctrl.Result{ Requeue: true, RequeueAfter: requeueAfter, }, nil } pod := new(corev1.Pod) if err := r.Get(ctx, req.NamespacedName, pod); err != nil { if !kerrors.IsNotFound(err) { log.Error(err, "Failed to fetch the pod") return ctrl.Result{}, err } log.Info("Ephemeral runner pod does not exist. Creating new ephemeral runner") result, err := r.createPod(ctx, &ephemeralRunner, secret, log) switch { case err == nil: return result, nil case kerrors.IsAlreadyExists(err): log.Info("Runner pod already exists. Waiting for the pod event to be received") return ctrl.Result{Requeue: true, RequeueAfter: 5 * time.Second}, nil case kerrors.IsInvalid(err): log.Error(err, "Failed to create a pod due to unrecoverable failure") errMessage := fmt.Sprintf("Failed to create the pod: %v", err) if err := r.markAsFailed(ctx, &ephemeralRunner, errMessage, ReasonInvalidPodFailure, log); err != nil { log.Error(err, "Failed to set ephemeral runner to phase Failed") return ctrl.Result{}, err } return ctrl.Result{}, nil case kerrors.IsForbidden(err): if status, ok := err.(kerrors.APIStatus); ok || errors.As(err, &status) { isResourceQuotaExceeded := strings.Contains(status.Status().Message, "exceeded quota:") isAboutToExpire := ephemeralRunner.CreationTimestamp.Time.Add(10 * time.Minute).Before(time.Now()) switch { case isResourceQuotaExceeded && isAboutToExpire: log.Error(err, "Failed to create a pod due to resource quota exceeded and the ephemeral runner is about to expire; re-creating the ephemeral runner") if err := r.Delete(ctx, &ephemeralRunner); err != nil { log.Error(err, "Failed to delete the ephemeral runner") return ctrl.Result{}, err } return ctrl.Result{}, nil case isResourceQuotaExceeded: log.Error(err, "Resource quota is exceeded; requeue in 30s to retry pod creation") return ctrl.Result{RequeueAfter: 30 * time.Second}, nil default: // other forbidden errors // fallthrough to the default handling below } } log.Error(err, "Failed to create a pod due to unrecoverable failure") errMessage := fmt.Sprintf("Failed to create the pod: %v", err) if err := r.markAsFailed(ctx, &ephemeralRunner, errMessage, ReasonInvalidPodFailure, log); err != nil { log.Error(err, "Failed to set ephemeral runner to phase Failed") return ctrl.Result{}, err } return ctrl.Result{}, nil default: log.Error(err, "Failed to create the pod") return ctrl.Result{}, err } } cs := runnerContainerStatus(pod) switch { case pod.Status.Phase == corev1.PodFailed: // All containers are stopped log.Info( "Pod is in failed phase, inspecting runner container status", "podReason", pod.Status.Reason, "podMessage", pod.Status.Message, "podConditions", pod.Status.Conditions, ) // If the runner pod did not have chance to start, terminated state may not be set. // Therefore, we should try to restart it. if cs == nil || cs.State.Terminated == nil { log.Info("Runner container does not have state set, deleting pod as failed so it can be restarted") return ctrl.Result{}, r.deleteEphemeralRunnerOrPod(ctx, &ephemeralRunner, pod, log) } switch cs.State.Terminated.ExitCode { case 0: log.Info("Runner container has succeeded but pod is in failed phase; Assume successful exit") // If the pod is in a failed state, that means that at least one container exited with non-zero exit code. // If the runner container exits with 0, we assume that the runner has finished successfully. // If side-car container exits with non-zero, it shouldn't affect the runner. Runner exit code // drives the controller's inference of whether the job has succeeded or failed. if err := r.markAsSucceeded(ctx, &ephemeralRunner, pod, log); err != nil { log.Error(err, "Failed to set ephemeral runner to phase Succeeded") return ctrl.Result{}, err } if err := r.Delete(ctx, &ephemeralRunner); err != nil { log.Error(err, "Failed to delete ephemeral runner after successful completion") return ctrl.Result{}, err } return ctrl.Result{}, nil case 7: if err := r.markAsOutdated(ctx, &ephemeralRunner, log); err != nil { log.Error(err, "Failed to set ephemeral runner to phase Outdated") return ctrl.Result{}, err } return ctrl.Result{}, nil } log.Error( errors.New("ephemeral runner container has failed, with runner container exit code non-zero"), "Ephemeral runner container has failed, and runner container termination exit code is non-zero", "containerTerminatedState", cs.State.Terminated, ) return ctrl.Result{}, r.deleteEphemeralRunnerOrPod(ctx, &ephemeralRunner, pod, log) case initContainerFailed(pod): log.Info( "Pod has a failed init container, deleting pod as failed so it can be restarted", "initContainerStatuses", pod.Status.InitContainerStatuses, ) return ctrl.Result{}, r.deleteEphemeralRunnerOrPod(ctx, &ephemeralRunner, pod, log) case cs == nil: // starting, no container state yet log.Info("Waiting for runner container status to be available") return ctrl.Result{}, nil case cs.State.Terminated == nil: // container is not terminated and pod phase is not failed, so runner is still running log.Info("Runner container is still running; updating ephemeral runner status") if err := r.updateRunStatusFromPod(ctx, &ephemeralRunner, pod, log); err != nil { log.Info("Failed to update ephemeral runner status. Requeue to not miss this event") return ctrl.Result{}, err } return ctrl.Result{}, nil case cs.State.Terminated.ExitCode == 7: // outdated if err := r.markAsOutdated(ctx, &ephemeralRunner, log); err != nil { log.Error(err, "Failed to set ephemeral runner to phase Outdated") return ctrl.Result{}, err } return ctrl.Result{}, nil case cs.State.Terminated.ExitCode != 0: // failed log.Info("Ephemeral runner container failed", "exitCode", cs.State.Terminated.ExitCode) return ctrl.Result{}, r.deleteEphemeralRunnerOrPod(ctx, &ephemeralRunner, pod, log) default: // succeeded log.Info("Ephemeral runner has finished successfully, deleting ephemeral runner", "exitCode", cs.State.Terminated.ExitCode) if err := r.markAsSucceeded(ctx, &ephemeralRunner, pod, log); err != nil { log.Error(err, "Failed to set ephemeral runner to phase Succeeded") return ctrl.Result{}, err } if err := r.Delete(ctx, &ephemeralRunner); err != nil { log.Error(err, "Failed to delete ephemeral runner after successful completion") return ctrl.Result{}, err } return ctrl.Result{}, nil } } func (r *EphemeralRunnerReconciler) deleteEphemeralRunnerOrPod(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, pod *corev1.Pod, log logr.Logger) error { if ephemeralRunner.HasJob() { log.Error( errors.New("ephemeral runner has a job assigned, but the pod has failed"), "Ephemeral runner either has faulty entrypoint or something external killing the runner", ) log.Info("Deleting the ephemeral runner that has a job assigned but the pod has failed") if err := r.Delete(ctx, ephemeralRunner); err != nil { log.Error(err, "Failed to delete the ephemeral runner that has a job assigned but the pod has failed") return err } // The runner is gone, and its pod failed with a job assigned, so the // registration is still held. The delete above runs the finalizer, which // queues its removal. log.Info("Deleted the ephemeral runner that has a job assigned but the pod has failed") return nil } if err := r.deletePodAsFailed(ctx, ephemeralRunner, pod, log); err != nil { log.Error(err, "Failed to delete runner pod on failure") return err } return nil } func (r *EphemeralRunnerReconciler) cleanupResources(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, log logr.Logger) error { log.Info("Cleaning up the runner pod") pod := new(corev1.Pod) err := r.Get(ctx, types.NamespacedName{Namespace: ephemeralRunner.Namespace, Name: ephemeralRunner.Name}, pod) switch { case err == nil: if pod.DeletionTimestamp.IsZero() { log.Info("Deleting the runner pod") if err := r.Delete(ctx, pod); err != nil && !kerrors.IsNotFound(err) { return fmt.Errorf("failed to delete pod: %w", err) } log.Info("Deleted the runner pod") } else { log.Info("Pod contains deletion timestamp") } case kerrors.IsNotFound(err): log.Info("Runner pod is deleted") default: return err } log.Info("Cleaning up the runner jitconfig secret") secret := new(corev1.Secret) err = r.Get(ctx, types.NamespacedName{Namespace: ephemeralRunner.Namespace, Name: ephemeralRunner.Name}, secret) switch { case err == nil: if secret.DeletionTimestamp.IsZero() { log.Info("Deleting the jitconfig secret") if err := r.Delete(ctx, secret); err != nil && !kerrors.IsNotFound(err) { return fmt.Errorf("failed to delete secret: %w", err) } log.Info("Deleted jitconfig secret") } else { log.Info("Secret contains deletion timestamp") } case kerrors.IsNotFound(err): log.Info("Runner jitconfig secret is deleted") default: return err } return nil } func (r *EphemeralRunnerReconciler) cleanupContainerHooksResources(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, log logr.Logger) error { log.Info("Cleaning up runner linked pods") var errs []error if err := r.cleanupRunnerLinkedPods(ctx, ephemeralRunner, log); err != nil { errs = append(errs, err) } log.Info("Cleaning up runner linked secrets") if err := r.cleanupRunnerLinkedSecrets(ctx, ephemeralRunner, log); err != nil { errs = append(errs, err) } return errors.Join(errs...) } func (r *EphemeralRunnerReconciler) cleanupRunnerLinkedPods(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, log logr.Logger) error { runnerLinedLabels := client.MatchingLabels( map[string]string{ "runner-pod": ephemeralRunner.Name, }, ) var runnerLinkedPodList corev1.PodList if err := r.List(ctx, &runnerLinkedPodList, client.InNamespace(ephemeralRunner.Namespace), runnerLinedLabels); err != nil { return fmt.Errorf("failed to list runner-linked pods: %w", err) } if len(runnerLinkedPodList.Items) == 0 { log.Info("Runner-linked pods are deleted") return nil } log.Info("Deleting container hooks runner-linked pods", "count", len(runnerLinkedPodList.Items)) var errs []error for i := range runnerLinkedPodList.Items { linkedPod := &runnerLinkedPodList.Items[i] if !linkedPod.DeletionTimestamp.IsZero() { continue } log.Info("Deleting container hooks runner-linked pod", "name", linkedPod.Name) if err := r.Delete(ctx, linkedPod); err != nil && !kerrors.IsNotFound(err) { errs = append(errs, fmt.Errorf("failed to delete runner linked pod %q: %w", linkedPod.Name, err)) } } return errors.Join(errs...) } func (r *EphemeralRunnerReconciler) cleanupRunnerLinkedSecrets(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, log logr.Logger) error { runnerLinkedLabels := client.MatchingLabels( map[string]string{ "runner-pod": ephemeralRunner.Name, }, ) var runnerLinkedSecretList corev1.SecretList if err := r.List(ctx, &runnerLinkedSecretList, client.InNamespace(ephemeralRunner.Namespace), runnerLinkedLabels); err != nil { return fmt.Errorf("failed to list runner-linked secrets: %w", err) } if len(runnerLinkedSecretList.Items) == 0 { log.Info("Runner-linked secrets are deleted") return nil } log.Info("Deleting container hooks runner-linked secrets", "count", len(runnerLinkedSecretList.Items)) var errs []error for i := range runnerLinkedSecretList.Items { s := &runnerLinkedSecretList.Items[i] if !s.DeletionTimestamp.IsZero() { continue } log.Info("Deleting container hooks runner-linked secret", "name", s.Name) if err := r.Delete(ctx, s); err != nil && !kerrors.IsNotFound(err) { errs = append(errs, fmt.Errorf("failed to delete runner linked secret %q: %w", s.Name, err)) } } return errors.Join(errs...) } func (r *EphemeralRunnerReconciler) markAsFailed(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, errMessage string, reason string, log logr.Logger) error { log.Info("Updating ephemeral runner status to Failed") original := ephemeralRunner.DeepCopy() ephemeralRunner.Status.Phase = v1alpha1.EphemeralRunnerPhaseFailed ephemeralRunner.Status.Reason = reason ephemeralRunner.Status.Message = errMessage if err := r.Status().Patch(ctx, ephemeralRunner, client.MergeFrom(original)); err != nil { return fmt.Errorf("failed to update ephemeral runner status Phase/Message: %w", err) } r.publishEphemeralRunnerPhaseMetric(ephemeralRunner, ephemeralRunner.Status.Phase, log) // A failed runner is not deleted here; it stays until the EphemeralRunnerSet // cleans it up, which can be a long time, so the registration is released now // rather than waiting for the finalizer. if err := r.queueUnregistration(ctx, ephemeralRunner, log); err != nil { return err } log.Info("EphemeralRunner is marked as Failed and queued for removal from the service") return nil } // queueUnregistration releases the runner's registration with the Actions // service: it hands the removal to the background workers and drops the // finalizer that exists to make it happen. // // A runner that exited with code 0 deregistered itself, so it has nothing to // hand over and only the finalizer goes. // // Dropping the finalizer is also what keeps this to a single removal. Without // it the deletion that eventually follows would queue the same runner again. // It doubles as the guard that makes this safe to call repeatedly: a runner // whose registration is already released is left alone. func (r *EphemeralRunnerReconciler) queueUnregistration(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, log logr.Logger) error { if !controllerutil.ContainsFinalizer(ephemeralRunner, ephemeralRunnerActionsFinalizerName) { return nil } var runnerID int if runnerSelfDeregistered(ephemeralRunner) { log.Info("Runner exited successfully and deregistered itself, skipping its removal from the service") } else { id, err := r.registeredRunnerID(ctx, ephemeralRunner, log) if err != nil { return err } runnerID = id } original := ephemeralRunner.DeepCopy() controllerutil.RemoveFinalizer(ephemeralRunner, ephemeralRunnerActionsFinalizerName) if err := r.Patch(ctx, ephemeralRunner, client.MergeFrom(original)); err != nil && !kerrors.IsNotFound(err) { return fmt.Errorf("failed to remove the runner registration finalizer: %w", err) } // Queued only once the finalizer is gone. A NotFound patch means another // actor already removed it and the runner finished deletion, while any other // failed patch leaves the removal to the retry rather than queueing it twice. if runnerID != 0 { r.UnregistrationQueue.Push(ephemeralRunner, runnerID) } return nil } func (r *EphemeralRunnerReconciler) markAsOutdated(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, log logr.Logger) error { log.Info("Updating ephemeral runner status to Outdated") original := ephemeralRunner.DeepCopy() ephemeralRunner.Status.Phase = v1alpha1.EphemeralRunnerPhaseOutdated ephemeralRunner.Status.Reason = "Outdated" ephemeralRunner.Status.Message = "Runner is deprecated" if err := r.Status().Patch(ctx, ephemeralRunner, client.MergeFrom(original)); err != nil { return fmt.Errorf("failed to update ephemeral runner status Phase/Message: %w", err) } r.publishEphemeralRunnerPhaseMetric(ephemeralRunner, ephemeralRunner.Status.Phase, log) // Queued rather than removed here, for the same reason as markAsFailed: an // outdated runner waits on the EphemeralRunnerSet to delete it, and the // phase transition has no reason to wait on the service. if err := r.queueUnregistration(ctx, ephemeralRunner, log); err != nil { return err } log.Info("EphemeralRunner is marked as Outdated and queued for removal from the service") return nil } func (r *EphemeralRunnerReconciler) markAsSucceeded(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, pod *corev1.Pod, log logr.Logger) error { log.Info("Updating ephemeral runner status to Succeeded") original := ephemeralRunner.DeepCopy() ephemeralRunner.Status.Phase = v1alpha1.EphemeralRunnerPhaseSucceeded ephemeralRunner.Status.Ready = false ephemeralRunner.Status.Reason = pod.Status.Reason ephemeralRunner.Status.Message = pod.Status.Message if err := r.Status().Patch(ctx, ephemeralRunner, client.MergeFrom(original)); err != nil { return fmt.Errorf("failed to update ephemeral runner status Phase/Message: %w", err) } r.publishEphemeralRunnerPhaseMetric(ephemeralRunner, ephemeralRunner.Status.Phase, log) log.Info("EphemeralRunner is marked as Succeeded") return nil } // deletePodAsFailed is responsible for deleting the pod and updating the .Status.Failures for tracking failure count. // It should not be responsible for setting the status to Failed. // // It should be called by deleteEphemeralRunnerOrPod which is responsible for deciding whether to delete the EphemeralRunner or just the Pod. func (r *EphemeralRunnerReconciler) deletePodAsFailed(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, pod *corev1.Pod, log logr.Logger) error { if pod.DeletionTimestamp.IsZero() { log.Info("Deleting the ephemeral runner pod", "podId", pod.UID) if err := r.Delete(ctx, pod); err != nil && !kerrors.IsNotFound(err) { return fmt.Errorf("failed to delete pod with status failed: %w", err) } } log.Info("Updating ephemeral runner status to track the failure count") original := ephemeralRunner.DeepCopy() if ephemeralRunner.Status.Failures == nil { ephemeralRunner.Status.Failures = make(map[string]metav1.Time) } ephemeralRunner.Status.Failures[string(pod.UID)] = metav1.Now() ephemeralRunner.Status.Ready = false ephemeralRunner.Status.Reason = pod.Status.Reason ephemeralRunner.Status.Message = pod.Status.Message if err := r.Status().Patch(ctx, ephemeralRunner, client.MergeFrom(original)); err != nil { return fmt.Errorf("failed to update ephemeral runner status with failure count: %w", err) } log.Info("EphemeralRunner pod is deleted and status is updated with failure count") return nil } func (r *EphemeralRunnerReconciler) createRunnerJitConfig(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, log logr.Logger) (*scaleset.RunnerScaleSetJitRunnerConfig, error) { // Runner is not registered with the service. We need to register it first log.Info("Creating ephemeral runner JIT config") actionsClient, err := r.GetActionsService(ctx, ephemeralRunner) if err != nil { return nil, fmt.Errorf("failed to get actions client for generating JIT config: %w", err) } jitSettings := &scaleset.RunnerScaleSetJitRunnerSetting{ Name: ephemeralRunner.Name, } for i := range ephemeralRunner.Spec.Spec.Containers { if ephemeralRunner.Spec.Spec.Containers[i].Name == v1alpha1.EphemeralRunnerContainerName && ephemeralRunner.Spec.Spec.Containers[i].WorkingDir != "" { jitSettings.WorkFolder = ephemeralRunner.Spec.Spec.Containers[i].WorkingDir } } jitConfig, err := actionsClient.GenerateJitRunnerConfig(ctx, jitSettings, ephemeralRunner.Spec.RunnerScaleSetID) if err == nil { // if NO error log.Info("Created ephemeral runner JIT config", "runnerId", jitConfig.Runner.ID) return jitConfig, nil } if !errors.Is(err, scaleset.RunnerExistsError) { return nil, fmt.Errorf("failed to generate JIT config with generic error: %w", err) } // If the runner with the name we want already exists it means: // - We might have a name collision. // - Our previous reconciliation loop failed to update the // status with the runnerId and runnerJITConfig after the `GenerateJitRunnerConfig` // created the runner registration on the service. // We will try to get the runner and see if it's belong to this AutoScalingRunnerSet, // if so, we can simply delete the runner registration and create a new one. log.Info("Getting runner jit config failed with conflict error, trying to get the runner by name", "runnerName", ephemeralRunner.Name) existingRunner, err := actionsClient.GetRunnerByName(ctx, ephemeralRunner.Name) if err != nil { return nil, fmt.Errorf("failed to get runner by name: %w", err) } if existingRunner == nil { log.Info("Runner with the same name does not exist anymore, re-queuing the reconciliation") return nil, fmt.Errorf("%w: runner existed, retry configuration", retryableError) } log.Info("Found the runner with the same name", "runnerId", existingRunner.ID, "runnerScaleSetId", existingRunner.RunnerScaleSetID) if existingRunner.RunnerScaleSetID == ephemeralRunner.Spec.RunnerScaleSetID { log.Info("Removing the runner with the same name") err := actionsClient.RemoveRunner(ctx, int64(existingRunner.ID)) if err != nil { return nil, fmt.Errorf("failed to remove runner from the service: %w", err) } log.Info("Removed the runner with the same name, re-queuing the reconciliation") return nil, fmt.Errorf("%w: runner existed belonging to the scale set, retry configuration", retryableError) } return nil, fmt.Errorf("%w: runner with the same name but doesn't belong to this RunnerScaleSet: %w", fatalError, err) } func (r *EphemeralRunnerReconciler) createPod(ctx context.Context, runner *v1alpha1.EphemeralRunner, secret *corev1.Secret, log logr.Logger) (ctrl.Result, error) { var envs []corev1.EnvVar if runner.Spec.ProxySecretRef != "" { http := corev1.EnvVar{ Name: "http_proxy", ValueFrom: &corev1.EnvVarSource{ SecretKeyRef: &corev1.SecretKeySelector{ LocalObjectReference: corev1.LocalObjectReference{ Name: runner.Spec.ProxySecretRef, }, Key: "http_proxy", }, }, } if runner.Spec.Proxy.HTTP != nil { envs = append(envs, http) } https := corev1.EnvVar{ Name: "https_proxy", ValueFrom: &corev1.EnvVarSource{ SecretKeyRef: &corev1.SecretKeySelector{ LocalObjectReference: corev1.LocalObjectReference{ Name: runner.Spec.ProxySecretRef, }, Key: "https_proxy", }, }, } if runner.Spec.Proxy.HTTPS != nil { envs = append(envs, https) } noProxy := corev1.EnvVar{ Name: "no_proxy", ValueFrom: &corev1.EnvVarSource{ SecretKeyRef: &corev1.SecretKeySelector{ LocalObjectReference: corev1.LocalObjectReference{ Name: runner.Spec.ProxySecretRef, }, Key: "no_proxy", }, }, } if len(runner.Spec.Proxy.NoProxy) > 0 { envs = append(envs, noProxy) } } log.Info("Creating new pod for ephemeral runner") newPod, err := r.newEphemeralRunnerPod(runner, secret, envs...) if err != nil { log.Error(err, "Failed to build new pod") return ctrl.Result{}, err } log.Info("Created new pod spec for ephemeral runner") if err := r.Create(ctx, newPod); err != nil { log.Error(err, "Failed to create pod resource for ephemeral runner.") return ctrl.Result{}, err } log.Info("Created ephemeral runner pod", "runnerScaleSetId", runner.Spec.RunnerScaleSetID, "runnerName", runner.Status.RunnerName, "runnerId", runner.Status.RunnerID, "configUrl", runner.Spec.GitHubConfigURL, "podName", newPod.Name) return ctrl.Result{}, nil } func (r *EphemeralRunnerReconciler) createSecret(ctx context.Context, runner *v1alpha1.EphemeralRunner, jitConfig *scaleset.RunnerScaleSetJitRunnerConfig, log logr.Logger) (*corev1.Secret, error) { log.Info("Creating new secret for ephemeral runner") jitSecret, err := r.newEphemeralRunnerJitSecret(runner, jitConfig) if err != nil { return nil, fmt.Errorf("failed to build jit secret: %w", err) } log.Info("Created new secret spec for ephemeral runner") if err := r.Create(ctx, jitSecret); err != nil { return nil, fmt.Errorf("failed to create jit secret: %w", err) } log.Info("Created ephemeral runner secret", "secretName", jitSecret.Name) return jitSecret, nil } // updateRunStatusFromPod is responsible for updating non-exiting statuses. // It should never update phase to Failed or Succeeded // It should never update phase to Running (the listener owns that transition) // // The event should not be re-queued since the termination status should be set // before proceeding with reconciliation logic func (r *EphemeralRunnerReconciler) updateRunStatusFromPod(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, pod *corev1.Pod, log logr.Logger) error { if pod.Status.Phase == corev1.PodSucceeded || pod.Status.Phase == corev1.PodFailed { return nil } ready := podReady(pod) // Publish Pending as soon as the runner is observed non-terminal, regardless of // the pod phase. The controller only reaches this point once the runner // container status exists, and by then the pod has usually already advanced to // Running, so keying the initial phase off PodPending would leave a runner // phase-empty for its whole life -- omitted from the phase metrics, and in // breach of the documented contract that Pending means "created, no job yet". // Guarding on the empty phase alone is sufficient: every terminal phase, and // Running itself, is non-empty, so this can never overwrite one. phase := ephemeralRunner.Status.Phase if phase == "" { phase = v1alpha1.EphemeralRunnerPhasePending } // The controller no longer promotes the runner to Running. The listener owns that // transition and applies it when a job is assigned to this runner. The controller // still publishes the initial Pending phase while the runner pod is starting. // The patch below is optimistically locked so a stale cached copy of this runner // cannot undo the listener's transition to Running. phaseChanged := phase != ephemeralRunner.Status.Phase readyChanged := ready != ephemeralRunner.Status.Ready if !phaseChanged && !readyChanged { return nil } log.Info( "Updating ephemeral runner status", "statusPhase", pod.Status.Phase, "statusReason", pod.Status.Reason, "statusMessage", pod.Status.Message, "ready", ready, ) original := ephemeralRunner.DeepCopy() ephemeralRunner.Status.Phase = phase ephemeralRunner.Status.Ready = ready ephemeralRunner.Status.Reason = pod.Status.Reason ephemeralRunner.Status.Message = pod.Status.Message if err := r.Status().Patch(ctx, ephemeralRunner, client.MergeFromWithOptions(original, client.MergeFromWithOptimisticLock{})); err != nil { return fmt.Errorf("failed to update runner status for Phase/Reason/Message/Ready: %w", err) } r.publishEphemeralRunnerPhaseMetric(ephemeralRunner, ephemeralRunner.Status.Phase, log) log.Info("Updated ephemeral runner status") return nil } func (r *EphemeralRunnerReconciler) publishEphemeralRunnerPhaseMetric(ephemeralRunner *v1alpha1.EphemeralRunner, phase v1alpha1.EphemeralRunnerPhase, log logr.Logger) { if !r.PublishMetrics { return } commonLabels, err := ephemeralRunnerMetricLabels(ephemeralRunner) if err != nil { log.Error(err, "Failed to build ephemeral runner metric labels") return } key := types.NamespacedName{Namespace: ephemeralRunner.Namespace, Name: ephemeralRunner.Name} ephemeralRunnerPhaseMetrics.Lock() defer ephemeralRunnerPhaseMetrics.Unlock() previousPhase, ok := ephemeralRunnerPhaseMetrics.phases[key] if ok && previousPhase == phase { return } if ok { metrics.SubEphemeralRunner(commonLabels, previousPhase) } if phase == "" { delete(ephemeralRunnerPhaseMetrics.phases, key) return } metrics.AddEphemeralRunner(commonLabels, phase) ephemeralRunnerPhaseMetrics.phases[key] = phase } func ephemeralRunnerMetricLabels(ephemeralRunner *v1alpha1.EphemeralRunner) (metrics.CommonLabels, error) { parsedURL, err := actions.ParseGitHubConfigFromURL(ephemeralRunner.Spec.GitHubConfigURL) if err != nil { return metrics.CommonLabels{}, fmt.Errorf("github config URL is invalid: %w", err) } return metrics.CommonLabels{ Name: ephemeralRunner.Labels[LabelKeyGitHubScaleSetName], Namespace: ephemeralRunner.Labels[LabelKeyGitHubScaleSetNamespace], Repository: parsedURL.Repository, Organization: parsedURL.Organization, Enterprise: parsedURL.Enterprise, }, nil } // registeredRunnerID returns the ID of the registration the runner holds with // the Actions service, or 0 when it never got one. // // Callers decide whether a removal is needed at all; this only names the // registration to remove. See runnerSelfDeregistered for the runners that do // not need one. // // The common answer comes from the runner's own status and costs nothing. The // exception is a runner whose status never recorded an ID: the registration is // created by GenerateJitRunnerConfig, and the ID it returns reaches the // jitconfig secret before the status patch that publishes it. A runner deleted // in that window holds a registration the status cannot name, so the secret is // read to recover it. That read only happens for a runner that got that far and // no further, never on the path a finishing job takes. // // A secret that cannot be read is an error rather than an answer. If it is // absent, the service is checked by name before concluding the runner was // never registered: GenerateJitRunnerConfig can register it just before // createSecret persists the ID. Any other uncertainty leaves the question // open, because answering 0 would drop the finalizer and lose the last record // of a registration that does exist. func (r *EphemeralRunnerReconciler) registeredRunnerID(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, log logr.Logger) (int, error) { if ephemeralRunner.Status.RunnerID != 0 { return ephemeralRunner.Status.RunnerID, nil } secret := new(corev1.Secret) if err := r.Get(ctx, types.NamespacedName{Namespace: ephemeralRunner.Namespace, Name: ephemeralRunner.Name}, secret); err != nil { if !kerrors.IsNotFound(err) { return 0, fmt.Errorf("failed to read the jitconfig secret of a runner without a recorded ID: %w", err) } actionsClient, err := r.GetActionsService(ctx, ephemeralRunner) if err != nil { return 0, fmt.Errorf("failed to get actions client for a runner without a recorded ID or jitconfig secret: %w", err) } existingRunner, err := actionsClient.GetRunnerByName(ctx, ephemeralRunner.Name) if err != nil { return 0, fmt.Errorf("failed to get runner by name for a runner without a recorded ID or jitconfig secret: %w", err) } if existingRunner == nil { log.Info("No runner registration found for a runner without a recorded ID or jitconfig secret") return 0, nil } if existingRunner.RunnerScaleSetID != ephemeralRunner.Spec.RunnerScaleSetID { return 0, fmt.Errorf( "runner registration %d found by name belongs to runner scale set %d, expected %d", existingRunner.ID, existingRunner.RunnerScaleSetID, ephemeralRunner.Spec.RunnerScaleSetID, ) } log.Info("Recovered the runner ID from the Actions service", "runnerId", existingRunner.ID) return existingRunner.ID, nil } runnerID, err := strconv.Atoi(string(secret.Data["runnerId"])) if err != nil { // Not retried, unlike a failed read. Nothing about waiting makes the // value parse, and there is no other record of the registration. log.Error(err, "Jitconfig secret of a runner without a recorded ID is corrupted; leaving the runner for the service to clean up") return 0, nil } log.Info("Recovered the runner ID from the jitconfig secret", "runnerId", runnerID) return runnerID, nil } // SetupWithManager sets up the controller with the Manager. func (r *EphemeralRunnerReconciler) SetupWithManager(mgr ctrl.Manager, opts ...Option) error { r.ResourceBuilder.setSchemeIfUnset(r.Scheme) if r.UnregistrationQueue == nil { r.UnregistrationQueue = NewRunnerUnregistrationQueue( r.Log.WithName("runner-unregistration"), r.SecretResolver, 0, ) if err := mgr.Add(r.UnregistrationQueue); err != nil { return fmt.Errorf("failed to add the runner unregistration workers to the manager: %w", err) } } return builderWithOptions( ctrl.NewControllerManagedBy(mgr). For(&v1alpha1.EphemeralRunner{}). Owns(&corev1.Pod{}, builder.WithPredicates(ephemeralRunnerOwnedPodPredicate())). WithEventFilter(predicate.ResourceVersionChangedPredicate{}), opts, ).Complete(r) } func runnerContainerStatus(pod *corev1.Pod) *corev1.ContainerStatus { for i := range pod.Status.ContainerStatuses { cs := &pod.Status.ContainerStatuses[i] if cs.Name == v1alpha1.EphemeralRunnerContainerName { return cs } } return nil } func initContainerFailed(pod *corev1.Pod) bool { for i := range pod.Status.InitContainerStatuses { cs := &pod.Status.InitContainerStatuses[i] if cs.State.Terminated != nil && cs.State.Terminated.ExitCode != 0 { return true } } return false }