Files
actions-runner-controller/controllers/actions.github.com/ephemeralrunner_controller.go
T

1122 lines
42 KiB
Go

/*
Copyright 2020 The actions-runner-controller authors.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
*/
package actionsgithubcom
import (
"context"
"errors"
"fmt"
"strconv"
"strings"
"time"
"github.com/actions/actions-runner-controller/apis/actions.github.com/v1alpha1"
"github.com/actions/actions-runner-controller/controllers/actions.github.com/metrics"
"github.com/actions/scaleset"
"github.com/go-logr/logr"
corev1 "k8s.io/api/core/v1"
kerrors "k8s.io/apimachinery/pkg/api/errors"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/runtime"
"k8s.io/apimachinery/pkg/types"
"k8s.io/client-go/util/workqueue"
ctrl "sigs.k8s.io/controller-runtime"
"sigs.k8s.io/controller-runtime/pkg/client"
"sigs.k8s.io/controller-runtime/pkg/controller/controllerutil"
"sigs.k8s.io/controller-runtime/pkg/controller/priorityqueue"
"sigs.k8s.io/controller-runtime/pkg/event"
"sigs.k8s.io/controller-runtime/pkg/handler"
"sigs.k8s.io/controller-runtime/pkg/predicate"
"sigs.k8s.io/controller-runtime/pkg/reconcile"
)
const (
ephemeralRunnerFinalizerName = "ephemeralrunner.actions.github.com/finalizer"
terminalPodUpdatePriority = 100
)
// EphemeralRunnerReconciler reconciles a EphemeralRunner object
type EphemeralRunnerReconciler struct {
client.Client
Log logr.Logger
Scheme *runtime.Scheme
PublishMetrics bool
*ResourceBuilder
}
// precompute backoff durations for failed ephemeral runners
// the len(failedRunnerBackoff) must be equal to maxFailures + 1
var failedRunnerBackoff = []time.Duration{
0,
5 * time.Second,
10 * time.Second,
20 * time.Second,
40 * time.Second,
80 * time.Second,
}
const maxFailures = 5
// +kubebuilder:rbac:groups=actions.github.com,resources=ephemeralrunners,verbs=get;list;watch;create;update;patch;delete
// +kubebuilder:rbac:groups=actions.github.com,resources=ephemeralrunners/status,verbs=get;update;patch
// +kubebuilder:rbac:groups=actions.github.com,resources=ephemeralrunners/finalizers,verbs=get;list;watch;create;update;patch;delete
// +kubebuilder:rbac:groups=core,resources=pods,verbs=get;list;watch;create;update;patch;delete
// +kubebuilder:rbac:groups=core,resources=pods/status,verbs=get
// +kubebuilder:rbac:groups=core,resources=secrets,verbs=create;get;list;watch;delete;deletecollection
// Reconcile is part of the main kubernetes reconciliation loop which aims to
// move the current state of the cluster closer to the desired state.
//
// For more details, check Reconcile and its Result here:
// - https://pkg.go.dev/sigs.k8s.io/controller-runtime@v0.6.4/pkg/reconcile
func (r *EphemeralRunnerReconciler) Reconcile(ctx context.Context, req ctrl.Request) (ctrl.Result, error) {
log := r.Log.WithValues("ephemeralrunner", req.NamespacedName)
var ephemeralRunner v1alpha1.EphemeralRunner
if err := r.Get(ctx, req.NamespacedName, &ephemeralRunner); err != nil {
return ctrl.Result{}, client.IgnoreNotFound(err)
}
if !ephemeralRunner.DeletionTimestamp.IsZero() {
r.emitLifecycleMetrics(ctx, &ephemeralRunner, log)
if !controllerutil.ContainsFinalizer(&ephemeralRunner, ephemeralRunnerFinalizerName) {
return ctrl.Result{}, nil
}
if ephemeralRunner.Status.Phase != v1alpha1.EphemeralRunnerPhaseSucceeded && ephemeralRunner.Status.RunnerID != 0 {
log.Info("Trying to remove runner from the service before finalizing")
if err := r.deleteRunnerFromService(ctx, &ephemeralRunner, log); err != nil {
log.Error(err, "Failed to remove runner from the service before finalizing")
}
}
log.Info("Finalizing ephemeral runner")
err := r.cleanupResources(ctx, &ephemeralRunner, log)
if err != nil {
log.Error(err, "Failed to clean up ephemeral runner owned resources")
return ctrl.Result{}, err
}
if ephemeralRunner.HasContainerHookConfigured() {
log.Info("Runner has container hook configured, cleaning up container hook resources")
err = r.cleanupContainerHooksResources(ctx, &ephemeralRunner, log)
if err != nil {
log.Error(err, "Failed to clean up container hooks resources")
return ctrl.Result{}, err
}
}
log.Info("Removing finalizer")
original := ephemeralRunner.DeepCopy()
controllerutil.RemoveFinalizer(&ephemeralRunner, ephemeralRunnerFinalizerName)
log.Info("Removed finalizer from ephemeral runner")
if err := r.Patch(ctx, &ephemeralRunner, client.MergeFrom(original)); client.IgnoreNotFound(err) != nil {
log.Error(err, "Failed to update ephemeral runner after removing finalizer")
return ctrl.Result{}, err
}
log.Info("Successfully removed finalizer after cleanup")
return ctrl.Result{}, nil
}
if ephemeralRunner.IsDone() {
log.Info("Cleaning up resources after after ephemeral runner termination", "phase", ephemeralRunner.Status.Phase)
err := r.cleanupResources(ctx, &ephemeralRunner, log)
if err != nil {
log.Error(err, "Failed to clean up ephemeral runner owned resources")
return ctrl.Result{}, err
}
if ephemeralRunner.HasContainerHookConfigured() {
log.Info("Runner has container hook configured, cleaning up container hook resources")
err = r.cleanupContainerHooksResources(ctx, &ephemeralRunner, log)
if err != nil {
log.Error(err, "Failed to clean up container hooks resources")
return ctrl.Result{}, err
}
}
log.Info("EphemeralRunner has already finished. Requesting deletion after cleanup", "phase", ephemeralRunner.Status.Phase)
if err := r.deleteCompletedEphemeralRunner(ctx, &ephemeralRunner, log); err != nil {
log.Error(err, "Failed to delete completed ephemeral runner")
return ctrl.Result{}, err
}
return ctrl.Result{}, nil
}
addFinalizers := !controllerutil.ContainsFinalizer(&ephemeralRunner, ephemeralRunnerFinalizerName)
if addFinalizers {
log.Info("Adding finalizers")
original := ephemeralRunner.DeepCopy()
var addedFinalizers bool
addedFinalizers = addedFinalizers || controllerutil.AddFinalizer(&ephemeralRunner, ephemeralRunnerFinalizerName)
if addedFinalizers {
if err := r.Patch(ctx, &ephemeralRunner, client.MergeFrom(original)); err != nil {
log.Error(err, "Failed to update with finalizer set")
return ctrl.Result{}, err
}
}
log.Info("Successfully added finalizers")
}
if ephemeralRunner.Status.RunnerID != 0 {
var pod corev1.Pod
err := r.Get(ctx, req.NamespacedName, &pod)
switch {
case err == nil:
return r.reconcilePod(ctx, &ephemeralRunner, &pod, log)
case !kerrors.IsNotFound(err):
log.Error(err, "Failed to fetch the pod")
return ctrl.Result{}, err
}
}
var secret corev1.Secret
if err := r.Get(ctx, req.NamespacedName, &secret); err != nil {
if !kerrors.IsNotFound(err) {
log.Error(err, "Failed to fetch secret")
return ctrl.Result{}, err
}
jitConfig, err := r.createRunnerJitConfig(ctx, &ephemeralRunner, log)
switch {
case err == nil:
// create secret if not created
log.Info("Creating new ephemeral runner secret for jitconfig.")
jitSecret, err := r.createSecret(ctx, &ephemeralRunner, jitConfig, log)
if err != nil {
return ctrl.Result{}, fmt.Errorf("failed to create secret: %w", err)
}
log.Info("Created new ephemeral runner secret for jitconfig.")
secret = *jitSecret
case errors.Is(err, retryableError):
log.Info("Encountered retryable error, requeueing", "error", err.Error())
return ctrl.Result{Requeue: true}, nil
case errors.Is(err, fatalError):
log.Info("JIT config cannot be created for this ephemeral runner, issuing delete", "error", err.Error())
if err := r.Delete(ctx, &ephemeralRunner); err != nil {
return ctrl.Result{}, fmt.Errorf("failed to delete the ephemeral runner: %w", err)
}
log.Info("Request to delete ephemeral runner has been issued")
return ctrl.Result{}, nil
default:
log.Error(err, "Failed to create ephemeral runners secret", "error", err.Error())
return ctrl.Result{}, err
}
}
if ephemeralRunner.Status.RunnerID == 0 {
log.Info("Updating ephemeral runner status with runnerId and runnerName")
runnerID, err := strconv.Atoi(string(secret.Data["runnerId"]))
if err != nil {
log.Error(err, "Runner config secret is corrupted: missing runnerId")
log.Info("Deleting corrupted runner config secret")
if err := r.Delete(ctx, &secret); err != nil {
return ctrl.Result{}, fmt.Errorf("failed to delete the corrupted runner config secret")
}
log.Info("Corrupted runner config secret has been deleted")
return ctrl.Result{Requeue: true}, nil
}
runnerName := string(secret.Data["runnerName"])
original := ephemeralRunner.DeepCopy()
ephemeralRunner.Status.RunnerID = runnerID
ephemeralRunner.Status.RunnerName = runnerName
if err := r.Status().Patch(ctx, &ephemeralRunner, client.MergeFrom(original)); err != nil {
return ctrl.Result{}, fmt.Errorf("failed to update runner status for RunnerId/RunnerName: %w", err)
}
log.Info("Updated ephemeral runner status with runnerId and runnerName")
}
if len(ephemeralRunner.Status.Failures) > maxFailures {
log.Info(fmt.Sprintf("EphemeralRunner has failed more than %d times. Deleting ephemeral runner so it can be re-created", maxFailures))
if err := r.Delete(ctx, &ephemeralRunner); err != nil {
log.Error(fmt.Errorf("failed to delete ephemeral runner after %d failures: %w", maxFailures, err), "Failed to delete ephemeral runner")
return ctrl.Result{}, err
}
return ctrl.Result{}, nil
}
lastFailure := ephemeralRunner.Status.LastFailure()
if !lastFailure.IsZero() {
now := metav1.Now()
backoffDuration := failedRunnerBackoff[len(ephemeralRunner.Status.Failures)]
nextReconciliation := lastFailure.Add(backoffDuration)
if now.Before(&metav1.Time{Time: nextReconciliation}) {
requeueAfter := nextReconciliation.Sub(now.Time)
log.Info(
"Backing off the next reconciliation due to failure",
"lastFailure", lastFailure,
"nextReconciliation", nextReconciliation,
"requeueAfter", requeueAfter,
)
return ctrl.Result{
Requeue: true,
RequeueAfter: requeueAfter,
}, nil
}
}
var pod corev1.Pod
if err := r.Get(ctx, req.NamespacedName, &pod); err != nil {
if !kerrors.IsNotFound(err) {
log.Error(err, "Failed to fetch the pod")
return ctrl.Result{}, err
}
log.Info("Ephemeral runner pod does not exist. Creating new ephemeral runner")
result, err := r.createPod(ctx, &ephemeralRunner, &secret, log)
switch {
case err == nil:
return result, nil
case kerrors.IsAlreadyExists(err):
log.Info("Runner pod already exists. Waiting for the pod event to be received")
return ctrl.Result{Requeue: true, RequeueAfter: 5 * time.Second}, nil
case kerrors.IsInvalid(err):
log.Error(err, "Failed to create a pod due to unrecoverable failure")
errMessage := fmt.Sprintf("Failed to create the pod: %v", err)
if err := r.markAsFailed(ctx, &ephemeralRunner, errMessage, ReasonInvalidPodFailure, log); err != nil {
log.Error(err, "Failed to set ephemeral runner to phase Failed")
return ctrl.Result{}, err
}
return ctrl.Result{}, nil
case kerrors.IsForbidden(err):
if status, ok := err.(kerrors.APIStatus); ok || errors.As(err, &status) {
isResourceQuotaExceeded := strings.Contains(status.Status().Message, "exceeded quota:")
isAboutToExpire := ephemeralRunner.CreationTimestamp.Time.Add(10 * time.Minute).Before(time.Now())
switch {
case isResourceQuotaExceeded && isAboutToExpire:
log.Error(err, "Failed to create a pod due to resource quota exceeded and the ephemeral runner is about to expire; re-creating the ephemeral runner")
if err := r.Delete(ctx, &ephemeralRunner); err != nil {
log.Error(err, "Failed to delete the ephemeral runner")
return ctrl.Result{}, err
}
return ctrl.Result{}, nil
case isResourceQuotaExceeded:
log.Error(err, "Resource quota is exceeded; requeue in 30s to retry pod creation")
return ctrl.Result{RequeueAfter: 30 * time.Second}, nil
default:
// other forbidden errors
// fallthrough to the default handling below
}
}
log.Error(err, "Failed to create a pod due to unrecoverable failure")
errMessage := fmt.Sprintf("Failed to create the pod: %v", err)
if err := r.markAsFailed(ctx, &ephemeralRunner, errMessage, ReasonInvalidPodFailure, log); err != nil {
log.Error(err, "Failed to set ephemeral runner to phase Failed")
return ctrl.Result{}, err
}
return ctrl.Result{}, nil
default:
log.Error(err, "Failed to create the pod")
return ctrl.Result{}, err
}
}
return r.reconcilePod(ctx, &ephemeralRunner, &pod, log)
}
func (r *EphemeralRunnerReconciler) reconcilePod(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, pod *corev1.Pod, log logr.Logger) (ctrl.Result, error) {
cs := runnerContainerStatus(pod)
switch {
case pod.Status.Phase == corev1.PodFailed: // All containers are stopped
log.Info(
"Pod is in failed phase, inspecting runner container status",
"podReason", pod.Status.Reason,
"podMessage", pod.Status.Message,
"podConditions", pod.Status.Conditions,
)
// If the runner pod did not have chance to start, terminated state may not be set.
// Therefore, we should try to restart it.
if cs == nil || cs.State.Terminated == nil {
log.Info("Runner container does not have state set, deleting pod as failed so it can be restarted")
return ctrl.Result{}, r.deleteEphemeralRunnerOrPod(ctx, ephemeralRunner, pod, log)
}
switch cs.State.Terminated.ExitCode {
case 0:
log.Info("Runner container has succeeded but pod is in failed phase; Assume successful exit")
// If the pod is in a failed state, that means that at least one container exited with non-zero exit code.
// If the runner container exits with 0, we assume that the runner has finished successfully.
// If side-car container exits with non-zero, it shouldn't affect the runner. Runner exit code
// drives the controller's inference of whether the job has succeeded or failed.
if err := r.markAsSucceededAndCleanup(ctx, ephemeralRunner, pod, log); err != nil {
log.Error(err, "Failed to clean up ephemeral runner resources after successful completion")
return ctrl.Result{}, err
}
return ctrl.Result{}, nil
case 7:
if err := r.markAsOutdated(ctx, ephemeralRunner, log); err != nil {
log.Error(err, "Failed to set ephemeral runner to phase Outdated")
return ctrl.Result{}, err
}
return ctrl.Result{}, nil
}
log.Error(
errors.New("ephemeral runner container has failed, with runner container exit code non-zero"),
"Ephemeral runner container has failed, and runner container termination exit code is non-zero",
"containerTerminatedState", cs.State.Terminated,
)
return ctrl.Result{}, r.deleteEphemeralRunnerOrPod(ctx, ephemeralRunner, pod, log)
case initContainerFailed(pod):
log.Info(
"Pod has a failed init container, deleting pod as failed so it can be restarted",
"initContainerStatuses", pod.Status.InitContainerStatuses,
)
return ctrl.Result{}, r.deleteEphemeralRunnerOrPod(ctx, ephemeralRunner, pod, log)
case cs == nil:
// starting, no container state yet
log.Info("Waiting for runner container status to be available")
return ctrl.Result{}, nil
case cs.State.Terminated == nil: // container is not terminated and pod phase is not failed, so runner is still running
log.Info("Runner container is still running; updating ephemeral runner status")
if err := r.updateRunStatusFromPod(ctx, ephemeralRunner, pod, log); err != nil {
log.Info("Failed to update ephemeral runner status. Requeue to not miss this event")
return ctrl.Result{}, err
}
return ctrl.Result{}, nil
case cs.State.Terminated.ExitCode == 7: // outdated
if err := r.markAsOutdated(ctx, ephemeralRunner, log); err != nil {
log.Error(err, "Failed to set ephemeral runner to phase Outdated")
return ctrl.Result{}, err
}
return ctrl.Result{}, nil
case cs.State.Terminated.ExitCode != 0: // failed
log.Info("Ephemeral runner container failed", "exitCode", cs.State.Terminated.ExitCode)
return ctrl.Result{}, r.deleteEphemeralRunnerOrPod(ctx, ephemeralRunner, pod, log)
default: // succeeded
log.Info("Ephemeral runner has finished successfully, cleaning up runner resources", "exitCode", cs.State.Terminated.ExitCode)
if err := r.markAsSucceededAndCleanup(ctx, ephemeralRunner, pod, log); err != nil {
log.Error(err, "Failed to clean up ephemeral runner resources after successful completion")
return ctrl.Result{}, err
}
return ctrl.Result{}, nil
}
}
func (r *EphemeralRunnerReconciler) deleteEphemeralRunnerOrPod(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, pod *corev1.Pod, log logr.Logger) error {
if ephemeralRunner.HasJob() {
log.Error(
errors.New("ephemeral runner has a job assigned, but the pod has failed"),
"Ephemeral runner either has faulty entrypoint or something external killing the runner",
)
if ephemeralRunner.Status.RunnerID != 0 {
log.Info("Trying to remove the runner from the service")
if err := r.deleteRunnerFromService(ctx, ephemeralRunner, log); err != nil {
log.Error(err, "Failed to remove the runner from the service")
}
}
log.Info("Deleting the ephemeral runner that has a job assigned but the pod has failed")
if err := r.Delete(ctx, ephemeralRunner); err != nil {
log.Error(err, "Failed to delete the ephemeral runner that has a job assigned but the pod has failed")
return err
}
return nil
}
failureCount := len(ephemeralRunner.Status.Failures)
if _, ok := ephemeralRunner.Status.Failures[string(pod.UID)]; !ok {
failureCount++
}
if failureCount > maxFailures {
log.Info(fmt.Sprintf("EphemeralRunner has failed more than %d times. Deleting ephemeral runner so it can be re-created", maxFailures))
if err := r.Delete(ctx, ephemeralRunner); err != nil {
log.Error(fmt.Errorf("failed to delete ephemeral runner after %d failures: %w", maxFailures, err), "Failed to delete ephemeral runner")
return err
}
return nil
}
if err := r.deletePodAsFailed(ctx, ephemeralRunner, pod, log); err != nil {
log.Error(err, "Failed to delete runner pod on failure")
return err
}
return nil
}
func (r *EphemeralRunnerReconciler) cleanupResources(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, log logr.Logger) error {
return r.cleanupResourcesForPod(ctx, ephemeralRunner, nil, log)
}
func (r *EphemeralRunnerReconciler) cleanupResourcesForPod(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, pod *corev1.Pod, log logr.Logger) error {
log.Info("Cleaning up the runner pod")
if pod != nil {
if pod.DeletionTimestamp.IsZero() {
log.Info("Deleting the runner pod")
if err := r.deletePodForCleanup(ctx, pod, log); err != nil {
return fmt.Errorf("failed to delete pod: %w", err)
}
log.Info("Deleted the runner pod")
} else {
log.Info("Runner pod is already being deleted")
}
} else {
var pod corev1.Pod
err := r.Get(ctx, types.NamespacedName{Namespace: ephemeralRunner.Namespace, Name: ephemeralRunner.Name}, &pod)
switch {
case err == nil && pod.DeletionTimestamp.IsZero():
log.Info("Deleting the runner pod")
if err := r.deletePodForCleanup(ctx, &pod, log); err != nil {
return fmt.Errorf("failed to delete pod: %w", err)
}
log.Info("Deleted the runner pod")
case err == nil && !pod.DeletionTimestamp.IsZero():
log.Info("Runner pod is already being deleted")
case kerrors.IsNotFound(err):
log.Info("Runner pod is deleted")
default:
return fmt.Errorf("failed to get pod: %w", err)
}
}
log.Info("Cleaning up the runner jitconfig secret")
secret := corev1.Secret{
ObjectMeta: metav1.ObjectMeta{
Name: ephemeralRunner.Name,
Namespace: ephemeralRunner.Namespace,
},
}
log.Info("Deleting the jitconfig secret")
if err := r.Delete(ctx, &secret); err != nil && !kerrors.IsNotFound(err) {
return fmt.Errorf("failed to delete secret: %w", err)
}
log.Info("Deleted jitconfig secret")
return nil
}
func (r *EphemeralRunnerReconciler) cleanupContainerHooksResources(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, log logr.Logger) error {
log.Info("Cleaning up runner linked pods")
var errs []error
if err := r.cleanupRunnerLinkedPods(ctx, ephemeralRunner, log); err != nil {
errs = append(errs, err)
}
log.Info("Cleaning up runner linked secrets")
if err := r.cleanupRunnerLinkedSecrets(ctx, ephemeralRunner, log); err != nil {
errs = append(errs, err)
}
return errors.Join(errs...)
}
func (r *EphemeralRunnerReconciler) cleanupRunnerLinkedPods(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, log logr.Logger) error {
runnerLinedLabels := client.MatchingLabels(
map[string]string{
"runner-pod": ephemeralRunner.Name,
},
)
var runnerLinkedPodList corev1.PodList
if err := r.List(ctx, &runnerLinkedPodList, client.InNamespace(ephemeralRunner.Namespace), runnerLinedLabels); err != nil {
return fmt.Errorf("failed to list runner-linked pods: %w", err)
}
if len(runnerLinkedPodList.Items) == 0 {
log.Info("Runner-linked pods are deleted")
return nil
}
log.Info("Deleting container hooks runner-linked pods", "count", len(runnerLinkedPodList.Items))
var errs []error
for i := range runnerLinkedPodList.Items {
linkedPod := &runnerLinkedPodList.Items[i]
if !linkedPod.DeletionTimestamp.IsZero() {
continue
}
log.Info("Deleting container hooks runner-linked pod", "name", linkedPod.Name)
if err := r.deletePodForCleanup(ctx, linkedPod, log); err != nil {
errs = append(errs, fmt.Errorf("failed to delete runner linked pod %q: %w", linkedPod.Name, err))
}
}
return errors.Join(errs...)
}
func (r *EphemeralRunnerReconciler) cleanupRunnerLinkedSecrets(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, log logr.Logger) error {
runnerLinkedLabels := client.MatchingLabels(
map[string]string{
"runner-pod": ephemeralRunner.Name,
},
)
log.Info("Deleting container hooks runner-linked secrets")
if err := r.DeleteAllOf(ctx, &corev1.Secret{}, client.InNamespace(ephemeralRunner.Namespace), runnerLinkedLabels); err != nil {
return fmt.Errorf("failed to delete runner-linked secrets: %w", err)
}
log.Info("Runner-linked secrets are deleted")
return nil
}
func (r *EphemeralRunnerReconciler) markAsFailed(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, errMessage string, reason string, log logr.Logger) error {
log.Info("Updating ephemeral runner status to Failed")
original := ephemeralRunner.DeepCopy()
ephemeralRunner.Status.Phase = v1alpha1.EphemeralRunnerPhaseFailed
ephemeralRunner.Status.Reason = reason
ephemeralRunner.Status.Message = errMessage
if err := r.Status().Patch(ctx, ephemeralRunner, client.MergeFrom(original)); err != nil {
return fmt.Errorf("failed to update ephemeral runner status Phase/Message: %w", err)
}
r.emitLifecycleMetrics(ctx, ephemeralRunner, log)
log.Info("Removing the runner from the service")
if err := r.deleteRunnerFromService(ctx, ephemeralRunner, log); err != nil {
return fmt.Errorf("failed to remove the runner from service: %w", err)
}
log.Info("EphemeralRunner is marked as Failed and deleted from the service")
return nil
}
func (r *EphemeralRunnerReconciler) markAsOutdated(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, log logr.Logger) error {
log.Info("Updating ephemeral runner status to Outdated")
original := ephemeralRunner.DeepCopy()
ephemeralRunner.Status.Phase = v1alpha1.EphemeralRunnerPhaseOutdated
ephemeralRunner.Status.Reason = "Outdated"
ephemeralRunner.Status.Message = "Runner is deprecated"
if err := r.Status().Patch(ctx, ephemeralRunner, client.MergeFrom(original)); err != nil {
return fmt.Errorf("failed to update ephemeral runner status Phase/Message: %w", err)
}
r.emitLifecycleMetrics(ctx, ephemeralRunner, log)
log.Info("Removing the runner from the service")
if err := r.deleteRunnerFromService(ctx, ephemeralRunner, log); err != nil {
return fmt.Errorf("failed to remove the runner from service: %w", err)
}
return nil
}
func (r *EphemeralRunnerReconciler) markAsSucceededAndCleanup(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, pod *corev1.Pod, log logr.Logger) error {
if ephemeralRunner.Status.Phase != v1alpha1.EphemeralRunnerPhaseSucceeded {
log.Info("Updating ephemeral runner status to Succeeded")
original := ephemeralRunner.DeepCopy()
ephemeralRunner.Status.Phase = v1alpha1.EphemeralRunnerPhaseSucceeded
ephemeralRunner.Status.Ready = false
ephemeralRunner.Status.Reason = ""
ephemeralRunner.Status.Message = ""
if err := r.Status().Patch(ctx, ephemeralRunner, client.MergeFrom(original)); err != nil {
return fmt.Errorf("failed to update ephemeral runner status to Succeeded: %w", err)
}
}
if err := r.cleanupResourcesForPod(ctx, ephemeralRunner, pod, log); err != nil {
return fmt.Errorf("failed to clean up ephemeral runner resources after successful completion: %w", err)
}
if err := r.deleteCompletedEphemeralRunner(ctx, ephemeralRunner, log); err != nil {
return fmt.Errorf("failed to delete ephemeral runner after successful completion: %w", err)
}
return nil
}
func (r *EphemeralRunnerReconciler) deleteCompletedEphemeralRunner(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, log logr.Logger) error {
log.Info("Deleting completed ephemeral runner")
if err := r.Delete(ctx, ephemeralRunner); err != nil && !kerrors.IsNotFound(err) {
return err
}
log.Info("Deleted completed ephemeral runner")
return nil
}
// deletePodAsFailed is responsible for deleting the pod and updating the .Status.Failures for tracking failure count.
// It should not be responsible for setting the status to Failed.
//
// It should be called by deleteEphemeralRunnerOrPod which is responsible for deciding whether to delete the EphemeralRunner or just the Pod.
func (r *EphemeralRunnerReconciler) deletePodAsFailed(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, pod *corev1.Pod, log logr.Logger) error {
if pod.DeletionTimestamp.IsZero() {
log.Info("Deleting the ephemeral runner pod", "podId", pod.UID)
if err := r.deletePodForCleanup(ctx, pod, log); err != nil {
return fmt.Errorf("failed to delete pod with status failed: %w", err)
}
}
log.Info("Updating ephemeral runner status to track the failure count")
original := ephemeralRunner.DeepCopy()
if ephemeralRunner.Status.Failures == nil {
ephemeralRunner.Status.Failures = make(map[string]metav1.Time)
}
ephemeralRunner.Status.Failures[string(pod.UID)] = metav1.Now()
ephemeralRunner.Status.Ready = false
ephemeralRunner.Status.Reason = pod.Status.Reason
ephemeralRunner.Status.Message = pod.Status.Message
if err := r.Status().Patch(ctx, ephemeralRunner, client.MergeFrom(original)); err != nil {
return fmt.Errorf("failed to update ephemeral runner status with failure count: %w", err)
}
r.emitLifecycleMetrics(ctx, ephemeralRunner, log)
log.Info("EphemeralRunner pod is deleted and status is updated with failure count")
return nil
}
var (
defaultPodDeleteOptions = []client.DeleteOption{}
forcePodDeleteOptions = []client.DeleteOption{client.GracePeriodSeconds(0)}
)
func (r *EphemeralRunnerReconciler) deletePodForCleanup(ctx context.Context, pod *corev1.Pod, log logr.Logger) error {
deleteOptions := defaultPodDeleteOptions
if shouldForceDeletePod(pod) {
log.Info("Force deleting terminal pod", "pod", types.NamespacedName{Namespace: pod.Namespace, Name: pod.Name})
deleteOptions = forcePodDeleteOptions
}
if err := r.Delete(ctx, pod, deleteOptions...); err != nil && !kerrors.IsNotFound(err) {
return err
}
return nil
}
func (r *EphemeralRunnerReconciler) createRunnerJitConfig(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, log logr.Logger) (*scaleset.RunnerScaleSetJitRunnerConfig, error) {
// Runner is not registered with the service. We need to register it first
log.Info("Creating ephemeral runner JIT config")
actionsClient, err := r.GetActionsService(ctx, ephemeralRunner)
if err != nil {
return nil, fmt.Errorf("failed to get actions client for generating JIT config: %w", err)
}
jitSettings := &scaleset.RunnerScaleSetJitRunnerSetting{
Name: ephemeralRunner.Name,
}
for i := range ephemeralRunner.Spec.Spec.Containers {
if ephemeralRunner.Spec.Spec.Containers[i].Name == v1alpha1.EphemeralRunnerContainerName &&
ephemeralRunner.Spec.Spec.Containers[i].WorkingDir != "" {
jitSettings.WorkFolder = ephemeralRunner.Spec.Spec.Containers[i].WorkingDir
}
}
jitConfig, err := actionsClient.GenerateJitRunnerConfig(ctx, jitSettings, ephemeralRunner.Spec.RunnerScaleSetID)
if err == nil { // if NO error
log.Info("Created ephemeral runner JIT config", "runnerId", jitConfig.Runner.ID)
return jitConfig, nil
}
if !errors.Is(err, scaleset.RunnerExistsError) {
return nil, fmt.Errorf("failed to generate JIT config with generic error: %w", err)
}
// If the runner with the name we want already exists it means:
// - We might have a name collision.
// - Our previous reconciliation loop failed to update the
// status with the runnerId and runnerJITConfig after the `GenerateJitRunnerConfig`
// created the runner registration on the service.
// We will try to get the runner and see if it's belong to this AutoScalingRunnerSet,
// if so, we can simply delete the runner registration and create a new one.
log.Info("Getting runner jit config failed with conflict error, trying to get the runner by name", "runnerName", ephemeralRunner.Name)
existingRunner, err := actionsClient.GetRunnerByName(ctx, ephemeralRunner.Name)
if err != nil {
return nil, fmt.Errorf("failed to get runner by name: %w", err)
}
if existingRunner == nil {
log.Info("Runner with the same name does not exist anymore, re-queuing the reconciliation")
return nil, fmt.Errorf("%w: runner existed, retry configuration", retryableError)
}
log.Info("Found the runner with the same name", "runnerId", existingRunner.ID, "runnerScaleSetId", existingRunner.RunnerScaleSetID)
if existingRunner.RunnerScaleSetID == ephemeralRunner.Spec.RunnerScaleSetID {
log.Info("Removing the runner with the same name")
err := actionsClient.RemoveRunner(ctx, int64(existingRunner.ID))
if err != nil {
return nil, fmt.Errorf("failed to remove runner from the service: %w", err)
}
log.Info("Removed the runner with the same name, re-queuing the reconciliation")
return nil, fmt.Errorf("%w: runner existed belonging to the scale set, retry configuration", retryableError)
}
return nil, fmt.Errorf("%w: runner with the same name but doesn't belong to this RunnerScaleSet: %w", fatalError, err)
}
func (r *EphemeralRunnerReconciler) createPod(ctx context.Context, runner *v1alpha1.EphemeralRunner, secret *corev1.Secret, log logr.Logger) (ctrl.Result, error) {
var envs []corev1.EnvVar
if runner.Spec.ProxySecretRef != "" {
http := corev1.EnvVar{
Name: "http_proxy",
ValueFrom: &corev1.EnvVarSource{
SecretKeyRef: &corev1.SecretKeySelector{
LocalObjectReference: corev1.LocalObjectReference{
Name: runner.Spec.ProxySecretRef,
},
Key: "http_proxy",
},
},
}
if runner.Spec.Proxy.HTTP != nil {
envs = append(envs, http)
}
https := corev1.EnvVar{
Name: "https_proxy",
ValueFrom: &corev1.EnvVarSource{
SecretKeyRef: &corev1.SecretKeySelector{
LocalObjectReference: corev1.LocalObjectReference{
Name: runner.Spec.ProxySecretRef,
},
Key: "https_proxy",
},
},
}
if runner.Spec.Proxy.HTTPS != nil {
envs = append(envs, https)
}
noProxy := corev1.EnvVar{
Name: "no_proxy",
ValueFrom: &corev1.EnvVarSource{
SecretKeyRef: &corev1.SecretKeySelector{
LocalObjectReference: corev1.LocalObjectReference{
Name: runner.Spec.ProxySecretRef,
},
Key: "no_proxy",
},
},
}
if len(runner.Spec.Proxy.NoProxy) > 0 {
envs = append(envs, noProxy)
}
}
log.Info("Creating new pod for ephemeral runner")
newPod, err := r.newEphemeralRunnerPod(runner, secret, envs...)
if err != nil {
log.Error(err, "Failed to build new pod")
return ctrl.Result{}, err
}
log.Info("Created new pod spec for ephemeral runner")
if err := r.Create(ctx, newPod); err != nil {
log.Error(err, "Failed to create pod resource for ephemeral runner.")
return ctrl.Result{}, err
}
log.Info("Created ephemeral runner pod",
"runnerScaleSetId", runner.Spec.RunnerScaleSetID,
"runnerName", runner.Status.RunnerName,
"runnerId", runner.Status.RunnerID,
"configUrl", runner.Spec.GitHubConfigURL,
"podName", newPod.Name)
return ctrl.Result{}, nil
}
func (r *EphemeralRunnerReconciler) createSecret(ctx context.Context, runner *v1alpha1.EphemeralRunner, jitConfig *scaleset.RunnerScaleSetJitRunnerConfig, log logr.Logger) (*corev1.Secret, error) {
log.Info("Creating new secret for ephemeral runner")
jitSecret, err := r.newEphemeralRunnerJitSecret(runner, jitConfig)
if err != nil {
return nil, fmt.Errorf("failed to build jit secret: %w", err)
}
log.Info("Created new secret spec for ephemeral runner")
if err := r.Create(ctx, jitSecret); err != nil {
return nil, fmt.Errorf("failed to create jit secret: %w", err)
}
log.Info("Created ephemeral runner secret", "secretName", jitSecret.Name)
return jitSecret, nil
}
// updateRunStatusFromPod is responsible for updating non-exiting statuses.
// It should never update phase to Failed or Succeeded
//
// The event should not be re-queued since the termination status should be set
// before proceeding with reconciliation logic
func (r *EphemeralRunnerReconciler) updateRunStatusFromPod(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, pod *corev1.Pod, log logr.Logger) error {
if pod.Status.Phase == corev1.PodSucceeded || pod.Status.Phase == corev1.PodFailed {
return nil
}
var ready bool
var lastTransitionTime time.Time
for _, condition := range pod.Status.Conditions {
if condition.Type == corev1.PodReady && condition.LastTransitionTime.After(lastTransitionTime) {
ready = condition.Status == corev1.ConditionTrue
lastTransitionTime = condition.LastTransitionTime.Time
}
}
phase := v1alpha1.EphemeralRunnerPhase(pod.Status.Phase)
phaseChanged := ephemeralRunner.Status.Phase != phase
readyChanged := ready != ephemeralRunner.Status.Ready
if !phaseChanged && !readyChanged {
return nil
}
log.Info(
"Updating ephemeral runner status",
"statusPhase", pod.Status.Phase,
"statusReason", pod.Status.Reason,
"statusMessage", pod.Status.Message,
"ready", ready,
)
original := ephemeralRunner.DeepCopy()
ephemeralRunner.Status.Phase = phase
ephemeralRunner.Status.Ready = ready
ephemeralRunner.Status.Reason = pod.Status.Reason
ephemeralRunner.Status.Message = pod.Status.Message
if err := r.Status().Patch(ctx, ephemeralRunner, client.MergeFrom(original)); err != nil {
return fmt.Errorf("failed to update runner status for Phase/Reason/Message/Ready: %w", err)
}
r.emitLifecycleMetrics(ctx, ephemeralRunner, log)
log.Info("Updated ephemeral runner status")
return nil
}
func (r *EphemeralRunnerReconciler) deleteRunnerFromService(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, log logr.Logger) error {
client, err := r.GetActionsService(ctx, ephemeralRunner)
if err != nil {
return fmt.Errorf("failed to get actions client for runner: %w", err)
}
log.Info("Removing runner from the service", "runnerId", ephemeralRunner.Status.RunnerID)
err = client.RemoveRunner(ctx, int64(ephemeralRunner.Status.RunnerID))
if err != nil {
return fmt.Errorf("failed to remove runner from the service: %w", err)
}
log.Info("Removed runner from the service", "runnerId", ephemeralRunner.Status.RunnerID)
return nil
}
// SetupWithManager sets up the controller with the Manager.
func (r *EphemeralRunnerReconciler) SetupWithManager(mgr ctrl.Manager, opts ...Option) error {
r.ResourceBuilder.setSchemeIfUnset(r.Scheme)
return builderWithOptions(
ctrl.NewControllerManagedBy(mgr).
For(&v1alpha1.EphemeralRunner{}).
Watches(&corev1.Pod{}, newEphemeralRunnerPodEventHandler(mgr)).
WithEventFilter(predicate.ResourceVersionChangedPredicate{}),
opts,
).Complete(r)
}
func newEphemeralRunnerPodEventHandler(mgr ctrl.Manager) handler.EventHandler {
return &prioritizedPodEventHandler{
owner: handler.EnqueueRequestForOwner(mgr.GetScheme(), mgr.GetRESTMapper(), &v1alpha1.EphemeralRunner{}, handler.OnlyControllerOwner()),
}
}
type prioritizedPodEventHandler struct {
owner handler.EventHandler
}
func (h *prioritizedPodEventHandler) Create(ctx context.Context, evt event.CreateEvent, q workqueue.TypedRateLimitingInterface[reconcile.Request]) {
h.owner.Create(ctx, evt, q)
}
func (h *prioritizedPodEventHandler) Update(ctx context.Context, evt event.UpdateEvent, q workqueue.TypedRateLimitingInterface[reconcile.Request]) {
if podBecameCleanupCandidate(evt.ObjectOld, evt.ObjectNew) {
h.owner.Update(ctx, evt, prioritizedWorkQueue{TypedRateLimitingInterface: q, priority: terminalPodUpdatePriority})
return
}
h.owner.Update(ctx, evt, q)
}
func (h *prioritizedPodEventHandler) Delete(ctx context.Context, evt event.DeleteEvent, q workqueue.TypedRateLimitingInterface[reconcile.Request]) {
h.owner.Delete(ctx, evt, q)
}
func (h *prioritizedPodEventHandler) Generic(ctx context.Context, evt event.GenericEvent, q workqueue.TypedRateLimitingInterface[reconcile.Request]) {
h.owner.Generic(ctx, evt, q)
}
type prioritizedWorkQueue struct {
workqueue.TypedRateLimitingInterface[reconcile.Request]
priority int
}
func (q prioritizedWorkQueue) Add(item reconcile.Request) {
priorityQueue, ok := q.TypedRateLimitingInterface.(priorityqueue.PriorityQueue[reconcile.Request])
if !ok {
q.TypedRateLimitingInterface.Add(item)
return
}
priorityQueue.AddWithOpts(priorityqueue.AddOpts{Priority: &q.priority}, item)
}
func (q prioritizedWorkQueue) AddAfter(item reconcile.Request, after time.Duration) {
priorityQueue, ok := q.TypedRateLimitingInterface.(priorityqueue.PriorityQueue[reconcile.Request])
if !ok {
q.TypedRateLimitingInterface.AddAfter(item, after)
return
}
priorityQueue.AddWithOpts(priorityqueue.AddOpts{After: after, Priority: &q.priority}, item)
}
func (q prioritizedWorkQueue) AddRateLimited(item reconcile.Request) {
priorityQueue, ok := q.TypedRateLimitingInterface.(priorityqueue.PriorityQueue[reconcile.Request])
if !ok {
q.TypedRateLimitingInterface.AddRateLimited(item)
return
}
priorityQueue.AddWithOpts(priorityqueue.AddOpts{RateLimited: true, Priority: &q.priority}, item)
}
func podBecameCleanupCandidate(oldObj, newObj client.Object) bool {
newPod, ok := newObj.(*corev1.Pod)
if !ok || newPod == nil || !shouldForceDeletePod(newPod) {
return false
}
oldPod, ok := oldObj.(*corev1.Pod)
return !ok || oldPod == nil || !shouldForceDeletePod(oldPod)
}
func runnerContainerStatus(pod *corev1.Pod) *corev1.ContainerStatus {
for i := range pod.Status.ContainerStatuses {
cs := &pod.Status.ContainerStatuses[i]
if cs.Name == v1alpha1.EphemeralRunnerContainerName {
return cs
}
}
return nil
}
func initContainerFailed(pod *corev1.Pod) bool {
for i := range pod.Status.InitContainerStatuses {
cs := &pod.Status.InitContainerStatuses[i]
if cs.State.Terminated != nil && cs.State.Terminated.ExitCode != 0 {
return true
}
}
return false
}
// emitLifecycleMetrics recomputes and emits lifecycle metrics for all EphemeralRunners
// owned by the same EphemeralRunnerSet as the given runner. This ensures metrics reflect
// the current lifecycle state of all sibling runners under the same AutoscalingRunnerSet.
//
// Label values are derived from the runner's labels (LabelKeyGitHubScaleSetName, etc.).
// If required labels are missing, logs a warning and skips metric emission.
func (r *EphemeralRunnerReconciler) emitLifecycleMetrics(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, log logr.Logger) {
if !r.PublishMetrics {
return
}
// Extract owner EphemeralRunnerSet from controller owner reference
ownerRef := metav1.GetControllerOfNoCopy(ephemeralRunner)
if ownerRef == nil || ownerRef.Kind != "EphemeralRunnerSet" {
log.V(1).Info("EphemeralRunner has no EphemeralRunnerSet owner, skipping metric emission")
return
}
// List all sibling EphemeralRunners owned by the same EphemeralRunnerSet
var runnerList v1alpha1.EphemeralRunnerList
if err := r.List(ctx, &runnerList,
client.InNamespace(ephemeralRunner.Namespace),
client.MatchingFields{resourceOwnerKey: ownerRef.Name},
); err != nil {
log.Error(err, "Failed to list sibling EphemeralRunners for metric emission")
return
}
// Aggregate lifecycle counts using the helper from Task 3
buckets := AggregateEphemeralRunnerLifecycle(runnerList.Items)
// Extract label values from the runner (all siblings share the same labels)
name := ephemeralRunner.Labels[LabelKeyGitHubScaleSetName]
namespace := ephemeralRunner.Labels[LabelKeyGitHubScaleSetNamespace]
repository := ephemeralRunner.Labels[LabelKeyGitHubRepository]
organization := ephemeralRunner.Labels[LabelKeyGitHubOrganization]
enterprise := ephemeralRunner.Labels[LabelKeyGitHubEnterprise]
// Gracefully handle missing labels: log warning and skip if name/namespace empty
if name == "" || namespace == "" {
log.Info("Missing required labels (name/namespace) for metric emission, skipping",
"name", name,
"namespace", namespace,
)
return
}
// Emit all six lifecycle metrics
metrics.SetEphemeralRunnerCountsByLifecycle(
metrics.CommonLabels{
Name: name,
Namespace: namespace,
Repository: repository,
Organization: organization,
Enterprise: enterprise,
},
buckets.Pending,
buckets.Running,
buckets.Succeeded,
buckets.Failed,
buckets.Outdated,
buckets.Deleting,
)
log.V(1).Info("Emitted lifecycle metrics",
"pending", buckets.Pending,
"running", buckets.Running,
"succeeded", buckets.Succeeded,
"failed", buckets.Failed,
"outdated", buckets.Outdated,
"deleting", buckets.Deleting,
)
}
func shouldForceDeletePod(pod *corev1.Pod) bool {
if pod.Status.Phase == corev1.PodSucceeded || pod.Status.Phase == corev1.PodFailed {
return true
}
if cs := runnerContainerStatus(pod); cs != nil && cs.State.Terminated != nil {
return true
}
return initContainerFailed(pod)
}