Reduce ephemeral runner status contention (#4692)

This commit is contained in:
Nikola Jokic
2026-09-29 12:31:31 +02:00
committed by GitHub
parent f858710caa
commit c48f2ca5e9
15 changed files with 244 additions and 258 deletions
@@ -71,6 +71,14 @@ func TestReconcileValidatesJITIdentityBeforePublication(t *testing.T) {
require.NoError(t, err)
require.NotNil(t, f.pod())
require.Zero(t, f.runner().Status.RunnerID)
pod := f.pod()
pod.Status.Phase = corev1.PodRunning
pod.Status.ContainerStatuses = []corev1.ContainerStatus{{
Name: v1alpha1.EphemeralRunnerContainerName,
State: corev1.ContainerState{Running: &corev1.ContainerStateRunning{}},
}}
require.NoError(t, f.c.Status().Update(t.Context(), pod))
}
_, err = f.reconcileRunner()
require.NoError(t, err)
@@ -444,21 +444,6 @@ func (r *EphemeralRunnerReconciler) Reconcile(ctx context.Context, req ctrl.Requ
}
}
// Validation above keeps malformed JIT secrets from reaching a Pod. The Pod
// only needs the valid secret, so publish the registration identity after
// the Pod exists. A retry can recover both fields from that secret.
if ephemeralRunner.Status.RunnerID == 0 {
log.Info("Updating ephemeral runner status with runnerId and runnerName")
original := ephemeralRunner.DeepCopy()
ephemeralRunner.Status.RunnerID = initialRunnerID
ephemeralRunner.Status.RunnerName = initialRunnerName
if err := r.Status().Patch(ctx, &ephemeralRunner, client.MergeFrom(original)); err != nil {
return ctrl.Result{}, fmt.Errorf("failed to update runner status for RunnerId/RunnerName: %w", err)
}
log.Info("Updated ephemeral runner status with runnerId and runnerName")
}
cs := runnerContainerStatus(pod)
switch {
case pod.Status.Phase == corev1.PodFailed: // All containers are stopped
@@ -520,7 +505,7 @@ func (r *EphemeralRunnerReconciler) Reconcile(ctx context.Context, req ctrl.Requ
case cs.State.Terminated == nil: // container is not terminated and pod phase is not failed, so runner is still running
log.Info("Runner container is still running; updating ephemeral runner status")
if err := r.updateRunStatusFromPod(ctx, &ephemeralRunner, pod, log); err != nil {
if err := r.updateRunStatusFromPod(ctx, &ephemeralRunner, pod, initialRunnerID, initialRunnerName, log); err != nil {
log.Info("Failed to update ephemeral runner status. Requeue to not miss this event")
return ctrl.Result{}, err
}
@@ -993,13 +978,13 @@ func (r *EphemeralRunnerReconciler) createSecret(ctx context.Context, runner *v1
return jitSecret, nil
}
// updateRunStatusFromPod is responsible for updating non-exiting statuses.
// It should never update phase to Failed or Succeeded
// It should never update phase to Running (the listener owns that transition)
// updateRunStatusFromPod is responsible for updating non-terminal statuses.
// It should never update phase to Failed or Succeeded.
//
// The event should not be re-queued since the termination status should be set
// before proceeding with reconciliation logic
func (r *EphemeralRunnerReconciler) updateRunStatusFromPod(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, pod *corev1.Pod, log logr.Logger) error {
// The JIT config secret is the durable registration record until the Pod first
// reports a non-terminal status. Publishing identity with that status update
// avoids a separate status-only reconciliation after Pod creation.
func (r *EphemeralRunnerReconciler) updateRunStatusFromPod(ctx context.Context, ephemeralRunner *v1alpha1.EphemeralRunner, pod *corev1.Pod, initialRunnerID int, initialRunnerName string, log logr.Logger) error {
if pod.Status.Phase == corev1.PodSucceeded || pod.Status.Phase == corev1.PodFailed {
return nil
}
@@ -1019,15 +1004,17 @@ func (r *EphemeralRunnerReconciler) updateRunStatusFromPod(ctx context.Context,
phase = v1alpha1.EphemeralRunnerPhasePending
}
// The controller no longer promotes the runner to Running. The listener owns that
// transition and applies it when a job is assigned to this runner. The controller
// still publishes the initial Pending phase while the runner pod is starting.
// The patch below is optimistically locked so a stale cached copy of this runner
// cannot undo the listener's transition to Running.
// The listener writes only job metadata. The runner controller owns phase
// transitions and promotes an assigned Pending runner to Running without
// racing the listener's status patch.
if phase == v1alpha1.EphemeralRunnerPhasePending && ephemeralRunner.HasJob() {
phase = v1alpha1.EphemeralRunnerPhaseRunning
}
phaseChanged := phase != ephemeralRunner.Status.Phase
readyChanged := ready != ephemeralRunner.Status.Ready
identityChanged := ephemeralRunner.Status.RunnerID == 0
if !phaseChanged && !readyChanged {
if !phaseChanged && !readyChanged && !identityChanged {
return nil
}
@@ -1043,8 +1030,12 @@ func (r *EphemeralRunnerReconciler) updateRunStatusFromPod(ctx context.Context,
ephemeralRunner.Status.Ready = ready
ephemeralRunner.Status.Reason = pod.Status.Reason
ephemeralRunner.Status.Message = pod.Status.Message
if identityChanged {
ephemeralRunner.Status.RunnerID = initialRunnerID
ephemeralRunner.Status.RunnerName = initialRunnerName
}
if err := r.Status().Patch(ctx, ephemeralRunner, client.MergeFromWithOptions(original, client.MergeFromWithOptimisticLock{})); err != nil {
if err := r.Status().Patch(ctx, ephemeralRunner, client.MergeFrom(original)); err != nil {
return fmt.Errorf("failed to update runner status for Phase/Reason/Message/Ready: %w", err)
}
r.publishEphemeralRunnerPhaseMetric(ephemeralRunner, ephemeralRunner.Status.Phase, log)
@@ -1249,7 +1240,7 @@ func (r *EphemeralRunnerReconciler) SetupWithManager(mgr ctrl.Manager, opts ...O
return builderWithOptions(
ctrl.NewControllerManagedBy(mgr).
For(&v1alpha1.EphemeralRunner{}).
For(&v1alpha1.EphemeralRunner{}, builder.WithPredicates(ephemeralRunnerPredicate())).
Owns(&corev1.Pod{}, builder.WithPredicates(ephemeralRunnerOwnedPodPredicate())).
WithEventFilter(predicate.ResourceVersionChangedPredicate{}),
opts,
@@ -796,7 +796,35 @@ var _ = Describe("EphemeralRunner", func() {
).Should(BeFalse(), "EphemeralRunner-owned resources should be removed from cache after deletion")
})
It("It should eventually have runner id set", func() {
It("It should record the runner identity with the first nonterminal pod status", func() {
pod := new(corev1.Pod)
Eventually(
func() error {
return k8sClient.Get(ctx, client.ObjectKey{Name: ephemeralRunner.Name, Namespace: ephemeralRunner.Namespace}, pod)
},
ephemeralRunnerTimeout,
ephemeralRunnerInterval,
).Should(Succeed())
Consistently(
func() (int, error) {
updatedEphemeralRunner := new(v1alpha1.EphemeralRunner)
if err := k8sClient.Get(ctx, client.ObjectKey{Name: ephemeralRunner.Name, Namespace: ephemeralRunner.Namespace}, updatedEphemeralRunner); err != nil {
return 0, err
}
return updatedEphemeralRunner.Status.RunnerID, nil
},
ephemeralRunnerInterval*3,
ephemeralRunnerInterval,
).Should(BeZero(), "Pod creation alone must not publish runner identity")
pod.Status.Phase = corev1.PodPending
pod.Status.ContainerStatuses = []corev1.ContainerStatus{{
Name: v1alpha1.EphemeralRunnerContainerName,
State: corev1.ContainerState{},
}}
Expect(k8sClient.Status().Update(ctx, pod)).To(Succeed())
Eventually(
func() (int, error) {
updatedEphemeralRunner := new(v1alpha1.EphemeralRunner)
@@ -1225,7 +1253,7 @@ var _ = Describe("EphemeralRunner", func() {
).Should(BeEquivalentTo(v1alpha1.EphemeralRunnerPhaseRunning))
})
It("Controller should not set Running phase from pod status - listener owns Running transition", func() {
It("Controller sets Running phase after the listener records a job assignment", func() {
pod := new(corev1.Pod)
Eventually(
func() (bool, error) {
@@ -1255,13 +1283,6 @@ var _ = Describe("EphemeralRunner", func() {
err := k8sClient.Status().Update(ctx, pod)
Expect(err).To(BeNil())
// Two-stage on purpose. Eventually establishes that the controller does
// publish Pending even though the pod was first observed already Running
// -- the common case once the image is cached, and the only chance the
// controller gets to publish an initial phase. Consistently then holds
// that it never advances to Running, which is the listener's transition
// to make. Asserting Pending is strictly stronger than asserting empty,
// because empty is also what a controller that never ran would leave.
updated := new(v1alpha1.EphemeralRunner)
Eventually(
func() (v1alpha1.EphemeralRunnerPhase, error) {
@@ -1274,7 +1295,13 @@ var _ = Describe("EphemeralRunner", func() {
ephemeralRunnerInterval,
).Should(BeEquivalentTo(v1alpha1.EphemeralRunnerPhasePending), "controller must publish the initial Pending phase")
Consistently(
Expect(k8sClient.Get(ctx, client.ObjectKey{Name: ephemeralRunner.Name, Namespace: ephemeralRunner.Namespace}, updated)).To(Succeed())
assignment := updated.DeepCopy()
assignment.Status.JobID = "job-1"
assignment.Status.WorkflowRunID = 1
Expect(k8sClient.Status().Patch(ctx, assignment, client.MergeFrom(updated))).To(Succeed())
Eventually(
func() (v1alpha1.EphemeralRunnerPhase, error) {
updated := new(v1alpha1.EphemeralRunner)
if err := k8sClient.Get(ctx, client.ObjectKey{Name: ephemeralRunner.Name, Namespace: ephemeralRunner.Namespace}, updated); err != nil {
@@ -1283,7 +1310,8 @@ var _ = Describe("EphemeralRunner", func() {
return updated.Status.Phase, nil
},
ephemeralRunnerTimeout,
).Should(BeEquivalentTo(v1alpha1.EphemeralRunnerPhasePending), "controller must not set Running from pod status")
ephemeralRunnerInterval,
).Should(BeEquivalentTo(v1alpha1.EphemeralRunnerPhaseRunning))
Eventually(
func() (bool, error) {
@@ -37,7 +37,7 @@ import (
"sigs.k8s.io/controller-runtime/pkg/client/interceptor"
)
func TestReconcileDefersRunnerIdentityUntilPodExists(t *testing.T) {
func TestReconcileDefersRunnerIdentityUntilPodReportsStatus(t *testing.T) {
ctx := context.Background()
key := types.NamespacedName{Namespace: "default", Name: "test-runner"}
@@ -73,7 +73,7 @@ func TestReconcileDefersRunnerIdentityUntilPodExists(t *testing.T) {
c := ctrlfake.NewClientBuilder().
WithScheme(scheme).
WithObjects(runner, secret).
WithStatusSubresource(&v1alpha1.EphemeralRunner{}).
WithStatusSubresource(&v1alpha1.EphemeralRunner{}, &corev1.Pod{}).
WithInterceptorFuncs(interceptor.Funcs{
SubResourcePatch: func(ctx context.Context, clt client.Client, subResourceName string, obj client.Object, patch client.Patch, opts ...client.SubResourcePatchOption) error {
if _, ok := obj.(*v1alpha1.EphemeralRunner); ok {
@@ -121,12 +121,20 @@ func TestReconcileDefersRunnerIdentityUntilPodExists(t *testing.T) {
assert.Empty(t, getRunner().Status.RunnerName)
assert.Zero(t, statusPatchAttempts, "the runner identity must not delay Pod creation")
// This is the same state after a controller crash following Pod creation:
// the next reconcile finds the Pod and restores the identity from the JIT
// secret. A transient patch failure returns an error for reconciliation to
// retry without creating another Pod.
// Identity remains deferred until the Pod reports a non-terminal container
// status. A transient failure of that coalesced status patch returns an
// error for reconciliation to retry without creating another Pod.
pod := new(corev1.Pod)
require.NoError(t, c.Get(ctx, key, pod))
pod.Status.Phase = corev1.PodRunning
pod.Status.ContainerStatuses = []corev1.ContainerStatus{{
Name: v1alpha1.EphemeralRunnerContainerName,
State: corev1.ContainerState{Running: &corev1.ContainerStateRunning{}},
}}
require.NoError(t, c.Status().Update(ctx, pod))
_, err = newReconciler().Reconcile(ctx, ctrl.Request{NamespacedName: key})
require.ErrorContains(t, err, "failed to update runner status for RunnerId/RunnerName")
require.ErrorContains(t, err, "failed to update runner status for Phase/Reason/Message/Ready")
assert.Equal(t, 1, podCount())
assert.Zero(t, getRunner().Status.RunnerID)
assert.Empty(t, getRunner().Status.RunnerName)
@@ -828,6 +828,15 @@ func (r *EphemeralRunnerSetReconciler) cleanUpEphemeralRunners(ctx context.Conte
var errs []error
log.Info("Cleanup pending or running ephemeral runners")
for _, ephemeralRunner := range ephemeralRunnerState.pending {
if ephemeralRunner.HasJob() {
log.Info(
"Skipping ephemeral runner since it is running a job",
"name", ephemeralRunner.Name,
"workflowRunId", ephemeralRunner.Status.WorkflowRunID,
"jobId", ephemeralRunner.Status.JobID,
)
continue
}
if waitForRunnerID(ephemeralRunner) {
continue
}
@@ -2414,7 +2414,7 @@ var _ = Describe("Test EphemeralRunnerSet actionable revision cleanup", func() {
}, time.Second, ephemeralRunnerSetTestInterval).Should(Equal(int64(0)))
})
It("deletes runner-a-idle, keeps runner-b-busy, and advances applied actionable revision 3 to 4", func() {
It("deletes runner-a-idle, keeps a job-bearing pending runner, and advances applied actionable revision 3 to 4", func() {
controller := &EphemeralRunnerSetReconciler{
Client: mgr.GetClient(),
APIReader: mgr.GetAPIReader(),
@@ -2479,7 +2479,7 @@ var _ = Describe("Test EphemeralRunnerSet actionable revision cleanup", func() {
err = k8sClient.Get(ctx, client.ObjectKeyFromObject(busyRunner), busyCurrent)
Expect(err).NotTo(HaveOccurred())
busyUpdated := busyCurrent.DeepCopy()
busyUpdated.Status.Phase = v1alpha1.EphemeralRunnerPhaseRunning
busyUpdated.Status.Phase = v1alpha1.EphemeralRunnerPhasePending
busyUpdated.Status.RunnerID = 102
busyUpdated.Status.JobID = "job-1"
busyUpdated.Status.WorkflowRunID = 9001
@@ -84,6 +84,28 @@ func ephemeralRunnerSetOwnedEphemeralRunnerPredicate() predicate.Predicate {
}
}
// ephemeralRunnerPredicate filters updates sent back to the EphemeralRunner
// controller. Pod events trigger its own status writes; its only status input
// from another writer is JobID, which the listener records on assignment.
func ephemeralRunnerPredicate() predicate.Predicate {
return predicate.Funcs{
UpdateFunc: func(e event.UpdateEvent) bool {
oldRunner, oldOK := e.ObjectOld.(*v1alpha1.EphemeralRunner)
newRunner, newOK := e.ObjectNew.(*v1alpha1.EphemeralRunner)
if !oldOK || !newOK {
return true
}
if !equalReconciledObjectMeta(&oldRunner.ObjectMeta, &newRunner.ObjectMeta) ||
!equality.Semantic.DeepEqual(&oldRunner.Spec, &newRunner.Spec) {
return true
}
return oldRunner.Status.JobID != newRunner.Status.JobID
},
}
}
// ephemeralRunnerOwnedPodPredicate filters updates of the pod owned by an
// EphemeralRunner.
//
@@ -156,6 +156,66 @@ func TestEphemeralRunnerSetOwnedEphemeralRunnerPredicate(t *testing.T) {
})
}
func TestEphemeralRunnerPredicate(t *testing.T) {
base := func() *v1alpha1.EphemeralRunner {
return &v1alpha1.EphemeralRunner{
ObjectMeta: metav1.ObjectMeta{
Name: "runner",
Namespace: "default",
Generation: 1,
Finalizers: []string{"finalizer"},
},
Spec: v1alpha1.EphemeralRunnerSpec{GitHubConfigURL: "https://github.com/org/repo"},
Status: v1alpha1.EphemeralRunnerStatus{
Phase: v1alpha1.EphemeralRunnerPhasePending,
Ready: true,
RunnerID: 42,
},
}
}
t.Run("reconciles on listener job assignment", func(t *testing.T) {
old, updated := base(), base()
updated.Status.JobID = "job"
assert.True(t, ephemeralRunnerPredicate().Update(event.UpdateEvent{ObjectOld: old, ObjectNew: updated}))
})
t.Run("reconciles on metadata and spec changes", func(t *testing.T) {
for name, mutate := range map[string]func(*v1alpha1.EphemeralRunner){
"finalizer": func(r *v1alpha1.EphemeralRunner) { r.Finalizers = nil },
"spec": func(r *v1alpha1.EphemeralRunner) { r.Spec.GitHubConfigURL = "https://github.com/other/repo" },
} {
t.Run(name, func(t *testing.T) {
old, updated := base(), base()
mutate(updated)
assert.True(t, ephemeralRunnerPredicate().Update(event.UpdateEvent{ObjectOld: old, ObjectNew: updated}))
})
}
})
t.Run("ignores controller-owned status writes", func(t *testing.T) {
old, updated := base(), base()
updated.Status.Phase = v1alpha1.EphemeralRunnerPhaseRunning
updated.Status.Ready = false
updated.Status.Reason = "reason"
updated.Status.Message = "message"
updated.Status.RunnerID = 43
updated.Status.RunnerName = "runner-name"
updated.Status.Failures = map[string]metav1.Time{"pod": metav1.Now()}
assert.False(t, ephemeralRunnerPredicate().Update(event.UpdateEvent{ObjectOld: old, ObjectNew: updated}))
})
t.Run("reconciles on unexpected types", func(t *testing.T) {
assert.True(t, ephemeralRunnerPredicate().Update(event.UpdateEvent{
ObjectOld: &corev1.Pod{},
ObjectNew: &corev1.Pod{},
}))
})
}
func TestEphemeralRunnerOwnedPodPredicate(t *testing.T) {
base := func() *corev1.Pod {
return &corev1.Pod{