mirror of
https://github.com/actions-runner-controller/actions-runner-controller.git
synced 2026-09-30 03:53:12 +02:00
Let the listener own the EphemeralRunner Running phase transition (#4646)
Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com>
This commit is contained in:
co-authored by
Copilot App
Copilot Autofix powered by AI
parent
d386789092
commit
9ce3169df3
@@ -834,6 +834,7 @@ func (r *EphemeralRunnerReconciler) createSecret(ctx context.Context, runner *v1
|
||||
|
||||
// updateRunStatusFromPod is responsible for updating non-exiting statuses.
|
||||
// It should never update phase to Failed or Succeeded
|
||||
// It should never update phase to Running (the listener owns that transition)
|
||||
//
|
||||
// The event should not be re-queued since the termination status should be set
|
||||
// before proceeding with reconciliation logic
|
||||
@@ -851,8 +852,25 @@ func (r *EphemeralRunnerReconciler) updateRunStatusFromPod(ctx context.Context,
|
||||
}
|
||||
}
|
||||
|
||||
phase := v1alpha1.EphemeralRunnerPhase(pod.Status.Phase)
|
||||
phaseChanged := ephemeralRunner.Status.Phase != phase
|
||||
// Publish Pending as soon as the runner is observed non-terminal, regardless of
|
||||
// the pod phase. The controller only reaches this point once the runner
|
||||
// container status exists, and by then the pod has usually already advanced to
|
||||
// Running, so keying the initial phase off PodPending would leave a runner
|
||||
// phase-empty for its whole life -- omitted from the phase metrics, and in
|
||||
// breach of the documented contract that Pending means "created, no job yet".
|
||||
// Guarding on the empty phase alone is sufficient: every terminal phase, and
|
||||
// Running itself, is non-empty, so this can never overwrite one.
|
||||
phase := ephemeralRunner.Status.Phase
|
||||
if phase == "" {
|
||||
phase = v1alpha1.EphemeralRunnerPhasePending
|
||||
}
|
||||
|
||||
// The controller no longer promotes the runner to Running. The listener owns that
|
||||
// transition and applies it when a job is assigned to this runner. The controller
|
||||
// still publishes the initial Pending phase while the runner pod is starting.
|
||||
// The patch below is optimistically locked so a stale cached copy of this runner
|
||||
// cannot undo the listener's transition to Running.
|
||||
phaseChanged := phase != ephemeralRunner.Status.Phase
|
||||
readyChanged := ready != ephemeralRunner.Status.Ready
|
||||
|
||||
if !phaseChanged && !readyChanged {
|
||||
@@ -872,7 +890,7 @@ func (r *EphemeralRunnerReconciler) updateRunStatusFromPod(ctx context.Context,
|
||||
ephemeralRunner.Status.Reason = pod.Status.Reason
|
||||
ephemeralRunner.Status.Message = pod.Status.Message
|
||||
|
||||
if err := r.Status().Patch(ctx, ephemeralRunner, client.MergeFrom(original)); err != nil {
|
||||
if err := r.Status().Patch(ctx, ephemeralRunner, client.MergeFromWithOptions(original, client.MergeFromWithOptimisticLock{})); err != nil {
|
||||
return fmt.Errorf("failed to update runner status for Phase/Reason/Message/Ready: %w", err)
|
||||
}
|
||||
r.publishEphemeralRunnerPhaseMetric(ephemeralRunner, ephemeralRunner.Status.Phase, log)
|
||||
|
||||
@@ -825,32 +825,47 @@ var _ = Describe("EphemeralRunner", func() {
|
||||
ephemeralRunnerInterval,
|
||||
).Should(BeEquivalentTo(true))
|
||||
|
||||
for _, phase := range []corev1.PodPhase{corev1.PodRunning, corev1.PodPending} {
|
||||
podCopy := pod.DeepCopy()
|
||||
pod.Status.Phase = phase
|
||||
// set container state to force status update
|
||||
pod.Status.ContainerStatuses = append(pod.Status.ContainerStatuses, corev1.ContainerStatus{
|
||||
Name: v1alpha1.EphemeralRunnerContainerName,
|
||||
State: corev1.ContainerState{},
|
||||
})
|
||||
podCopy := pod.DeepCopy()
|
||||
pod.Status.Phase = corev1.PodPending
|
||||
// set container state to force status update
|
||||
pod.Status.ContainerStatuses = append(pod.Status.ContainerStatuses, corev1.ContainerStatus{
|
||||
Name: v1alpha1.EphemeralRunnerContainerName,
|
||||
State: corev1.ContainerState{},
|
||||
})
|
||||
|
||||
err := k8sClient.Status().Patch(ctx, pod, client.MergeFrom(podCopy))
|
||||
Expect(err).To(BeNil(), "failed to patch pod status")
|
||||
err := k8sClient.Status().Patch(ctx, pod, client.MergeFrom(podCopy))
|
||||
Expect(err).To(BeNil(), "failed to patch pod status")
|
||||
|
||||
var updated *v1alpha1.EphemeralRunner
|
||||
Eventually(
|
||||
func() (v1alpha1.EphemeralRunnerPhase, error) {
|
||||
updated = new(v1alpha1.EphemeralRunner)
|
||||
err := k8sClient.Get(ctx, client.ObjectKey{Name: ephemeralRunner.Name, Namespace: ephemeralRunner.Namespace}, updated)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return updated.Status.Phase, nil
|
||||
},
|
||||
ephemeralRunnerTimeout,
|
||||
ephemeralRunnerInterval,
|
||||
).Should(BeEquivalentTo(phase))
|
||||
}
|
||||
Eventually(
|
||||
func() (v1alpha1.EphemeralRunnerPhase, error) {
|
||||
updated := new(v1alpha1.EphemeralRunner)
|
||||
err := k8sClient.Get(ctx, client.ObjectKey{Name: ephemeralRunner.Name, Namespace: ephemeralRunner.Namespace}, updated)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return updated.Status.Phase, nil
|
||||
},
|
||||
ephemeralRunnerTimeout,
|
||||
ephemeralRunnerInterval,
|
||||
).Should(BeEquivalentTo(v1alpha1.EphemeralRunnerPhasePending))
|
||||
|
||||
podCopy = pod.DeepCopy()
|
||||
pod.Status.Phase = corev1.PodRunning
|
||||
err = k8sClient.Status().Patch(ctx, pod, client.MergeFrom(podCopy))
|
||||
Expect(err).To(BeNil(), "failed to patch pod status")
|
||||
|
||||
Consistently(
|
||||
func() (v1alpha1.EphemeralRunnerPhase, error) {
|
||||
updated := new(v1alpha1.EphemeralRunner)
|
||||
err := k8sClient.Get(ctx, client.ObjectKey{Name: ephemeralRunner.Name, Namespace: ephemeralRunner.Namespace}, updated)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return updated.Status.Phase, nil
|
||||
},
|
||||
ephemeralRunnerInterval*3,
|
||||
ephemeralRunnerInterval,
|
||||
).Should(BeEquivalentTo(v1alpha1.EphemeralRunnerPhasePending), "controller should not set Running from pod status")
|
||||
})
|
||||
|
||||
It("It should update ready based on the latest condition", func() {
|
||||
@@ -1173,7 +1188,6 @@ var _ = Describe("EphemeralRunner", func() {
|
||||
ephemeralRunnerInterval,
|
||||
).Should(BeEquivalentTo(true))
|
||||
|
||||
// first set phase to running
|
||||
pod.Status.ContainerStatuses = append(pod.Status.ContainerStatuses, corev1.ContainerStatus{
|
||||
Name: v1alpha1.EphemeralRunnerContainerName,
|
||||
State: corev1.ContainerState{
|
||||
@@ -1186,19 +1200,15 @@ var _ = Describe("EphemeralRunner", func() {
|
||||
err := k8sClient.Status().Update(ctx, pod)
|
||||
Expect(err).To(BeNil())
|
||||
|
||||
Eventually(
|
||||
func() (v1alpha1.EphemeralRunnerPhase, error) {
|
||||
updated := new(v1alpha1.EphemeralRunner)
|
||||
if err := k8sClient.Get(ctx, client.ObjectKey{Name: ephemeralRunner.Name, Namespace: ephemeralRunner.Namespace}, updated); err != nil {
|
||||
return "", err
|
||||
}
|
||||
return updated.Status.Phase, nil
|
||||
},
|
||||
ephemeralRunnerTimeout,
|
||||
ephemeralRunnerInterval,
|
||||
).Should(BeEquivalentTo(v1alpha1.EphemeralRunnerPhaseRunning))
|
||||
updated := new(v1alpha1.EphemeralRunner)
|
||||
err = k8sClient.Get(ctx, client.ObjectKey{Name: ephemeralRunner.Name, Namespace: ephemeralRunner.Namespace}, updated)
|
||||
Expect(err).To(BeNil())
|
||||
|
||||
original := updated.DeepCopy()
|
||||
updated.Status.Phase = v1alpha1.EphemeralRunnerPhaseRunning
|
||||
err = k8sClient.Status().Patch(ctx, updated, client.MergeFrom(original))
|
||||
Expect(err).To(BeNil())
|
||||
|
||||
// set phase to succeeded
|
||||
pod.Status.Phase = corev1.PodSucceeded
|
||||
err = k8sClient.Status().Update(ctx, pod)
|
||||
Expect(err).To(BeNil())
|
||||
@@ -1214,6 +1224,78 @@ var _ = Describe("EphemeralRunner", func() {
|
||||
ephemeralRunnerTimeout,
|
||||
).Should(BeEquivalentTo(v1alpha1.EphemeralRunnerPhaseRunning))
|
||||
})
|
||||
|
||||
It("Controller should not set Running phase from pod status - listener owns Running transition", func() {
|
||||
pod := new(corev1.Pod)
|
||||
Eventually(
|
||||
func() (bool, error) {
|
||||
if err := k8sClient.Get(ctx, client.ObjectKey{Name: ephemeralRunner.Name, Namespace: ephemeralRunner.Namespace}, pod); err != nil {
|
||||
return false, err
|
||||
}
|
||||
return true, nil
|
||||
},
|
||||
ephemeralRunnerTimeout,
|
||||
ephemeralRunnerInterval,
|
||||
).Should(BeEquivalentTo(true))
|
||||
|
||||
pod.Status.ContainerStatuses = append(pod.Status.ContainerStatuses, corev1.ContainerStatus{
|
||||
Name: v1alpha1.EphemeralRunnerContainerName,
|
||||
State: corev1.ContainerState{
|
||||
Running: &corev1.ContainerStateRunning{
|
||||
StartedAt: metav1.Now(),
|
||||
},
|
||||
},
|
||||
})
|
||||
pod.Status.Phase = corev1.PodRunning
|
||||
pod.Status.Conditions = append(pod.Status.Conditions, corev1.PodCondition{
|
||||
Type: corev1.PodReady,
|
||||
Status: corev1.ConditionTrue,
|
||||
LastTransitionTime: metav1.Now(),
|
||||
})
|
||||
err := k8sClient.Status().Update(ctx, pod)
|
||||
Expect(err).To(BeNil())
|
||||
|
||||
// Two-stage on purpose. Eventually establishes that the controller does
|
||||
// publish Pending even though the pod was first observed already Running
|
||||
// -- the common case once the image is cached, and the only chance the
|
||||
// controller gets to publish an initial phase. Consistently then holds
|
||||
// that it never advances to Running, which is the listener's transition
|
||||
// to make. Asserting Pending is strictly stronger than asserting empty,
|
||||
// because empty is also what a controller that never ran would leave.
|
||||
updated := new(v1alpha1.EphemeralRunner)
|
||||
Eventually(
|
||||
func() (v1alpha1.EphemeralRunnerPhase, error) {
|
||||
if err := k8sClient.Get(ctx, client.ObjectKey{Name: ephemeralRunner.Name, Namespace: ephemeralRunner.Namespace}, updated); err != nil {
|
||||
return "Unknown", err
|
||||
}
|
||||
return updated.Status.Phase, nil
|
||||
},
|
||||
ephemeralRunnerTimeout,
|
||||
ephemeralRunnerInterval,
|
||||
).Should(BeEquivalentTo(v1alpha1.EphemeralRunnerPhasePending), "controller must publish the initial Pending phase")
|
||||
|
||||
Consistently(
|
||||
func() (v1alpha1.EphemeralRunnerPhase, error) {
|
||||
updated := new(v1alpha1.EphemeralRunner)
|
||||
if err := k8sClient.Get(ctx, client.ObjectKey{Name: ephemeralRunner.Name, Namespace: ephemeralRunner.Namespace}, updated); err != nil {
|
||||
return "Unknown", err
|
||||
}
|
||||
return updated.Status.Phase, nil
|
||||
},
|
||||
ephemeralRunnerTimeout,
|
||||
).Should(BeEquivalentTo(v1alpha1.EphemeralRunnerPhasePending), "controller must not set Running from pod status")
|
||||
|
||||
Eventually(
|
||||
func() (bool, error) {
|
||||
if err := k8sClient.Get(ctx, client.ObjectKey{Name: ephemeralRunner.Name, Namespace: ephemeralRunner.Namespace}, updated); err != nil {
|
||||
return false, err
|
||||
}
|
||||
return updated.Status.Ready, nil
|
||||
},
|
||||
ephemeralRunnerTimeout,
|
||||
ephemeralRunnerInterval,
|
||||
).Should(BeEquivalentTo(true))
|
||||
})
|
||||
})
|
||||
|
||||
Describe("Checking the API", func() {
|
||||
|
||||
@@ -1840,6 +1840,14 @@ var _ = Describe("EphemeralRunner phase metrics", func() {
|
||||
err = k8sClient.Status().Patch(ctx, podRunning, client.MergeFrom(podPending))
|
||||
Expect(err).NotTo(HaveOccurred(), "failed to patch pod to running")
|
||||
|
||||
runnerRunning := new(v1alpha1.EphemeralRunner)
|
||||
err = k8sClient.Get(ctx, client.ObjectKey{Name: ephemeralRunner.Name, Namespace: ephemeralRunner.Namespace}, runnerRunning)
|
||||
Expect(err).NotTo(HaveOccurred(), "failed to get ephemeral runner before listener-owned running patch")
|
||||
runnerRunningOriginal := runnerRunning.DeepCopy()
|
||||
runnerRunning.Status.Phase = v1alpha1.EphemeralRunnerPhaseRunning
|
||||
err = k8sClient.Status().Patch(ctx, runnerRunning, client.MergeFrom(runnerRunningOriginal))
|
||||
Expect(err).NotTo(HaveOccurred(), "failed to simulate listener running phase patch")
|
||||
|
||||
_, err = controller.Reconcile(ctx, request)
|
||||
Expect(err).NotTo(HaveOccurred(), "failed to reconcile running pod")
|
||||
expectEphemeralRunnerPhase(ctx, ephemeralRunner, v1alpha1.EphemeralRunnerPhaseRunning)
|
||||
|
||||
@@ -39,7 +39,7 @@ var (
|
||||
prometheus.GaugeOpts{
|
||||
Subsystem: githubScaleSetControllerSubsystem,
|
||||
Name: "pending_ephemeral_runners",
|
||||
Help: "Number of ephemeral runners in a pending state.",
|
||||
Help: "Number of ephemeral runners that have not been assigned a job yet.",
|
||||
},
|
||||
labels,
|
||||
)
|
||||
@@ -47,7 +47,7 @@ var (
|
||||
prometheus.GaugeOpts{
|
||||
Subsystem: githubScaleSetControllerSubsystem,
|
||||
Name: "running_ephemeral_runners",
|
||||
Help: "Number of ephemeral runners in a running state.",
|
||||
Help: "Number of ephemeral runners that have been assigned a job.",
|
||||
},
|
||||
labels,
|
||||
)
|
||||
|
||||
@@ -1022,7 +1022,12 @@ func rulesForListenerRole(resourceNames []string) []rbacv1.PolicyRule {
|
||||
},
|
||||
{
|
||||
APIGroups: []string{"actions.github.com"},
|
||||
Resources: []string{"ephemeralrunners", "ephemeralrunners/status"},
|
||||
Resources: []string{"ephemeralrunners"},
|
||||
Verbs: []string{"get", "patch"},
|
||||
},
|
||||
{
|
||||
APIGroups: []string{"actions.github.com"},
|
||||
Resources: []string{"ephemeralrunners/status"},
|
||||
Verbs: []string{"patch"},
|
||||
},
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user