mirror of
https://github.com/actions-runner-controller/actions-runner-controller.git
synced 2026-10-09 12:36:28 +02:00
Switch the scale set off instead of rebuilding it when runners are outdated
When the runners reject the runner spec they were given, the listener has to stop acquiring jobs the scale set cannot run. The controller removed the listener but then deleted the EphemeralRunnerSet, and left the AutoscalingRunnerSet phase on Running. AutoscalingRunnerSetPhaseOutdated was declared and read, but never assigned by anything. Nothing held the scale set switched off as a result. The next reconcile saw a missing EphemeralRunnerSet, created it, created a listener for it, and the fresh runners rejected the same spec again, so the scale set churned through create and teardown cycles against the Actions service instead of resting. Record the outdated phase and keep the EphemeralRunnerSet, pinned to zero replicas and patch id. It releases every runner that is not executing a job while the phase keeps the listener from being rebuilt, and the revision bookkeeping that decides when the scale set may run again is preserved. Recovery is driven by the spec update that the phase is waiting for: it moves the phase back to pending, and the runner spec is then republished to the set with an advanced revision even when the runner spec itself did not change. The revision is what tells the EphemeralRunnerSet to stop judging itself by the runners that failed, so without advancing it a scale set could only be recovered by editing the pod template, and an edit to anything else would switch the listener back on against a set parked at zero. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com>
This commit is contained in:
co-authored by
Copilot App
parent
b3e44a1c39
commit
e2768dddbf
@@ -2854,3 +2854,235 @@ func unblockDeletion(listener *v1alpha1.AutoscalingListener) {
|
||||
}
|
||||
Expect(k8sClient.Patch(context.Background(), current, client.MergeFrom(original))).To(Succeed(), "failed to remove the test finalizer from the listener")
|
||||
}
|
||||
|
||||
var _ = Describe("Test AutoscalingRunnerSet outdated lifecycle", Ordered, func() {
|
||||
var originalBuildVersion string
|
||||
buildVersion := "0.1.0"
|
||||
|
||||
BeforeAll(func() {
|
||||
originalBuildVersion = build.Version
|
||||
build.Version = buildVersion
|
||||
})
|
||||
|
||||
AfterAll(func() {
|
||||
build.Version = originalBuildVersion
|
||||
})
|
||||
|
||||
Context("When the runners reject the runner spec they were given", func() {
|
||||
var ctx context.Context
|
||||
var mgr ctrl.Manager
|
||||
var autoscalingNS *corev1.Namespace
|
||||
var autoscalingRunnerSet *v1alpha1.AutoscalingRunnerSet
|
||||
|
||||
ephemeralRunnerSetKey := func() client.ObjectKey {
|
||||
return client.ObjectKey{Name: autoscalingRunnerSet.Name, Namespace: autoscalingRunnerSet.Namespace}
|
||||
}
|
||||
|
||||
listenerKey := func() client.ObjectKey {
|
||||
return client.ObjectKey{Name: scaleSetListenerName(autoscalingRunnerSet), Namespace: autoscalingRunnerSet.Namespace}
|
||||
}
|
||||
|
||||
getEphemeralRunnerSet := func() *v1alpha1.EphemeralRunnerSet {
|
||||
GinkgoHelper()
|
||||
|
||||
runnerSet := new(v1alpha1.EphemeralRunnerSet)
|
||||
Expect(k8sClient.Get(ctx, ephemeralRunnerSetKey(), runnerSet)).To(Succeed(), "failed to get the ephemeral runner set")
|
||||
return runnerSet
|
||||
}
|
||||
|
||||
autoscalingRunnerSetPhase := func() (v1alpha1.AutoscalingRunnerSetPhase, error) {
|
||||
updated := new(v1alpha1.AutoscalingRunnerSet)
|
||||
if err := k8sClient.Get(ctx, client.ObjectKeyFromObject(autoscalingRunnerSet), updated); err != nil {
|
||||
return "", err
|
||||
}
|
||||
return v1alpha1.AutoscalingRunnerSetPhase(updated.Status.Phase), nil
|
||||
}
|
||||
|
||||
// markRunnersOutdated stands in for the EphemeralRunnerSet controller
|
||||
// reporting that the runners it created rejected the runner spec. The
|
||||
// applied revision is moved up to the spec revision because that is the
|
||||
// state the report is only meaningful in: the set is running the spec it
|
||||
// is complaining about.
|
||||
markRunnersOutdated := func() {
|
||||
GinkgoHelper()
|
||||
|
||||
runnerSet := getEphemeralRunnerSet()
|
||||
original := runnerSet.DeepCopy()
|
||||
runnerSet.Status.Phase = v1alpha1.EphemeralRunnerSetPhaseOutdated
|
||||
runnerSet.Status.AppliedActionableRevision = runnerSet.Spec.ActionableRevision
|
||||
Expect(k8sClient.Status().Patch(ctx, runnerSet, client.MergeFrom(original))).To(Succeed(), "failed to mark the ephemeral runner set outdated")
|
||||
}
|
||||
|
||||
expectSwitchedOff := func() int64 {
|
||||
GinkgoHelper()
|
||||
|
||||
Eventually(autoscalingRunnerSetPhase, autoscalingRunnerSetTestTimeout, autoscalingRunnerSetTestInterval).
|
||||
Should(BeEquivalentTo(v1alpha1.AutoscalingRunnerSetPhaseOutdated), "the autoscaling runner set should report the outdated phase")
|
||||
|
||||
Eventually(
|
||||
func() bool {
|
||||
return errors.IsNotFound(k8sClient.Get(ctx, listenerKey(), new(v1alpha1.AutoscalingListener)))
|
||||
},
|
||||
autoscalingRunnerSetTestTimeout,
|
||||
autoscalingRunnerSetTestInterval,
|
||||
).Should(BeTrue(), "the listener should be removed so no further jobs are acquired")
|
||||
|
||||
// The set is kept, not deleted: deleting it would make the controller
|
||||
// rebuild it from the same rejected spec on the very next reconcile.
|
||||
var runnerSet *v1alpha1.EphemeralRunnerSet
|
||||
Eventually(
|
||||
func() (bool, error) {
|
||||
runnerSet = new(v1alpha1.EphemeralRunnerSet)
|
||||
if err := k8sClient.Get(ctx, ephemeralRunnerSetKey(), runnerSet); err != nil {
|
||||
return false, err
|
||||
}
|
||||
return runnerSet.Spec.Replicas == 0 && runnerSet.Spec.PatchID == 0, nil
|
||||
},
|
||||
autoscalingRunnerSetTestTimeout,
|
||||
autoscalingRunnerSetTestInterval,
|
||||
).Should(BeTrue(), "the ephemeral runner set should be held at zero replicas")
|
||||
|
||||
Consistently(
|
||||
func() error {
|
||||
return k8sClient.Get(ctx, ephemeralRunnerSetKey(), new(v1alpha1.EphemeralRunnerSet))
|
||||
},
|
||||
2*time.Second,
|
||||
autoscalingRunnerSetTestInterval,
|
||||
).Should(Succeed(), "the ephemeral runner set should not be deleted while the scale set is outdated")
|
||||
|
||||
return runnerSet.Spec.ActionableRevision
|
||||
}
|
||||
|
||||
expectRecovered := func(outdatedRevision int64) {
|
||||
GinkgoHelper()
|
||||
|
||||
Eventually(
|
||||
func() (int64, error) {
|
||||
runnerSet := new(v1alpha1.EphemeralRunnerSet)
|
||||
if err := k8sClient.Get(ctx, ephemeralRunnerSetKey(), runnerSet); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return runnerSet.Spec.ActionableRevision, nil
|
||||
},
|
||||
autoscalingRunnerSetTestTimeout,
|
||||
autoscalingRunnerSetTestInterval,
|
||||
).Should(BeNumerically(">", outdatedRevision), "the runner spec revision should advance so the runner set stops judging itself by the rejected runners")
|
||||
|
||||
Eventually(
|
||||
func() error {
|
||||
return k8sClient.Get(ctx, listenerKey(), new(v1alpha1.AutoscalingListener))
|
||||
},
|
||||
autoscalingRunnerSetTestTimeout,
|
||||
autoscalingRunnerSetTestInterval,
|
||||
).Should(Succeed(), "the listener should be created again so the scale set can acquire jobs")
|
||||
|
||||
Eventually(autoscalingRunnerSetPhase, autoscalingRunnerSetTestTimeout, autoscalingRunnerSetTestInterval).
|
||||
Should(BeEquivalentTo(v1alpha1.AutoscalingRunnerSetPhaseRunning), "the autoscaling runner set should leave the outdated phase")
|
||||
}
|
||||
|
||||
BeforeEach(func() {
|
||||
ctx = context.Background()
|
||||
autoscalingNS, mgr = createNamespace(GinkgoT(), k8sClient)
|
||||
configSecret := createDefaultSecret(GinkgoT(), k8sClient, autoscalingNS.Name)
|
||||
|
||||
controller := &AutoscalingRunnerSetReconciler{
|
||||
Client: mgr.GetClient(),
|
||||
Scheme: mgr.GetScheme(),
|
||||
Log: logf.Log,
|
||||
ControllerNamespace: autoscalingNS.Name,
|
||||
DefaultRunnerScaleSetListenerImage: "ghcr.io/actions/arc",
|
||||
ResourceBuilder: ResourceBuilder{
|
||||
ResourceCache: newTestResourceCache(),
|
||||
SecretResolver: secretresolver.New(mgr.GetClient(), scalefake.NewMultiClient(
|
||||
scalefake.WithClient(
|
||||
scalefake.NewClient(
|
||||
scalefake.WithGetRunnerGroupByName(&scaleset.RunnerGroup{ID: 1, Name: "testgroup"}, nil),
|
||||
scalefake.WithGetRunnerScaleSet(nil, nil),
|
||||
scalefake.WithCreateRunnerScaleSet(&scaleset.RunnerScaleSet{ID: 1, Name: "test-asrs", RunnerGroupID: 1, RunnerGroupName: "testgroup"}, nil),
|
||||
scalefake.WithDeleteRunnerScaleSet(nil),
|
||||
),
|
||||
),
|
||||
)),
|
||||
},
|
||||
}
|
||||
Expect(controller.SetupWithManager(mgr)).To(Succeed(), "failed to setup controller")
|
||||
startManagers(GinkgoT(), mgr)
|
||||
|
||||
min := 1
|
||||
max := 10
|
||||
autoscalingRunnerSet = &v1alpha1.AutoscalingRunnerSet{
|
||||
ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "test-asrs",
|
||||
Namespace: autoscalingNS.Name,
|
||||
Labels: map[string]string{LabelKeyKubernetesVersion: buildVersion},
|
||||
},
|
||||
Spec: v1alpha1.AutoscalingRunnerSetSpec{
|
||||
GitHubConfigUrl: "https://github.com/owner/repo",
|
||||
GitHubConfigSecret: configSecret.Name,
|
||||
MaxRunners: &max,
|
||||
MinRunners: &min,
|
||||
RunnerGroup: "testgroup",
|
||||
Template: corev1.PodTemplateSpec{
|
||||
Spec: corev1.PodSpec{
|
||||
Containers: []corev1.Container{
|
||||
{
|
||||
Name: "runner",
|
||||
Image: "ghcr.io/actions/runner",
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
Expect(k8sClient.Create(ctx, autoscalingRunnerSet)).To(Succeed(), "failed to create AutoScalingRunnerSet")
|
||||
|
||||
Eventually(
|
||||
func() error {
|
||||
return k8sClient.Get(ctx, ephemeralRunnerSetKey(), new(v1alpha1.EphemeralRunnerSet))
|
||||
},
|
||||
autoscalingRunnerSetTestTimeout,
|
||||
autoscalingRunnerSetTestInterval,
|
||||
).Should(Succeed(), "the ephemeral runner set should be created")
|
||||
|
||||
Eventually(autoscalingRunnerSetPhase, autoscalingRunnerSetTestTimeout, autoscalingRunnerSetTestInterval).
|
||||
Should(BeEquivalentTo(v1alpha1.AutoscalingRunnerSetPhaseRunning), "the autoscaling runner set should settle before the runners reject the spec")
|
||||
})
|
||||
|
||||
It("switches the scale set off and keeps the ephemeral runner set", func() {
|
||||
markRunnersOutdated()
|
||||
expectSwitchedOff()
|
||||
})
|
||||
|
||||
It("recovers when the runner spec is corrected", func() {
|
||||
markRunnersOutdated()
|
||||
outdatedRevision := expectSwitchedOff()
|
||||
|
||||
updated := new(v1alpha1.AutoscalingRunnerSet)
|
||||
Expect(k8sClient.Get(ctx, client.ObjectKeyFromObject(autoscalingRunnerSet), updated)).To(Succeed())
|
||||
original := updated.DeepCopy()
|
||||
updated.Spec.Template.Spec.Containers[0].Image = "ghcr.io/actions/runner:fixed"
|
||||
Expect(k8sClient.Patch(ctx, updated, client.MergeFrom(original))).To(Succeed(), "failed to correct the runner spec")
|
||||
|
||||
expectRecovered(outdatedRevision)
|
||||
})
|
||||
|
||||
// The runner spec is not the only reason a scale set can be stuck: the
|
||||
// runners may have been rejected because of the scale set registration
|
||||
// rather than the pod template. Any spec edit therefore has to be enough
|
||||
// to retry, otherwise the scale set can only be recovered by touching a
|
||||
// field that has nothing to do with the failure.
|
||||
It("recovers when a field outside the runner spec is updated", func() {
|
||||
markRunnersOutdated()
|
||||
outdatedRevision := expectSwitchedOff()
|
||||
|
||||
updated := new(v1alpha1.AutoscalingRunnerSet)
|
||||
Expect(k8sClient.Get(ctx, client.ObjectKeyFromObject(autoscalingRunnerSet), updated)).To(Succeed())
|
||||
original := updated.DeepCopy()
|
||||
max := 20
|
||||
updated.Spec.MaxRunners = &max
|
||||
Expect(k8sClient.Patch(ctx, updated, client.MergeFrom(original))).To(Succeed(), "failed to update the autoscaling runner set")
|
||||
|
||||
expectRecovered(outdatedRevision)
|
||||
})
|
||||
})
|
||||
})
|
||||
|
||||
Reference in New Issue
Block a user