Hand a finished runner's pod back as soon as its job is over (#4673)

This commit is contained in:
Nikola Jokic
2026-09-23 14:12:11 +02:00
committed by GitHub
parent 390ceb03f6
commit 2adef8f8c4
13 changed files with 797 additions and 52 deletions
@@ -59,6 +59,9 @@ args:
{{- with .Values.controller.manager.config.ephemeralRunnerMaxConcurrentReconciles }}
- "--ephemeral-runner-max-concurrent-reconciles={{ . }}"
{{- end }}
{{- with .Values.controller.manager.config.terminatedRunnerPodGracePeriodSeconds }}
- "--terminated-runner-pod-grace-period-seconds={{ . }}"
{{- end }}
{{- if .Values.controller.metrics }}
{{- with .Values.controller.metrics }}
- "--listener-metrics-addr={{ .listenerAddr }}"
@@ -119,3 +119,34 @@ tests:
- contains:
path: spec.template.spec.containers[0].args
content: "--ephemeral-runner-max-concurrent-reconciles=20"
# Unset, the controller keeps its own default of removing a finished runner pod
# immediately. The exact match above is what pins that no flag is rendered for it.
- it: should render the terminated runner pod grace period when configured
set:
controller:
manager:
config:
terminatedRunnerPodGracePeriodSeconds: 30
release:
name: "test-arc"
namespace: "test-ns"
asserts:
- contains:
path: spec.template.spec.containers[0].args
content: "--terminated-runner-pod-grace-period-seconds=30"
# A negative value hands the grace period back to the pod itself.
- it: should render a negative terminated runner pod grace period
set:
controller:
manager:
config:
terminatedRunnerPodGracePeriodSeconds: -1
release:
name: "test-arc"
namespace: "test-ns"
asserts:
- contains:
path: spec.template.spec.containers[0].args
content: "--terminated-runner-pod-grace-period-seconds=-1"
@@ -39,6 +39,15 @@ controller:
# three only pay off when many runner scale sets exist.
ephemeralRunnerMaxConcurrentReconciles: 4
# How long a runner pod whose containers have all exited is kept before it is deleted.
# Zero, the default, removes the pod as soon as its job is over instead of leaving it
# Terminating while the kubelet cleans up locally, handing its name, its scheduling slot
# and its share of any ResourceQuota straight to the runner replacing it. Raise it to keep
# finished pods around for longer, for example to give a log collector time to read them.
# A negative value leaves the pod's own terminationGracePeriodSeconds in charge. Pods that
# are still running are always deleted gracefully.
terminatedRunnerPodGracePeriodSeconds: null
# List of label prefixes that should NOT be propagated to internal resources.
excludeLabelPropagationPrefixes: []
# Example:
@@ -79,6 +79,9 @@ spec:
{{- with .Values.flags.ephemeralRunnerMaxConcurrentReconciles }}
- "--ephemeral-runner-max-concurrent-reconciles={{ . }}"
{{- end }}
{{- with .Values.flags.terminatedRunnerPodGracePeriodSeconds }}
- "--terminated-runner-pod-grace-period-seconds={{ . }}"
{{- end }}
{{- if .Values.metrics }}
{{- with .Values.metrics }}
- "--listener-metrics-addr={{ .listenerAddr }}"
@@ -866,6 +866,55 @@ func TestTemplate_ControllerDeployment_MaxConcurrentReconciles(t *testing.T) {
})
}
// TestTemplate_ControllerDeployment_TerminatedRunnerPodGracePeriod pins how the
// grace period of a finished runner pod reaches the controller.
//
// Leaving the value unset has to render no flag at all, so the controller keeps
// its own default of removing those pods immediately. Setting it is what a
// cluster that wants finished pods to stay readable for a while does, and a
// negative value is how it asks for the pod's own grace period instead.
func TestTemplate_ControllerDeployment_TerminatedRunnerPodGracePeriod(t *testing.T) {
t.Parallel()
helmChartPath, err := filepath.Abs("../../gha-runner-scale-set-controller")
require.NoError(t, err)
releaseName := "test-arc"
namespaceName := "test-" + strings.ToLower(random.UniqueID())
renderArgs := func(t *testing.T, values map[string]string) []string {
options := &helm.Options{
Logger: logger.Discard,
SetValues: values,
KubectlOptions: k8s.NewKubectlOptions("", "", namespaceName),
}
output := helm.RenderTemplateContext(t, t.Context(), options, helmChartPath, releaseName, []string{"templates/deployment.yaml"})
var deployment appsv1.Deployment
helm.UnmarshalK8SYaml(t, output, &deployment)
require.Len(t, deployment.Spec.Template.Spec.Containers, 1)
return deployment.Spec.Template.Spec.Containers[0].Args
}
t.Run("no flag renders by default", func(t *testing.T) {
for _, arg := range renderArgs(t, nil) {
assert.NotContains(t, arg, "--terminated-runner-pod-grace-period-seconds")
}
})
t.Run("the flag renders when configured", func(t *testing.T) {
args := renderArgs(t, map[string]string{"flags.terminatedRunnerPodGracePeriodSeconds": "30"})
assert.Contains(t, args, "--terminated-runner-pod-grace-period-seconds=30")
})
t.Run("the flag renders when it hands the grace period back to the pod", func(t *testing.T) {
args := renderArgs(t, map[string]string{"flags.terminatedRunnerPodGracePeriodSeconds": "-1"})
assert.Contains(t, args, "--terminated-runner-pod-grace-period-seconds=-1")
})
}
func TestTemplate_ControllerContainerEnvironmentVariables(t *testing.T) {
t.Parallel()
@@ -126,6 +126,15 @@ flags:
## sets exist.
ephemeralRunnerMaxConcurrentReconciles: 4
## How long a runner pod whose containers have all exited is kept before it is deleted.
## Zero, the default, removes the pod as soon as its job is over instead of leaving it
## Terminating while the kubelet cleans up locally, handing its name, its scheduling slot
## and its share of any ResourceQuota straight to the runner replacing it.
## Raise it to keep finished pods around for longer, for example to give a log collector
## time to read them. A negative value leaves the pod's own terminationGracePeriodSeconds
## in charge. Pods that are still running are always deleted gracefully.
# terminatedRunnerPodGracePeriodSeconds: 0
## Defines a list of prefixes that should not be propagated to internal resources.
## This is useful when you have labels that are used for internal purposes and should not be propagated to internal resources.
## See https://github.com/actions/actions-runner-controller/issues/3533 for more information.