feat: add --repo-retries for helm repo and registry login commands (#2683)

* feat: add --repo-retries for retrying helm repo and registry login commands

Add a configurable retry mechanism for chart repository operations to
handle unstable networks (corporate proxies, slow internal registries).

Closes #1894

- New --repo-retries N flag and HELMFILE_REPO_RETRIES env var (default 0
  = opt-in, backward compatible)
- Retry applies to helm repo add, helm repo update (incl. ACR), and
  helm registry login with exponential backoff (1s, 2s, 4s, ..., capped 30s)
- Single retryRepoOp helper; per-attempt args/buffer are local to avoid
  state leaking across retries
- Tests cover succeed-after-retry, exhausted-retries, disabled-by-default,
  and regression guards for password-buffer and args-accumulation

Signed-off-by: yxxhero <aiopsclub@163.com>

* fix: address PR review (overflow guard, cancellable sleep, flag-override, docs)

Address Copilot review feedback on #2683:

- Cap backoff shift exponent at 5 to prevent time.Duration overflow on
  large --repo-retries values
- Make retry sleep context-aware (sleepCtx) so Ctrl+C aborts the retry
  loop promptly via the ShellRunner context
- Log a concise exit status instead of the verbose ExitError dump, and
  clarify the retry-counter wording ('retry N/M')
- Use -1 sentinel as the CLI default so --repo-retries=0 can explicitly
  disable retries even when HELMFILE_REPO_RETRIES is set
- Align help text and docs: retry applies 'on failure' (not just
  transient errors), document the 0-disables behavior
- Add tests for overflow guard, cancellable sleep, and flag-zero-disables

Signed-off-by: yxxhero <aiopsclub@163.com>

* fix: abort retries on canceled context, hide sentinel default, align comment

Address follow-up Copilot review on #2683:

- Fix tight-loop bug: sleepCtx now returns whether it completed vs was
  interrupted by context cancellation, and retryRepoOp aborts the retry
  loop on interruption so Ctrl+C no longer spins into rapid helm calls
- Hide the -1 sentinel from --help by overriding the displayed default
  to 0 (pflag DefValue), matching the documented default while keeping
  the flag-override semantics
- Correct HelmExecOptions.RepoRetry comment: 'on failure' not 'transient
  network errors', matching the actual retry behavior
- Add Test_Retry_AbortsOnCanceledContext covering the no-tight-loop path

Signed-off-by: yxxhero <aiopsclub@163.com>

* fix: copy args per retry in RegistryLogin, make cancel test deterministic

Address follow-up Copilot review on #2683:

- RegistryLogin: pass a per-attempt copy of args to execStdIn so its
  internal append (for helm.extra) can't alias the shared slice across
  retries
- Test_Retry_AbortsOnCanceledContext: cancel the context deterministically
  inside the op closure after the first attempt, replacing the flaky
  time.Sleep(20ms) goroutine

Signed-off-by: yxxhero <aiopsclub@163.com>

* fix: return error on unknown managed repo type instead of silent skip

Address Copilot review on #2683: AddRepo logged an error for an unknown
managed type but returned nil, silently succeeding while skipping the
repo add. Now returns an error so misconfigurations fail loudly.

Signed-off-by: yxxhero <aiopsclub@163.com>

---------

Signed-off-by: yxxhero <aiopsclub@163.com>
This commit is contained in:
yxxhero
2026-07-23 18:10:04 +08:00
committed by GitHub
parent f64f3ec197
commit 7bebfab71a
9 changed files with 425 additions and 41 deletions
+3
View File
@@ -37,6 +37,7 @@ type App struct {
DisableForceUpdate bool
EnforcePluginVerification bool
HelmOCIPlainHTTP bool
RepoRetry int
DisableKubeVersionAutoDetection bool
SequentialHelmfiles bool
@@ -87,6 +88,7 @@ func New(conf ConfigProvider) *App {
DisableForceUpdate: conf.DisableForceUpdate(),
EnforcePluginVerification: conf.EnforcePluginVerification(),
HelmOCIPlainHTTP: conf.HelmOCIPlainHTTP(),
RepoRetry: conf.RepoRetry(),
SequentialHelmfiles: conf.SequentialHelmfiles(),
Logger: conf.Logger(),
Kubeconfig: conf.Kubeconfig(),
@@ -983,6 +985,7 @@ func (a *App) getHelm(st *state.HelmState) (helmexec.Interface, error) {
DisableForceUpdate: a.DisableForceUpdate,
EnforcePluginVerification: a.EnforcePluginVerification,
HelmOCIPlainHTTP: a.HelmOCIPlainHTTP,
RepoRetry: a.RepoRetry,
}, a.Logger, kubeconfig, kubectx, &helmexec.ShellRunner{
Logger: a.Logger,
Ctx: a.ctx,
+1
View File
@@ -15,6 +15,7 @@ type ConfigProvider interface {
DisableForceUpdate() bool
EnforcePluginVerification() bool
HelmOCIPlainHTTP() bool
RepoRetry() int
SkipDeps() bool
SkipRefresh() bool
SequentialHelmfiles() bool
+21
View File
@@ -5,6 +5,7 @@ import (
"fmt"
"io"
"os"
"strconv"
"go.uber.org/zap"
"golang.org/x/term"
@@ -42,6 +43,11 @@ type GlobalOptions struct {
EnforcePluginVerification bool
// HelmOCIPlainHTTP is true if Helm should use plain HTTP for OCI registries
HelmOCIPlainHTTP bool
// RepoRetry is the number of times to retry "helm repo add/update" and
// "helm registry login" on failure with exponential backoff.
// A negative value (the CLI default sentinel) means "unset" and falls back
// to the HELMFILE_REPO_RETRIES env var; 0 explicitly disables retries.
RepoRetry int
// Quiet is true if the output should be quiet.
Quiet bool
// Kubeconfig is the path to the kubeconfig file to use.
@@ -278,6 +284,21 @@ func (g *GlobalImpl) HelmOCIPlainHTTP() bool {
return g.GlobalOptions.HelmOCIPlainHTTP
}
// RepoRetry returns the number of times to retry helm repo and registry login
// operations on failure, with exponential backoff. A negative value means the
// flag was not specified, so the HELMFILE_REPO_RETRIES env var is consulted;
// this lets --repo-retries=0 explicitly disable retries even when the env var
// is set.
func (g *GlobalImpl) RepoRetry() int {
if g.GlobalOptions.RepoRetry >= 0 {
return g.GlobalOptions.RepoRetry
}
if v, err := strconv.Atoi(os.Getenv(envvar.RepoRetry)); err == nil && v > 0 {
return v
}
return 0
}
// SequentialHelmfiles returns whether to process helmfile.d files sequentially
func (g *GlobalImpl) SequentialHelmfiles() bool {
return g.GlobalOptions.SequentialHelmfiles
+28
View File
@@ -420,3 +420,31 @@ func TestColorFlagOverridesNoColorEnv(t *testing.T) {
})
}
}
// TestRepoRetry tests the repo-retries flag and HELMFILE_REPO_RETRIES env var fallback
func TestRepoRetry(t *testing.T) {
// RepoRetry < 0 means the flag was not specified (CLI default sentinel).
unset := GlobalOptions{RepoRetry: -1}
tests := []struct {
name string
opts GlobalOptions
env string
expected int
}{
{name: "default (unset)", opts: unset, env: "", expected: 0},
{name: "env set", opts: unset, env: "3", expected: 3},
{name: "flag set", opts: GlobalOptions{RepoRetry: 5}, env: "", expected: 5},
{name: "flag overrides env", opts: GlobalOptions{RepoRetry: 5}, env: "3", expected: 5},
{name: "flag zero disables env", opts: GlobalOptions{RepoRetry: 0}, env: "3", expected: 0},
{name: "invalid env ignored", opts: unset, env: "abc", expected: 0},
{name: "negative env ignored", opts: unset, env: "-1", expected: 0},
}
for _, test := range tests {
t.Run(test.name, func(t *testing.T) {
t.Setenv(envvar.RepoRetry, test.env)
received := NewGlobalImpl(&test.opts).RepoRetry()
require.Equalf(t, test.expected, received, "RepoRetry expected %d, received %d", test.expected, received)
})
}
}
+1
View File
@@ -25,6 +25,7 @@ const (
GoYamlV3 = "HELMFILE_GO_YAML_V3"
CacheHome = "HELMFILE_CACHE_HOME"
Interactive = "HELMFILE_INTERACTIVE"
RepoRetry = "HELMFILE_REPO_RETRIES"
RenderYaml = "HELMFILE_RENDER_YAML" // force helmfile.yaml to be rendered as template regardless of extension, expecting "true" lower case
// AWSSDKLogLevel controls AWS SDK logging level
+136 -39
View File
@@ -3,6 +3,7 @@ package helmexec
import (
"bytes"
"context"
"errors"
"fmt"
"io"
"net/url"
@@ -12,6 +13,7 @@ import (
"strconv"
"strings"
"sync"
"time"
"unicode"
"github.com/Masterminds/semver/v3"
@@ -38,6 +40,9 @@ type HelmExecOptions struct {
DisableForceUpdate bool // If true, do not force helm repos to update when executing "helm repo add" (Helm 3)
EnforcePluginVerification bool // If true, fail plugin installation if verification is not supported
HelmOCIPlainHTTP bool // If true, use plain HTTP for OCI registries
// RepoRetry is the number of times to retry helm repo and registry login
// operations on failure, with exponential backoff. 0 disables retries.
RepoRetry int
}
type execer struct {
@@ -272,11 +277,91 @@ func (helm *execer) SetDisableForceUpdate(forceUpdate bool) {
helm.options.DisableForceUpdate = forceUpdate
}
func (helm *execer) AddRepo(name, repository, cafile, certfile, keyfile, username, password string, managed string, passCredentials, skipTLSVerify bool) error {
var args []string
// repoRetryBaseBackoff is the base unit for exponential backoff between repo
// operation retries (doubles each attempt, capped at 30x). Exposed as a
// package-level variable so tests can shrink it to avoid real sleeps.
var repoRetryBaseBackoff = time.Second
// retryRepoOp runs op, retrying up to helm.options.RepoRetry times on failure
// with exponential backoff (1x, 2x, 4x, ..., capped at 30x the base unit). It
// returns the output from the (last) attempt. A zero/negative RepoRetry
// disables retrying — op runs exactly once.
func (helm *execer) retryRepoOp(name string, op func() ([]byte, error)) ([]byte, error) {
maxRetries := helm.options.RepoRetry
var out []byte
var err error
for attempt := 0; ; attempt++ {
out, err = op()
if err == nil || maxRetries <= 0 || attempt >= maxRetries {
return out, err
}
// Cap the shift exponent at 5 (2^5 = 32x already exceeds the 30x cap)
// so very large --repo-retries values can't overflow time.Duration.
shift := attempt
if shift > 5 {
shift = 5
}
backoff := repoRetryBaseBackoff * time.Duration(1<<shift)
if backoff > 30*repoRetryBaseBackoff {
backoff = 30 * repoRetryBaseBackoff
}
helm.logger.Warnf("repo operation %q failed (%s); retry %d/%d in %v",
name, conciseError(err), attempt+1, maxRetries, backoff)
// sleepCtx returns false if interrupted by context cancellation; in that
// case stop retrying so a canceled context (Ctrl+C) doesn't spin into a
// tight loop of rapid helm invocations.
if !helm.sleepCtx(backoff) {
return out, err
}
}
}
// conciseError returns a short reason for logging. For a helm ExitError it
// reports just the exit status, avoiding the very verbose PATH/ARGS/OUTPUT
// dump that Error() produces.
func conciseError(err error) string {
var ee ExitError
if errors.As(err, &ee) {
return fmt.Sprintf("exit status %d", ee.ExitStatus())
}
return err.Error()
}
// sleepCtx sleeps for d, returning early if the runner's context is canceled,
// so an interrupt (Ctrl+C) aborts the retry loop promptly rather than blocking
// until the full backoff elapses. It returns true if the full duration elapsed,
// or false if it was interrupted by context cancellation. Falls back to
// time.Sleep (returning true) for runners without a context (e.g. the test
// mockRunner).
func (helm *execer) sleepCtx(d time.Duration) bool {
ctx := helm.runnerContext()
if ctx == nil {
time.Sleep(d)
return true
}
timer := time.NewTimer(d)
defer timer.Stop()
select {
case <-timer.C:
return true
case <-ctx.Done():
return false
}
}
// runnerContext returns the ShellRunner's context if the runner is a
// ShellRunner (pointer or value), else nil.
func (helm *execer) runnerContext() context.Context {
switch r := helm.runner.(type) {
case *ShellRunner:
return r.Ctx
case ShellRunner:
return r.Ctx
}
return nil
}
func (helm *execer) AddRepo(name, repository, cafile, certfile, keyfile, username, password string, managed string, passCredentials, skipTLSVerify bool) error {
if name == "" && repository != "" {
helm.logger.Infof("empty field name\n")
return fmt.Errorf("empty field name")
@@ -288,46 +373,52 @@ func (helm *execer) AddRepo(name, repository, cafile, certfile, keyfile, usernam
helm.extra = savedExtra
}()
var out []byte
var err error
switch managed {
case "acr":
helm.logger.Infof("Adding repo %v (acr)", name)
out, err = helm.azcli(name)
out, err = helm.retryRepoOp(fmt.Sprintf("add %s (acr)", name), func() ([]byte, error) {
return helm.azcli(name)
})
case "":
args = append(args, "repo", "add", name, repository)
// --force-update is needed for both Helm 3.3.2+ and Helm 4
// to ensure repository indexes are updated when a repository already exists
// See https://github.com/helm/helm/pull/8777
if !helm.options.DisableForceUpdate && (helm.IsHelm4() || helm.IsVersionAtLeast("3.3.2")) {
args = append(args, "--force-update")
}
if certfile != "" && keyfile != "" {
args = append(args, "--cert-file", certfile, "--key-file", keyfile)
}
if cafile != "" {
args = append(args, "--ca-file", cafile)
}
if passCredentials {
args = append(args, "--pass-credentials")
}
if skipTLSVerify {
args = append(args, "--insecure-skip-tls-verify")
}
helm.logger.Infof("Adding repo %v %v", name, repository)
if username != "" && password != "" {
args = append(args, "--username", username, "--password-stdin")
buffer := bytes.Buffer{}
fmt.Fprintf(&buffer, "%s\n", password)
out, err = helm.execStdIn(args, map[string]string{}, &buffer)
} else {
out, err = helm.exec(args, map[string]string{}, nil)
}
out, err = helm.retryRepoOp(fmt.Sprintf("add %s", name), func() ([]byte, error) {
// args is local to each attempt, so the username/password append
// below can't accumulate across retries.
args := []string{"repo", "add", name, repository}
// --force-update is needed for both Helm 3.3.2+ and Helm 4
// to ensure repository indexes are updated when a repository already exists
// See https://github.com/helm/helm/pull/8777
if !helm.options.DisableForceUpdate && (helm.IsHelm4() || helm.IsVersionAtLeast("3.3.2")) {
args = append(args, "--force-update")
}
if certfile != "" && keyfile != "" {
args = append(args, "--cert-file", certfile, "--key-file", keyfile)
}
if cafile != "" {
args = append(args, "--ca-file", cafile)
}
if passCredentials {
args = append(args, "--pass-credentials")
}
if skipTLSVerify {
args = append(args, "--insecure-skip-tls-verify")
}
if username != "" && password != "" {
args = append(args, "--username", username, "--password-stdin")
buffer := bytes.Buffer{}
fmt.Fprintf(&buffer, "%s\n", password)
return helm.execStdIn(args, map[string]string{}, &buffer)
}
return helm.exec(args, map[string]string{}, nil)
})
default:
helm.logger.Errorf("ERROR: unknown type '%v' for repository %v", managed, name)
out = nil
err = nil
err = fmt.Errorf("unknown managed type %q for repository %v", managed, name)
}
helm.info(out)
@@ -341,7 +432,9 @@ func (helm *execer) UpdateRepo() error {
defer func() {
helm.extra = savedExtra
}()
out, err := helm.exec([]string{"repo", "update"}, map[string]string{}, nil)
out, err := helm.retryRepoOp("update", func() ([]byte, error) {
return helm.exec([]string{"repo", "update"}, map[string]string{}, nil)
})
helm.info(out)
return err
}
@@ -373,11 +466,15 @@ func (helm *execer) RegistryLogin(repository, username, password, caFile, certFi
}
args = append(args, "--username", username, "--password-stdin")
buffer := bytes.Buffer{}
fmt.Fprintf(&buffer, "%s\n", password)
helm.logger.Info("Logging in to registry")
out, err := helm.execStdIn(args, map[string]string{"HELM_EXPERIMENTAL_OCI": "1"}, &buffer)
out, err := helm.retryRepoOp(fmt.Sprintf("registry login %s", repository), func() ([]byte, error) {
buffer := bytes.Buffer{}
fmt.Fprintf(&buffer, "%s\n", password)
// Copy args so execStdIn's internal append (for helm.extra) can't alias
// the shared slice across retries.
return helm.execStdIn(append([]string{}, args...), map[string]string{"HELM_EXPERIMENTAL_OCI": "1"}, &buffer)
})
helm.info(out)
return err
}
+226 -2
View File
@@ -2,6 +2,7 @@ package helmexec
import (
"bytes"
"context"
"fmt"
"io"
"os"
@@ -10,6 +11,7 @@ import (
"regexp"
"strings"
"testing"
"time"
"github.com/Masterminds/semver/v3"
"github.com/google/go-cmp/cmp"
@@ -24,9 +26,31 @@ type mockRunner struct {
output []byte
versionOutput []byte // if set, returned for "helm version --short" probe; overrides default Helm 4 fallback
err error
// failCount is the number of times Execute/ExecuteStdIn fail (with
// errTransientNet) before succeeding. Used to exercise retry logic.
failCount int
calls int
// stdins records the stdin content passed to each ExecuteStdIn call,
// so tests can assert that retries still pass the full payload (e.g. the
// password is not consumed by a prior attempt).
stdins []string
// execArgs records the args passed to each Execute/ExecuteStdIn call, so
// tests can assert that retries don't accumulate duplicated flags.
execArgs [][]string
}
var errTransientNet = fmt.Errorf("transient network error")
func (mock *mockRunner) ExecuteStdIn(cmd string, args []string, env map[string]string, stdin io.Reader) ([]byte, error) {
mock.calls++
mock.execArgs = append(mock.execArgs, append([]string{}, args...))
if stdin != nil {
b, _ := io.ReadAll(stdin)
mock.stdins = append(mock.stdins, string(b))
}
if mock.failCount > 0 && mock.calls <= mock.failCount {
return nil, errTransientNet
}
return mock.output, mock.err
}
@@ -39,6 +63,11 @@ func (mock *mockRunner) Execute(cmd string, args []string, env map[string]string
return []byte("v4.0.1+g12500dd"), nil
}
}
mock.calls++
mock.execArgs = append(mock.execArgs, append([]string{}, args...))
if mock.failCount > 0 && mock.calls <= mock.failCount {
return nil, errTransientNet
}
return mock.output, mock.err
}
@@ -260,8 +289,8 @@ exec: az acr helm repo add --name acrRepo:
err = helm.AddRepo("otherRepo", "", "", "", "", "", "", "unknown", false, false)
expected = `ERROR: unknown type 'unknown' for repository otherRepo
`
if err != nil {
t.Errorf("unexpected error: %v", err)
if err == nil {
t.Errorf("expected error for unknown managed type, got nil")
}
if buffer.String() != expected {
t.Errorf("helmexec.AddRepo()\nactual = %v\nexpect = %v", buffer.String(), expected)
@@ -388,6 +417,201 @@ exec: helm --kubeconfig config --kube-context dev registry login repo.example.co
}
}
// newRetryExecer builds an execer with the given RepoRetry option and a
// mockRunner that fails failCount times before succeeding.
func newRetryExecer(t *testing.T, repoRetry, failCount int) (*execer, *mockRunner) {
t.Helper()
logger := NewLogger(os.Stdout, "debug")
runner := &mockRunner{failCount: failCount}
helm, err := New("helm", HelmExecOptions{RepoRetry: repoRetry}, logger, "config", "dev", runner)
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
return helm, runner
}
// withTestBackoff shrinks repoRetryBaseBackoff for the test's duration so retry
// tests don't sleep for real. The restore is via t.Cleanup so it always runs.
func withTestBackoff(t *testing.T) {
t.Helper()
original := repoRetryBaseBackoff
repoRetryBaseBackoff = time.Millisecond
t.Cleanup(func() { repoRetryBaseBackoff = original })
}
// retryTestCase is the shared shape for the AddRepo/UpdateRepo/RegistryLogin
// retry table tests.
type retryTestCase struct {
name string
repoRetry int
failCount int
wantErr bool
wantCalls int // expected number of exec (non-version) calls
managed string
}
func runRetryCases(t *testing.T, run func(*testing.T, *execer, *mockRunner, retryTestCase), cases []retryTestCase) {
t.Helper()
withTestBackoff(t)
for _, tt := range cases {
t.Run(tt.name, func(t *testing.T) {
helm, runner := newRetryExecer(t, tt.repoRetry, tt.failCount)
run(t, helm, runner, tt)
if runner.calls != tt.wantCalls {
t.Errorf("runner.calls = %d, want %d", runner.calls, tt.wantCalls)
}
})
}
}
func Test_AddRepo_Retry(t *testing.T) {
runRetryCases(t, func(t *testing.T, helm *execer, runner *mockRunner, tt retryTestCase) {
err := helm.AddRepo("myRepo", "https://repo.example.com/", "", "", "", "user", "pass", tt.managed, false, false)
if (err != nil) != tt.wantErr {
t.Errorf("AddRepo() error = %v, wantErr %v", err, tt.wantErr)
}
// For the normal (non-acr) path, each attempt's args must contain
// --username exactly once: guards against the username/password flags
// accumulating across retries.
if tt.managed == "" {
for i, a := range runner.execArgs {
if c := strings.Count(strings.Join(a, " "), "--username"); c != 1 {
t.Errorf("AddRepo() attempt %d: --username count = %d, want 1 (args=%v)", i+1, c, a)
}
}
}
}, []retryTestCase{
{name: "succeeds after one retry", repoRetry: 2, failCount: 1, wantErr: false, wantCalls: 2},
{name: "retries exhausted", repoRetry: 1, failCount: 2, wantErr: true, wantCalls: 2},
{name: "retry disabled runs once", repoRetry: 0, failCount: 1, wantErr: true, wantCalls: 1},
{name: "acr managed retries", repoRetry: 3, failCount: 2, wantErr: false, wantCalls: 3, managed: "acr"},
})
}
func Test_UpdateRepo_Retry(t *testing.T) {
runRetryCases(t, func(t *testing.T, helm *execer, _ *mockRunner, tt retryTestCase) {
err := helm.UpdateRepo()
if (err != nil) != tt.wantErr {
t.Errorf("UpdateRepo() error = %v, wantErr %v", err, tt.wantErr)
}
}, []retryTestCase{
{name: "succeeds after one retry", repoRetry: 2, failCount: 1, wantErr: false, wantCalls: 2},
{name: "retries exhausted", repoRetry: 1, failCount: 2, wantErr: true, wantCalls: 2},
{name: "retry disabled runs once", repoRetry: 0, failCount: 1, wantErr: true, wantCalls: 1},
})
}
func Test_RegistryLogin_Retry(t *testing.T) {
runRetryCases(t, func(t *testing.T, helm *execer, runner *mockRunner, tt retryTestCase) {
err := helm.RegistryLogin("repo.example.com", "user", "pass", "", "", "", false)
if (err != nil) != tt.wantErr {
t.Errorf("RegistryLogin() error = %v, wantErr %v", err, tt.wantErr)
}
// Every attempt must carry the full password (regression guard:
// the password buffer must not be consumed by a prior attempt).
for i, s := range runner.stdins {
if s != "pass\n" {
t.Errorf("RegistryLogin() attempt %d stdin = %q, want %q", i+1, s, "pass\n")
}
}
}, []retryTestCase{
{name: "succeeds after one retry", repoRetry: 2, failCount: 1, wantErr: false, wantCalls: 2},
{name: "retries exhausted", repoRetry: 1, failCount: 2, wantErr: true, wantCalls: 2},
{name: "retry disabled runs once", repoRetry: 0, failCount: 1, wantErr: true, wantCalls: 1},
})
}
// Test_Retry_LargeCount_NoOverflow verifies that a large --repo-retries value
// doesn't overflow the backoff duration (the shift exponent must be capped).
func Test_Retry_LargeCount_NoOverflow(t *testing.T) {
withTestBackoff(t)
// failCount > repoRetry so every attempt fails; attempt indices exceed the
// shift cap of 5, exercising the overflow guard without panicking.
const retries = 10
helm, runner := newRetryExecer(t, retries, retries+1)
err := helm.UpdateRepo()
if err == nil {
t.Fatal("UpdateRepo() expected error, got nil")
}
if runner.calls != retries+1 { // 1 initial + retries
t.Errorf("runner.calls = %d, want %d", runner.calls, retries+1)
}
}
// Test_SleepCtx_Canceled verifies sleepCtx returns promptly (and false) when
// the runner's context is already canceled, rather than blocking for the full
// duration.
func Test_SleepCtx_Canceled(t *testing.T) {
ctx, cancel := context.WithCancel(context.Background())
cancel() // cancel immediately
helm := &execer{
runner: &ShellRunner{Ctx: ctx},
logger: NewLogger(io.Discard, "debug"),
}
start := time.Now()
completed := helm.sleepCtx(30 * time.Second)
elapsed := time.Since(start)
if completed {
t.Error("sleepCtx() = true with a canceled context; want false")
}
if elapsed > time.Second {
t.Errorf("sleepCtx blocked for %v with a canceled context; expected prompt return", elapsed)
}
}
// Test_SleepCtx_NoContext verifies sleepCtx falls back to time.Sleep (returning
// true) when the runner has no context (e.g. mockRunner).
func Test_SleepCtx_NoContext(t *testing.T) {
helm := &execer{
runner: &mockRunner{},
logger: NewLogger(io.Discard, "debug"),
}
start := time.Now()
completed := helm.sleepCtx(5 * time.Millisecond)
elapsed := time.Since(start)
if !completed {
t.Error("sleepCtx() = false without a context; want true")
}
if elapsed < 4*time.Millisecond {
t.Errorf("sleepCtx returned in %v without a context; expected to sleep ~5ms", elapsed)
}
}
// Test_Retry_AbortsOnCanceledContext verifies that retryRepoOp stops retrying
// once the runner's context is canceled, rather than spinning into a tight
// loop of rapid helm invocations.
func Test_Retry_AbortsOnCanceledContext(t *testing.T) {
withTestBackoff(t)
ctx, cancel := context.WithCancel(context.Background())
defer cancel()
runner := &mockRunner{failCount: 100} // always fails
helm := &execer{
options: HelmExecOptions{RepoRetry: 100},
version: semver.MustParse("v4.0.1"),
runner: &ShellRunner{Ctx: ctx},
logger: NewLogger(io.Discard, "debug"),
}
// Cancel deterministically inside the op after the first attempt runs, so
// the retry loop must bail out at sleepCtx — no timing-based flakiness.
calls := 0
_, err := helm.retryRepoOp("test", func() ([]byte, error) {
calls++
out, e := runner.Execute("helm", []string{"repo", "update"}, nil, false)
if calls == 1 {
cancel()
}
return out, e
})
if err == nil {
t.Fatal("expected error, got nil")
}
// The context is canceled during the first attempt; sleepCtx must observe
// it and abort, so only the first attempt should run.
if calls > 2 {
t.Errorf("calls = %d after cancellation; expected <= 2 (no tight loop)", calls)
}
}
func Test_SyncRelease(t *testing.T) {
var buffer bytes.Buffer
logger := NewLogger(&buffer, "debug")