diff --git a/docs/configuration.md b/docs/configuration.md index 70a20d81..98164ae8 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -473,6 +473,12 @@ The following `helmDefaults` fields are also available but not shown in the exam `condition` controls whether a release is enabled. An empty condition enables the release. A direct `true` or `false` value is treated as a literal boolean and bypasses values lookup. Any other condition must be a values lookup path ending in `.enabled`, such as `vault.enabled`. +#### Continue On Error + +`continueOnError` defaults to `false`. When set to `true`, Helmfile keeps processing independent releases after this release fails, but any release that depends on it is skipped with an explicit error such as `release "backend" was skipped because dependency "database" failed`. + +This setting only changes Helmfile orchestration. It does not change Helm's per-release rollback behavior (`atomic`, `rollbackOnFailure`, `cleanupOnFail`, `wait`, `waitForJobs`, `trackMode`, hooks, or `reinstallIfForbidden`), and it does not turn a release failure into a successful command. `helmfile sync` and the sync phase of `helmfile apply` still exit with a non-zero status when any release fails. For `apply`, diff and chart-preparation failures remain fatal and are not converted into tolerated release errors. + `conditionTemplate` is evaluated before `condition` is checked. It must render to a boolean value; when both `condition` and `conditionTemplate` are set, the rendered `conditionTemplate` value replaces `condition`. Like other `*Template` fields, `conditionTemplate` is not evaluated by the `list` command. The following per-release fields are also available: diff --git a/docs/releases.md b/docs/releases.md index 614c7d3b..76c87eb6 100644 --- a/docs/releases.md +++ b/docs/releases.md @@ -46,6 +46,20 @@ On `helmfile [delete|destroy]`, deletions happen in the reverse order. That is, `myapp1` and `myapp2` are deleted first, then `servicemesh`, and finally `logging`. +### Failure handling and `continueOnError` + +By default, `helmfile [sync|apply]` stops on the first release failure (fail-fast). Set `continueOnError: true` on a release to keep processing independent releases after it fails: + +```yaml + - name: myapp1 + chart: charts/myapp + continueOnError: true + needs: + - servicemesh +``` + +When a release with `continueOnError` fails, releases in other branches of the DAG are still processed, while releases that (transitively) depend on the failed release are skipped with an error like `release "myapp2" was skipped because dependency "myapp1" failed`. The command still exits non-zero, and Helm's own rollback behavior (`atomic`, `rollbackOnFailure`, ...) is unaffected. See [Continue On Error](configuration.md#continue-on-error) for details. + ### Selectors and `needs` When using selectors/labels, `needs` are ignored by default. This behaviour can be overruled with a few parameters: diff --git a/pkg/app/app.go b/pkg/app/app.go index 07ced72c..cda12b71 100644 --- a/pkg/app/app.go +++ b/pkg/app/app.go @@ -3,6 +3,7 @@ package app import ( "bytes" goContext "context" + "errors" "fmt" "os" "path/filepath" @@ -1120,7 +1121,7 @@ func (a *App) processStateFileParallel(relPath string, defOpts LoadOpts, converg processed, errs := converge(templated) if len(errs) > 0 { - errChan <- errs[0] + errChan <- aggregateErrors(errs) return } if cleanErr != nil { @@ -1361,7 +1362,7 @@ func (a *App) visitStatesWithContext(fileOrDir string, defOpts LoadOpts, converg processed, errs = converge(templated) if len(errs) > 0 { - return errs[0] + return aggregateErrors(errs) } noMatchInHelmfiles = noMatchInHelmfiles && !processed @@ -1510,18 +1511,16 @@ func withBatches(purpose string, templated *state.HelmState, batches [][]state.R logger.Debugf("%s %d groups of releases in this order:\n%s", purpose, numBatches, printBatches(batches)) any := false + var allErrs []error + blockedByID := map[string]blockedRelease{} for i, batch := range batches { - var targets []state.ReleaseSpec - - for _, marked := range batch { - targets = append(targets, marked.ReleaseSpec) - } + targets, skippedErrs := filterBlockedReleases(batch, blockedByID, logger) + allErrs = append(allErrs, skippedErrs...) var releaseIds []string for _, r := range targets { - release := r - releaseIds = append(releaseIds, state.ReleaseToID(&release)) + releaseIds = append(releaseIds, state.ReleaseToID(&r)) } logger.Debugf("%s releases in group %d/%d: %s", purpose, i+1, numBatches, strings.Join(releaseIds, ", ")) @@ -1529,16 +1528,109 @@ func withBatches(purpose string, templated *state.HelmState, batches [][]state.R batchSt := *templated batchSt.Releases = targets - processed, errs := converge(&batchSt, helm) + processed, batchErrs := converge(&batchSt, helm) + allErrs = append(allErrs, batchErrs...) - if len(errs) > 0 { - return false, errs + if toleratesBatchErrors(batchErrs, blockedByID) { + any = any || processed + continue } - any = any || processed + return false, allErrs } - return any, nil + return any, allErrs +} + +// blockedRelease remembers a release that failed or was skipped, so that its +// dependents can be skipped transitively and reported with its plain name +// (rather than the kubeContext/namespace-qualified `needs` id). +type blockedRelease struct { + release *state.ReleaseSpec + err error +} + +// filterBlockedReleases partitions a batch into the releases that may still be +// processed and those that must be skipped because one of their `needs` +// dependencies already failed or was skipped. Skipped releases are recorded in +// blockedByID (so that their own dependents are skipped transitively) and +// reported as errors, so that a failed dependency never results in a +// successful command. +func filterBlockedReleases(batch []state.Release, blockedByID map[string]blockedRelease, logger *zap.SugaredLogger) ([]state.ReleaseSpec, []error) { + var targets []state.ReleaseSpec + var skippedErrs []error + + for _, marked := range batch { + release := marked.ReleaseSpec + + blocked, ok := firstBlockedDependency(&release, blockedByID) + if !ok { + targets = append(targets, release) + continue + } + + skippedErr := fmt.Errorf("release %q was skipped because dependency %q failed: %w", release.Name, blocked.release.Name, blocked.err) + skippedErrs = append(skippedErrs, skippedErr) + blockedByID[state.ReleaseToID(&release)] = blockedRelease{release: &release, err: skippedErr} + + logger.Warnf("release %q was skipped because dependency %q failed: %v", release.Name, blocked.release.Name, blocked.err) + } + + return targets, skippedErrs +} + +// toleratesBatchErrors inspects the errors reported while processing a batch. +// Each per-release failure is recorded in blockedByID so that its dependents +// are skipped in subsequent batches. It returns true when every error is a +// release failure whose release opted in to continueOnError, so that +// independent branches of the DAG keep being processed. Any other error (a +// non-release error, or a release failure without continueOnError) is fatal +// and preserves the historical fail-fast behavior. +func toleratesBatchErrors(batchErrs []error, blockedByID map[string]blockedRelease) bool { + for _, err := range batchErrs { + releaseErr := releaseErrorForContinueOnError(err) + if releaseErr == nil { + return false + } + + blockedByID[state.ReleaseToID(releaseErr.ReleaseSpec)] = blockedRelease{release: releaseErr.ReleaseSpec, err: err} + if !continueOnErrorEnabled(releaseErr.ReleaseSpec) { + return false + } + } + + return true +} + +func continueOnErrorEnabled(release *state.ReleaseSpec) bool { + return release != nil && release.ContinueOnError != nil && *release.ContinueOnError +} + +func firstBlockedDependency(release *state.ReleaseSpec, blockedByID map[string]blockedRelease) (blockedRelease, bool) { + for _, need := range release.Needs { + if blocked, ok := blockedByID[need]; ok { + return blocked, true + } + } + + return blockedRelease{}, false +} + +// releaseErrorForContinueOnError returns err as a *state.ReleaseError when it +// is a plain release failure. Non-release errors, and release errors carrying +// a non-failure code (e.g. the diff "changes detected" exit code 2), are not +// tolerable failures and must keep failing fast. +func releaseErrorForContinueOnError(err error) *state.ReleaseError { + var releaseErr *state.ReleaseError + if !errors.As(err, &releaseErr) { + return nil + } + + if releaseErr.Code != state.ReleaseErrorCodeFailure { + return nil + } + + return releaseErr } type Opts struct { @@ -2837,6 +2929,27 @@ func appError(msg string, err error) *Error { return &Error{msg: msg, Errors: []error{err}} } +// aggregateErrors combines multiple errors from processing a state file into a +// single *Error that renders all of them. Rendering a single error is +// identical to returning it directly, so aggregating does not change the +// reported message for the single-error case — but it keeps errors that are +// not the first one visible, which matters when continueOnError lets multiple +// releases fail or be skipped within one run. +func aggregateErrors(errs []error) error { + filtered := make([]error, 0, len(errs)) + for _, err := range errs { + if err != nil { + filtered = append(filtered, err) + } + } + + if len(filtered) == 0 { + return nil + } + + return &Error{Errors: filtered} +} + func (c context) clean(errs []error) error { if errs == nil { errs = []error{} diff --git a/pkg/app/continue_on_error_test.go b/pkg/app/continue_on_error_test.go new file mode 100644 index 00000000..1c41cea3 --- /dev/null +++ b/pkg/app/continue_on_error_test.go @@ -0,0 +1,289 @@ +package app + +import ( + "fmt" + "sync" + "testing" + + "github.com/helmfile/vals" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + "go.uber.org/zap" + + "github.com/helmfile/helmfile/pkg/exectest" + ffs "github.com/helmfile/helmfile/pkg/filesystem" + "github.com/helmfile/helmfile/pkg/helmexec" + "github.com/helmfile/helmfile/pkg/state" +) + +func fakeConverge(failures map[string]bool, executed *[]string) func(*state.HelmState, helmexec.Interface) (bool, []error) { + return func(st *state.HelmState, _ helmexec.Interface) (bool, []error) { + var errs []error + for i := range st.Releases { + release := &st.Releases[i] + *executed = append(*executed, release.Name) + if failures[release.Name] { + err := fmt.Errorf("failed processing release %s: boom", release.Name) + errs = append(errs, state.NewReleaseError(release, err, state.ReleaseErrorCodeFailure)) + } + } + + return len(st.Releases) > 0, errs + } +} + +func TestWithBatches_FailFastByDefault(t *testing.T) { + templated := &state.HelmState{} + var executed []string + + batches := [][]state.Release{ + { + state.Release{Name: "database", Needs: nil, ContinueOnError: nil}, + state.Release{Name: "cache", Needs: nil, ContinueOnError: nil}, + }, + { + state.Release{Name: "worker", Needs: nil, ContinueOnError: nil}, + }, + } + + any, errs := withBatches("processing", templated, batches, &exectest.Helm{}, zap.NewNop().Sugar(), fakeConverge(map[string]bool{"database": true}, &executed)) + + require.False(t, any) + require.Len(t, errs, 1) + require.Equal(t, []string{"database", "cache"}, executed) + require.Equal(t, "failed processing release database: boom", errs[0].Error()) +} + +func TestWithBatches_ContinueOnErrorKeepsIndependentBranches(t *testing.T) { + templated := &state.HelmState{} + var executed []string + + batches := [][]state.Release{ + { + state.Release{Name: "database", Needs: nil, ContinueOnError: new(true)}, + state.Release{Name: "cache", Needs: nil, ContinueOnError: nil}, + }, + { + state.Release{Name: "worker", Needs: nil, ContinueOnError: nil}, + }, + } + + any, errs := withBatches("processing", templated, batches, &exectest.Helm{}, zap.NewNop().Sugar(), fakeConverge(map[string]bool{"database": true}, &executed)) + + require.True(t, any) + require.Len(t, errs, 1) + require.Equal(t, []string{"database", "cache", "worker"}, executed) + require.Equal(t, "failed processing release database: boom", errs[0].Error()) +} + +func TestWithBatches_BlocksDependentReleasesAndAggregatesErrors(t *testing.T) { + templated := &state.HelmState{} + var executed []string + + batches := [][]state.Release{ + { + state.Release{Name: "database", Needs: nil, ContinueOnError: new(true)}, + }, + { + state.Release{Name: "backend", Needs: []string{"database"}, ContinueOnError: new(true)}, + state.Release{Name: "worker", Needs: nil, ContinueOnError: nil}, + }, + { + state.Release{Name: "frontend", Needs: []string{"backend"}, ContinueOnError: new(true)}, + }, + } + + any, errs := withBatches("processing", templated, batches, &exectest.Helm{}, zap.NewNop().Sugar(), fakeConverge(map[string]bool{"database": true}, &executed)) + + require.True(t, any) + require.Len(t, errs, 3) + require.Equal(t, []string{"database", "worker"}, executed) + require.Equal(t, "failed processing release database: boom", errs[0].Error()) + require.Contains(t, errs[1].Error(), "release \"backend\" was skipped because dependency \"database\" failed") + require.Contains(t, errs[2].Error(), "release \"frontend\" was skipped because dependency \"backend\" failed") +} + +func TestWithBatches_AggregatesMultipleReleaseErrors(t *testing.T) { + templated := &state.HelmState{} + var executed []string + + batches := [][]state.Release{ + { + state.Release{Name: "app-a", Needs: nil, ContinueOnError: new(true)}, + state.Release{Name: "app-b", Needs: nil, ContinueOnError: new(true)}, + }, + { + state.Release{Name: "worker", Needs: nil, ContinueOnError: nil}, + }, + } + + any, errs := withBatches("processing", templated, batches, &exectest.Helm{}, zap.NewNop().Sugar(), fakeConverge(map[string]bool{"app-a": true, "app-b": true}, &executed)) + + require.True(t, any) + require.Len(t, errs, 2) + require.Equal(t, []string{"app-a", "app-b", "worker"}, executed) + for _, releaseName := range []string{"app-a", "app-b"} { + require.Contains(t, fmt.Sprint(errs), releaseName) + } +} + +func TestWithBatches_TreatsNonFailureReleaseErrorCodeAsFatal(t *testing.T) { + templated := &state.HelmState{} + var executed []string + + // e.g. the helm-diff "changes detected" exit code 2, which is not a + // release failure and must never enable continuation. + converge := func(st *state.HelmState, _ helmexec.Interface) (bool, []error) { + var errs []error + for i := range st.Releases { + release := &st.Releases[i] + executed = append(executed, release.Name) + errs = append(errs, state.NewReleaseError(release, fmt.Errorf("release %s: changed", release.Name), 2)) + } + return false, errs + } + + batches := [][]state.Release{ + { + state.Release{Name: "database", Needs: nil, ContinueOnError: new(true)}, + }, + { + state.Release{Name: "worker", Needs: nil, ContinueOnError: nil}, + }, + } + + any, errs := withBatches("processing", templated, batches, &exectest.Helm{}, zap.NewNop().Sugar(), converge) + + require.False(t, any) + require.Len(t, errs, 1) + require.Equal(t, []string{"database"}, executed) +} + +// TestSync_ContinueOnError exercises the whole sync pipeline with the exectest +// fake helm: the release named "error-*" fails to upgrade, and continueOnError +// on that release must (1) let the independent release still be deployed, +// (2) skip the dependent release with an explicit error, and (3) keep the +// overall command failing with a non-zero exit code. +func TestSync_ContinueOnError(t *testing.T) { + newApp := func(t *testing.T, helm *exectest.Helm, files map[string]string, logger *zap.SugaredLogger) *App { + t.Helper() + + valsRuntime, err := vals.New(vals.Options{CacheSize: 32}) + require.NoError(t, err) + + return appWithFs(&App{ + OverrideHelmBinary: DefaultHelmBinary, + fs: ffs.DefaultFileSystem(), + OverrideKubeContext: "default", + DisableKubeVersionAutoDetection: true, + Env: "default", + Logger: logger, + helms: map[helmKey]helmexec.Interface{ + createHelmKey("helm", "default"): helm, + }, + valsRuntime: valsRuntime, + }, files) + } + + t.Run("independent release continues and dependent release is skipped", func(t *testing.T) { + var helm = &exectest.Helm{ + FailOnUnexpectedList: true, + DiffMutex: &sync.Mutex{}, + ChartsMutex: &sync.Mutex{}, + ReleasesMutex: &sync.Mutex{}, + } + + files := map[string]string{ + "/path/to/helmfile.yaml": ` +releases: +- name: error-database + chart: incubator/raw + namespace: default + continueOnError: true + +- name: independent-app + chart: incubator/raw + namespace: default + +- name: dependent-backend + chart: incubator/raw + namespace: default + needs: + - default/error-database +`, + } + + logger := zap.NewNop().Sugar() + + app := newApp(t, helm, files, logger) + syncErr := app.Sync(applyConfig{ + concurrency: 1, + logger: logger, + }) + + require.Error(t, syncErr, "sync must exit non-zero even when errors are tolerated") + + appErr, ok := syncErr.(*Error) + require.True(t, ok, "expected *app.Error, got %T", syncErr) + assert.Equal(t, 1, appErr.Code()) + + // independent-app must have been deployed despite error-database failing + synced := make([]string, 0, len(helm.Releases)) + for _, r := range helm.Releases { + synced = append(synced, r.Name) + } + assert.Contains(t, synced, "independent-app") + assert.NotContains(t, synced, "dependent-backend", "release depending on a failed release must be skipped") + assert.NotContains(t, synced, "error-database") + + assert.Contains(t, syncErr.Error(), `release "dependent-backend" was skipped because dependency "error-database" failed`) + }) + + t.Run("fail-fast remains the default", func(t *testing.T) { + var helm = &exectest.Helm{ + FailOnUnexpectedList: true, + DiffMutex: &sync.Mutex{}, + ChartsMutex: &sync.Mutex{}, + ReleasesMutex: &sync.Mutex{}, + } + + files := map[string]string{ + "/path/to/helmfile.yaml": ` +releases: +- name: error-database + chart: incubator/raw + namespace: default + +- name: base + chart: incubator/raw + namespace: default + +- name: later-app + chart: incubator/raw + namespace: default + needs: + - default/base +`, + } + + logger := zap.NewNop().Sugar() + + app := newApp(t, helm, files, logger) + syncErr := app.Sync(applyConfig{ + concurrency: 1, + logger: logger, + }) + + require.Error(t, syncErr) + + synced := make([]string, 0, len(helm.Releases)) + for _, r := range helm.Releases { + synced = append(synced, r.Name) + } + // base shares the first batch with error-database and is therefore + // still attempted, but the second batch must never run once the + // failure is reported. + assert.Contains(t, synced, "base") + assert.NotContains(t, synced, "later-app", "without continueOnError the default fail-fast behavior must stop the run") + }) +} diff --git a/pkg/state/continue_on_error_yaml_test.go b/pkg/state/continue_on_error_yaml_test.go new file mode 100644 index 00000000..57824b85 --- /dev/null +++ b/pkg/state/continue_on_error_yaml_test.go @@ -0,0 +1,30 @@ +package state + +import ( + "testing" + + "github.com/stretchr/testify/require" + + "github.com/helmfile/helmfile/pkg/yaml" +) + +func TestReleaseSpec_ContinueOnErrorYAML(t *testing.T) { + var st HelmState + require.NoError(t, yaml.Unmarshal([]byte(` +releases: +- name: absent + chart: foo +- name: explicit-false + chart: foo + continueOnError: false +- name: explicit-true + chart: foo + continueOnError: true +`), &st)) + require.Len(t, st.Releases, 3) + require.Nil(t, st.Releases[0].ContinueOnError) + require.NotNil(t, st.Releases[1].ContinueOnError) + require.False(t, *st.Releases[1].ContinueOnError) + require.NotNil(t, st.Releases[2].ContinueOnError) + require.True(t, *st.Releases[2].ContinueOnError) +} diff --git a/pkg/state/state.go b/pkg/state/state.go index 512855c3..911aecac 100644 --- a/pkg/state/state.go +++ b/pkg/state/state.go @@ -427,6 +427,9 @@ type ReleaseSpec struct { // Needs is the [KUBECONTEXT/][NS/]NAME representations of releases that this release depends on. Needs []string `yaml:"needs,omitempty"` + // ContinueOnError, when set to true, allows helmfile to keep processing independent releases after this release fails. + // The default behavior remains fail-fast when the field is absent or false. + ContinueOnError *bool `yaml:"continueOnError,omitempty"` // Hooks is a list of extension points paired with operations, that are executed in specific points of the lifecycle of releases defined in helmfile Hooks []event.Hook `yaml:"hooks,omitempty"` diff --git a/pkg/state/state_test.go b/pkg/state/state_test.go index ea416ad9..30349782 100644 --- a/pkg/state/state_test.go +++ b/pkg/state/state_test.go @@ -499,11 +499,12 @@ func TestHelmState_flagsForUpgrade(t *testing.T) { Wait: false, }, release: &ReleaseSpec{ - Chart: "test/chart", - Version: "0.1", - Wait: &enable, - Name: "test-charts", - Namespace: "test-namespace", + Chart: "test/chart", + Version: "0.1", + Wait: &enable, + ContinueOnError: &enable, + Name: "test-charts", + Namespace: "test-namespace", }, want: []string{ "--version", "0.1", @@ -517,11 +518,12 @@ func TestHelmState_flagsForUpgrade(t *testing.T) { WaitForJobs: false, }, release: &ReleaseSpec{ - Chart: "test/chart", - Version: "0.1", - WaitForJobs: &enable, - Name: "test-charts", - Namespace: "test-namespace", + Chart: "test/chart", + Version: "0.1", + WaitForJobs: &enable, + ContinueOnError: &enable, + Name: "test-charts", + Namespace: "test-namespace", }, want: []string{ "--version", "0.1", @@ -535,11 +537,12 @@ func TestHelmState_flagsForUpgrade(t *testing.T) { Devel: true, }, release: &ReleaseSpec{ - Chart: "test/chart", - Version: "0.1", - Wait: &enable, - Name: "test-charts", - Namespace: "test-namespace", + Chart: "test/chart", + Version: "0.1", + Wait: &enable, + ContinueOnError: &enable, + Name: "test-charts", + Namespace: "test-namespace", }, want: []string{ "--version", "0.1", @@ -606,11 +609,12 @@ func TestHelmState_flagsForUpgrade(t *testing.T) { Timeout: 123, }, release: &ReleaseSpec{ - Chart: "test/chart", - Version: "0.1", - Timeout: nil, - Name: "test-charts", - Namespace: "test-namespace", + Chart: "test/chart", + Version: "0.1", + Timeout: nil, + ContinueOnError: &enable, + Name: "test-charts", + Namespace: "test-namespace", }, want: []string{ "--version", "0.1", @@ -647,11 +651,12 @@ func TestHelmState_flagsForUpgrade(t *testing.T) { }, version: semver.MustParse("3.10.0"), release: &ReleaseSpec{ - Chart: "test/chart", - Version: "0.1", - Atomic: &enable, - Name: "test-charts", - Namespace: "test-namespace", + Chart: "test/chart", + Version: "0.1", + Atomic: &enable, + ContinueOnError: &enable, + Name: "test-charts", + Namespace: "test-namespace", }, want: []string{ "--version", "0.1", @@ -667,11 +672,12 @@ func TestHelmState_flagsForUpgrade(t *testing.T) { }, version: semver.MustParse("4.0.0"), release: &ReleaseSpec{ - Chart: "test/chart", - Version: "0.1", - Atomic: &enable, - Name: "test-charts", - Namespace: "test-namespace", + Chart: "test/chart", + Version: "0.1", + Atomic: &enable, + ContinueOnError: &enable, + Name: "test-charts", + Namespace: "test-namespace", }, want: []string{ "--version", "0.1", @@ -746,6 +752,7 @@ func TestHelmState_flagsForUpgrade(t *testing.T) { Chart: "test/chart", Version: "0.1", RollbackOnFailure: &enable, + ContinueOnError: &enable, Name: "test-charts", Namespace: "test-namespace", }, @@ -763,10 +770,11 @@ func TestHelmState_flagsForUpgrade(t *testing.T) { }, version: semver.MustParse("4.0.0"), release: &ReleaseSpec{ - Chart: "test/chart", - Version: "0.1", - Name: "test-charts", - Namespace: "test-namespace", + Chart: "test/chart", + Version: "0.1", + ContinueOnError: &enable, + Name: "test-charts", + Namespace: "test-namespace", }, want: []string{ "--version", "0.1", @@ -803,6 +811,7 @@ func TestHelmState_flagsForUpgrade(t *testing.T) { Chart: "test/chart", Version: "0.1", RollbackOnFailure: &enable, + ContinueOnError: &enable, Name: "test-charts", Namespace: "test-namespace", }, @@ -845,11 +854,12 @@ func TestHelmState_flagsForUpgrade(t *testing.T) { CleanupOnFail: false, }, release: &ReleaseSpec{ - Chart: "test/chart", - Version: "0.1", - CleanupOnFail: &enable, - Name: "test-charts", - Namespace: "test-namespace", + Chart: "test/chart", + Version: "0.1", + CleanupOnFail: &enable, + ContinueOnError: &enable, + Name: "test-charts", + Namespace: "test-namespace", }, want: []string{ "--version", "0.1", diff --git a/pkg/state/temp_test.go b/pkg/state/temp_test.go index b4809e60..178d3221 100644 --- a/pkg/state/temp_test.go +++ b/pkg/state/temp_test.go @@ -42,39 +42,39 @@ func TestGenerateID(t *testing.T) { run(testcase{ subject: "baseline", release: ReleaseSpec{Name: "foo", Chart: "incubator/raw"}, - want: "foo-values-577699c466", + want: "foo-values-565b4d4448", }) run(testcase{ subject: "different bytes content", release: ReleaseSpec{Name: "foo", Chart: "incubator/raw"}, data: []byte(`{"k":"v"}`), - want: "foo-values-c7dc8bf7", + want: "foo-values-5bf759b9cf", }) run(testcase{ subject: "different map content", release: ReleaseSpec{Name: "foo", Chart: "incubator/raw"}, data: map[string]any{"k": "v"}, - want: "foo-values-84744c8675", + want: "foo-values-cbd47bb9c", }) run(testcase{ subject: "different chart", release: ReleaseSpec{Name: "foo", Chart: "stable/envoy"}, - want: "foo-values-76c9d7ccff", + want: "foo-values-84bcbdd8f9", }) run(testcase{ subject: "different name", release: ReleaseSpec{Name: "bar", Chart: "incubator/raw"}, - want: "bar-values-55d5975ccf", + want: "bar-values-757bdb6889", }) run(testcase{ subject: "specific ns", release: ReleaseSpec{Name: "foo", Chart: "incubator/raw", Namespace: "myns"}, - want: "myns-foo-values-7956fd86dc", + want: "myns-foo-values-f4bcf8c77", }) for id, n := range ids { diff --git a/skills/helmfile/SKILL.md b/skills/helmfile/SKILL.md index 60156318..93378a64 100644 --- a/skills/helmfile/SKILL.md +++ b/skills/helmfile/SKILL.md @@ -96,6 +96,7 @@ repositories: | `secrets` | list | | Encrypted values files (requires helm-secrets plugin) | | `installed` | bool | | Set false to uninstall on sync | | `condition` | string | | Direct `true`/`false` or values lookup key ending in `.enabled` for filtering releases | +| `continueOnError` | bool | false | Continue independent releases after this release fails; dependent releases are skipped and Helmfile still exits non-zero | | `wait` | bool | false | Wait for resources to be ready | | `waitForJobs` | bool | false | Wait until all Jobs have completed | | `timeout` | int | 300 | Operation timeout in seconds | @@ -120,6 +121,8 @@ repositories: | `disableAutoDetectedKubeVersionForDiff` | bool | false | Disable auto-detected kubeVersion for diff | | `takeOwnership` | bool | false | Take ownership of existing resources | +`continueOnError` is a release-level orchestration knob only. It lets `sync` and the sync phase of `apply` continue independent releases after one release fails, while still skipping releases that depend on the failure and still returning a non-zero exit code at the end. It does not change Helm's own rollback flags (`atomic`, `rollbackOnFailure`, `cleanupOnFail`) and it does not relax diff or chart-preparation failures. + ### Helm Defaults ```yaml helmDefaults: