Files
helmfile/pkg/agent/doctor/analyze_test.go
T
yxxhero 9b943adc9e feat: add helmfile doctor command for AI-assisted diff analysis (#2660)
* feat: add `helmfile doctor` command for AI-assisted diff analysis

`helmfile doctor` runs `helmfile diff` and asks an OpenAI-compatible LLM to
summarize the changes and flag risks (data loss, security exposure, breaking
changes, downtime, performance, best-practice issues).

Key design decisions:
- When no LLM is configured, doctor is equivalent to `helmfile diff` with
  one exception: --show-secrets is always forced off (secrets never reach
  stdout, even without an LLM).
- Secrets are ALWAYS redacted via two layers: (1) ShowSecrets() forced to
  false so helm-diff emits <REDACTED> placeholders; (2) a defense-in-depth
  text redactor strips residual secret-looking content (Secret YAML blocks,
  sensitive key/value lines, base64 blobs, JWT tokens) before LLM transmission.
- LLM configuration precedence: env (HELMFILE_LLM_*) < helmfile.yaml (llm:)
  < CLI flags (--llm-*).
- Supports any OpenAI-compatible backend (OpenAI, Azure, One-API, LiteLLM,
  Ollama, etc.) with automatic response_format fallback for backends that
  don't support JSON mode.
- Prompt injection defense: release names and environment values are
  JSON-encoded before insertion into the LLM prompt.
- Exit codes: 0 (success/low-risk), 2 (high-risk gate, bypass with --force),
  1 (other errors). Helm-diff's 'detected changes' exit-2 is swallowed.

New packages:
- pkg/agent/llm: OpenAI-compatible client with JSON response parsing, mock
  client for testing, prompt builder with injection defense.
- pkg/agent/doctor: secret redactor (state machine + regex), report renderer
  (markdown + JSON), config resolver (env < yaml < flag merge).

Testing: 70+ unit tests covering redaction patterns, prompt injection,
response_format fallback, JSON parsing, yaml roundtrip, concurrency safety,
panic recovery, and error propagation. go test -race passes.

Documentation: full doctor section in docs/cli.md, llm: block reference in
docs/configuration.md, updated skills/helmfile for AI agents.

Signed-off-by: yxxhero <aiopsclub@163.com>

* docs: fix doctor equivalence wording per PR review

Per review feedback (PR #2660): the docs claimed doctor is 'equivalent to
helmfile diff — same flags, same output, same exit codes' in the unconfigured
path, but this over-promises because:

  1. doctor --output is the report format (not helm-diff's output format)
  2. helm-diff's --output is exposed as --diff-output in doctor
  3. --show-secrets is silently ignored

Updated all three locations (cli.md, cmd/doctor.go Long + godoc, pkg/app/doctor.go
godoc) to say 'falls back to helmfile diff with --show-secrets forced off' and
explicitly note the --output / --diff-output flag difference.

Signed-off-by: yxxhero <aiopsclub@163.com>

---------

Signed-off-by: yxxhero <aiopsclub@163.com>
2026-06-22 16:52:35 +08:00

111 lines
3.1 KiB
Go

package doctor
import (
"context"
"errors"
"testing"
"time"
"github.com/helmfile/helmfile/pkg/agent/llm"
)
// errMockClient is a Client that always returns errBoom to exercise the
// degraded-output path.
type errMockClient struct{ err error }
func (m *errMockClient) Analyze(_ context.Context, _ string, _ llm.AnalyzeInput) (llm.Analysis, error) {
return llm.Analysis{}, m.err
}
func TestAnalyze_EmptyDiffReturnsEmptyResult(t *testing.T) {
r := Analyze(context.Background(), "", Options{Client: nil})
// Empty Result means RawDiff empty AND Analysis nil — doctor.go uses
// these field checks directly (rather than a Result.IsEmpty helper) to
// decide whether to print anything.
if r.RawDiff != "" || r.Analysis != nil {
t.Fatalf("expected empty result, got %+v", r)
}
}
func TestAnalyze_NilClientReturnsRawDiff(t *testing.T) {
r := Analyze(context.Background(), "diff body", Options{Client: nil})
if r.Analysis != nil {
t.Errorf("Analysis should be nil, got %+v", r.Analysis)
}
if r.RawDiff != "diff body" {
t.Errorf("RawDiff = %q, want %q", r.RawDiff, "diff body")
}
if r.LLMCallFailed {
t.Error("LLMCallFailed should be false when client is nil")
}
}
func TestAnalyze_Success(t *testing.T) {
want := llm.Analysis{Summary: "ok", Risks: []llm.Risk{{Level: llm.RiskLevelLow}}}
c := llm.NewMockClient(want)
r := Analyze(context.Background(), "diff", Options{
Client: c,
Model: "gpt-4o",
})
if r.Analysis == nil {
t.Fatal("Analysis is nil")
}
if r.Analysis.Summary != "ok" {
t.Errorf("Summary = %q, want ok", r.Analysis.Summary)
}
if r.LLMCallFailed {
t.Error("LLMCallFailed should be false on success")
}
if r.Model != "gpt-4o" {
t.Errorf("Model = %q, want gpt-4o", r.Model)
}
}
func TestAnalyze_LLMFailureSetsFlag(t *testing.T) {
boom := errors.New("upstream 500")
c := &errMockClient{err: boom}
r := Analyze(context.Background(), "diff", Options{Client: c, Model: "m"})
if !r.LLMCallFailed {
t.Fatal("LLMCallFailed should be true")
}
if r.LLMError != boom {
t.Errorf("LLMError = %v, want %v", r.LLMError, boom)
}
if r.RawDiff != "diff" {
t.Errorf("RawDiff should still be populated, got %q", r.RawDiff)
}
}
func TestResult_HasHighRisk(t *testing.T) {
tests := []struct {
name string
r Result
want bool
}{
{name: "nil analysis", r: Result{}, want: false},
{name: "no risks", r: Result{Analysis: &llm.Analysis{}}, want: false},
{name: "only low", r: Result{Analysis: &llm.Analysis{Risks: []llm.Risk{{Level: llm.RiskLevelLow}}}}, want: false},
{name: "has high", r: Result{Analysis: &llm.Analysis{Risks: []llm.Risk{{Level: llm.RiskLevelHigh}}}}, want: true},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
if got := tt.r.HasHighRisk(); got != tt.want {
t.Errorf("HasHighRisk() = %v, want %v", got, tt.want)
}
})
}
}
// TestRoundDuration guards against roundDuration misbehaving on common
// durations.
func TestRoundDuration(t *testing.T) {
if got := roundDuration(500 * time.Millisecond); got != "500ms" {
t.Errorf("got %q want 500ms", got)
}
if got := roundDuration(2200 * time.Millisecond); got != "2.2s" {
t.Errorf("got %q want 2.2s", got)
}
}