Phase 5 Batch 4: analyzer 规则引擎 + 3 个高层 Tool
- internal/analyzer: 纯函数规则引擎
- Finding / Severity / Thresholds / Input / StageMetric / ExecutorMetric
- 3 规则: data_skew (max/min ratio), gc_pressure (GC/CPU ratio),
bottleneck (shuffle read+write GB)
- 严重度阶梯: warning / critical (阈值 ×2)
- Analyze 入口数据驱动 for ... range
- 11 个单测 (3 规则 × normal/warning/critical + 零除防御)
- fetch_spark_metrics: SHS 业务封装, format=summary 触发 analyzer
- fetch_cluster_env: RM /cluster/info + /cluster/metrics 聚合
- analyze_spark_log: RM 日志降级链 + LLM-Ready Prompt 拼装
- prompt 结构: cluster 元信息 + log source + log tail +
heuristic findings + suggested LLM analysis
- findings 是 first-pass filter, LLM 做最终判断
- deps.go: +AnalyzerThresholds
- main.go: 注入 cfg 的 3 个阈值
- 端到端实测: analyze_spark_log 完整 prompt 渲染
完成 11 个 Tool 表面 (list_clusters / spark_submit / 4 RM /
fetch_spark_metrics / fetch_cluster_env / analyze_spark_log /
fetch_url / upload_file)
Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,149 @@
|
||||
package analyzer
|
||||
|
||||
import (
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestAnalyze(t *testing.T) {
|
||||
defaultThresholds := Thresholds{
|
||||
DataSkewRatio: 3.0,
|
||||
GCPressureRatio: 0.1,
|
||||
BottleneckShuffleGB: 50.0,
|
||||
}
|
||||
|
||||
t.Run("data_skew", func(t *testing.T) {
|
||||
t.Run("normal", func(t *testing.T) {
|
||||
in := Input{
|
||||
StageMetrics: map[string]StageMetric{
|
||||
"stage 0": {MaxPartitionBytes: 100, MinPartitionBytes: 50},
|
||||
},
|
||||
}
|
||||
if got := Analyze(in, defaultThresholds); len(got) != 0 {
|
||||
t.Fatalf("expected no findings, got %+v", got)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("critical", func(t *testing.T) {
|
||||
in := Input{
|
||||
StageMetrics: map[string]StageMetric{
|
||||
"stage 1": {MaxPartitionBytes: 1000, MinPartitionBytes: 10},
|
||||
},
|
||||
}
|
||||
got := Analyze(in, defaultThresholds)
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("expected 1 finding, got %+v", got)
|
||||
}
|
||||
if got[0].Rule != "data_skew" || got[0].Severity != SeverityCritical || got[0].Stage != "stage 1" {
|
||||
t.Fatalf("unexpected finding: %+v", got[0])
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("zero_min_partition_bytes", func(t *testing.T) {
|
||||
in := Input{
|
||||
StageMetrics: map[string]StageMetric{
|
||||
"stage 2": {MaxPartitionBytes: 1000, MinPartitionBytes: 0},
|
||||
},
|
||||
}
|
||||
if got := Analyze(in, defaultThresholds); len(got) != 0 {
|
||||
t.Fatalf("expected no findings for zero min, got %+v", got)
|
||||
}
|
||||
})
|
||||
})
|
||||
|
||||
t.Run("gc_pressure", func(t *testing.T) {
|
||||
t.Run("normal", func(t *testing.T) {
|
||||
in := Input{
|
||||
ExecutorMetrics: map[string]ExecutorMetric{
|
||||
"1": {GCTimeMS: 100, CPUTimeMS: 10000},
|
||||
},
|
||||
}
|
||||
if got := Analyze(in, defaultThresholds); len(got) != 0 {
|
||||
t.Fatalf("expected no findings, got %+v", got)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("warning", func(t *testing.T) {
|
||||
in := Input{
|
||||
ExecutorMetrics: map[string]ExecutorMetric{
|
||||
"1": {GCTimeMS: 2000, CPUTimeMS: 10000},
|
||||
},
|
||||
}
|
||||
got := Analyze(in, defaultThresholds)
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("expected 1 finding, got %+v", got)
|
||||
}
|
||||
if got[0].Rule != "gc_pressure" || got[0].Severity != SeverityWarning {
|
||||
t.Fatalf("unexpected finding: %+v", got[0])
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("critical", func(t *testing.T) {
|
||||
in := Input{
|
||||
ExecutorMetrics: map[string]ExecutorMetric{
|
||||
"1": {GCTimeMS: 5000, CPUTimeMS: 10000},
|
||||
},
|
||||
}
|
||||
got := Analyze(in, defaultThresholds)
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("expected 1 finding, got %+v", got)
|
||||
}
|
||||
if got[0].Rule != "gc_pressure" || got[0].Severity != SeverityCritical {
|
||||
t.Fatalf("unexpected finding: %+v", got[0])
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("zero_cpu_time", func(t *testing.T) {
|
||||
in := Input{
|
||||
ExecutorMetrics: map[string]ExecutorMetric{
|
||||
"1": {GCTimeMS: 5000, CPUTimeMS: 0},
|
||||
},
|
||||
}
|
||||
if got := Analyze(in, defaultThresholds); len(got) != 0 {
|
||||
t.Fatalf("expected no findings for zero cpu, got %+v", got)
|
||||
}
|
||||
})
|
||||
})
|
||||
|
||||
t.Run("bottleneck", func(t *testing.T) {
|
||||
t.Run("normal", func(t *testing.T) {
|
||||
in := Input{
|
||||
StageMetrics: map[string]StageMetric{
|
||||
"stage 0": {ShuffleReadBytes: 1 << 30, ShuffleWriteBytes: 0},
|
||||
},
|
||||
}
|
||||
if got := Analyze(in, defaultThresholds); len(got) != 0 {
|
||||
t.Fatalf("expected no findings, got %+v", got)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("warning", func(t *testing.T) {
|
||||
in := Input{
|
||||
StageMetrics: map[string]StageMetric{
|
||||
"stage 1": {ShuffleReadBytes: 120 << 30, ShuffleWriteBytes: 0},
|
||||
},
|
||||
}
|
||||
got := Analyze(in, defaultThresholds)
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("expected 1 finding, got %+v", got)
|
||||
}
|
||||
if got[0].Rule != "bottleneck" || got[0].Severity != SeverityWarning {
|
||||
t.Fatalf("unexpected finding: %+v", got[0])
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("critical", func(t *testing.T) {
|
||||
in := Input{
|
||||
StageMetrics: map[string]StageMetric{
|
||||
"stage 2": {ShuffleReadBytes: 300 << 30, ShuffleWriteBytes: 0},
|
||||
},
|
||||
}
|
||||
got := Analyze(in, defaultThresholds)
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("expected 1 finding, got %+v", got)
|
||||
}
|
||||
if got[0].Rule != "bottleneck" || got[0].Severity != SeverityCritical {
|
||||
t.Fatalf("unexpected finding: %+v", got[0])
|
||||
}
|
||||
})
|
||||
})
|
||||
}
|
||||
Reference in New Issue
Block a user