Phase 5 Batch 4: analyzer 规则引擎 + 3 个高层 Tool
- internal/analyzer: 纯函数规则引擎
- Finding / Severity / Thresholds / Input / StageMetric / ExecutorMetric
- 3 规则: data_skew (max/min ratio), gc_pressure (GC/CPU ratio),
bottleneck (shuffle read+write GB)
- 严重度阶梯: warning / critical (阈值 ×2)
- Analyze 入口数据驱动 for ... range
- 11 个单测 (3 规则 × normal/warning/critical + 零除防御)
- fetch_spark_metrics: SHS 业务封装, format=summary 触发 analyzer
- fetch_cluster_env: RM /cluster/info + /cluster/metrics 聚合
- analyze_spark_log: RM 日志降级链 + LLM-Ready Prompt 拼装
- prompt 结构: cluster 元信息 + log source + log tail +
heuristic findings + suggested LLM analysis
- findings 是 first-pass filter, LLM 做最终判断
- deps.go: +AnalyzerThresholds
- main.go: 注入 cfg 的 3 个阈值
- 端到端实测: analyze_spark_log 完整 prompt 渲染
完成 11 个 Tool 表面 (list_clusters / spark_submit / 4 RM /
fetch_spark_metrics / fetch_cluster_env / analyze_spark_log /
fetch_url / upload_file)
Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,116 @@
|
||||
package analyzer
|
||||
|
||||
import "fmt"
|
||||
|
||||
// Analyze runs all heuristic rules and returns the triggered findings.
|
||||
func Analyze(in Input, t Thresholds) []Finding {
|
||||
var out []Finding
|
||||
out = append(out, checkDataSkew(in, t)...)
|
||||
out = append(out, checkGCPressure(in, t)...)
|
||||
out = append(out, checkBottleneck(in, t)...)
|
||||
return out
|
||||
}
|
||||
|
||||
func checkDataSkew(in Input, t Thresholds) []Finding {
|
||||
var out []Finding
|
||||
for stage, m := range in.StageMetrics {
|
||||
if m.MinPartitionBytes <= 0 || m.MaxPartitionBytes <= 0 {
|
||||
continue
|
||||
}
|
||||
ratio := float64(m.MaxPartitionBytes) / float64(m.MinPartitionBytes)
|
||||
if ratio > t.DataSkewRatio {
|
||||
sev := SeverityWarning
|
||||
if ratio > t.DataSkewRatio*2 {
|
||||
sev = SeverityCritical
|
||||
}
|
||||
out = append(out, Finding{
|
||||
Rule: "data_skew",
|
||||
Severity: sev,
|
||||
Stage: stage,
|
||||
Evidence: map[string]any{
|
||||
"max_partition_bytes": m.MaxPartitionBytes,
|
||||
"min_partition_bytes": m.MinPartitionBytes,
|
||||
"ratio": ratio,
|
||||
"threshold": t.DataSkewRatio,
|
||||
},
|
||||
Message: fmt.Sprintf("stage %q 数据倾斜: max/min = %.1fx (阈值 %.1fx)", stage, ratio, t.DataSkewRatio),
|
||||
})
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func checkGCPressure(in Input, t Thresholds) []Finding {
|
||||
var out []Finding
|
||||
var totalGC, totalCPU int64
|
||||
var worstExec string
|
||||
var worstRatio float64
|
||||
for id, m := range in.ExecutorMetrics {
|
||||
if m.CPUTimeMS <= 0 {
|
||||
continue
|
||||
}
|
||||
r := float64(m.GCTimeMS) / float64(m.CPUTimeMS)
|
||||
totalGC += m.GCTimeMS
|
||||
totalCPU += m.CPUTimeMS
|
||||
if r > worstRatio {
|
||||
worstRatio = r
|
||||
worstExec = id
|
||||
}
|
||||
}
|
||||
if totalCPU > 0 {
|
||||
aggRatio := float64(totalGC) / float64(totalCPU)
|
||||
if aggRatio > t.GCPressureRatio {
|
||||
sev := SeverityWarning
|
||||
if aggRatio > t.GCPressureRatio*2 {
|
||||
sev = SeverityCritical
|
||||
}
|
||||
out = append(out, Finding{
|
||||
Rule: "gc_pressure",
|
||||
Severity: sev,
|
||||
Evidence: map[string]any{
|
||||
"aggregate_gc_ratio": aggRatio,
|
||||
"worst_executor": worstExec,
|
||||
"worst_ratio": worstRatio,
|
||||
"threshold": t.GCPressureRatio,
|
||||
},
|
||||
Message: fmt.Sprintf("GC 压力过高: 聚合 GC/CPU 比率 = %.1f%% (阈值 %.1f%%, 最差 executor %s = %.1f%%)",
|
||||
aggRatio*100, t.GCPressureRatio*100, worstExec, worstRatio*100),
|
||||
})
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func checkBottleneck(in Input, t Thresholds) []Finding {
|
||||
var out []Finding
|
||||
thresholdBytes := int64(t.BottleneckShuffleGB * (1 << 30))
|
||||
for stage, m := range in.StageMetrics {
|
||||
total := m.ShuffleReadBytes + m.ShuffleWriteBytes
|
||||
if total < 0 {
|
||||
continue // overflow guard
|
||||
}
|
||||
if total > thresholdBytes {
|
||||
sev := SeverityInfo
|
||||
if total > thresholdBytes*2 {
|
||||
sev = SeverityWarning
|
||||
}
|
||||
if total > thresholdBytes*5 {
|
||||
sev = SeverityCritical
|
||||
}
|
||||
out = append(out, Finding{
|
||||
Rule: "bottleneck",
|
||||
Severity: sev,
|
||||
Stage: stage,
|
||||
Evidence: map[string]any{
|
||||
"shuffle_read_bytes": m.ShuffleReadBytes,
|
||||
"shuffle_write_bytes": m.ShuffleWriteBytes,
|
||||
"shuffle_total_gb": float64(total) / (1 << 30),
|
||||
"threshold_gb": t.BottleneckShuffleGB,
|
||||
},
|
||||
Message: fmt.Sprintf("stage %q 是潜在瓶颈: shuffle 总 %.1f GB (阈值 %.1f GB)",
|
||||
stage, float64(total)/(1<<30), t.BottleneckShuffleGB),
|
||||
})
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
@@ -0,0 +1,149 @@
|
||||
package analyzer
|
||||
|
||||
import (
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestAnalyze(t *testing.T) {
|
||||
defaultThresholds := Thresholds{
|
||||
DataSkewRatio: 3.0,
|
||||
GCPressureRatio: 0.1,
|
||||
BottleneckShuffleGB: 50.0,
|
||||
}
|
||||
|
||||
t.Run("data_skew", func(t *testing.T) {
|
||||
t.Run("normal", func(t *testing.T) {
|
||||
in := Input{
|
||||
StageMetrics: map[string]StageMetric{
|
||||
"stage 0": {MaxPartitionBytes: 100, MinPartitionBytes: 50},
|
||||
},
|
||||
}
|
||||
if got := Analyze(in, defaultThresholds); len(got) != 0 {
|
||||
t.Fatalf("expected no findings, got %+v", got)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("critical", func(t *testing.T) {
|
||||
in := Input{
|
||||
StageMetrics: map[string]StageMetric{
|
||||
"stage 1": {MaxPartitionBytes: 1000, MinPartitionBytes: 10},
|
||||
},
|
||||
}
|
||||
got := Analyze(in, defaultThresholds)
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("expected 1 finding, got %+v", got)
|
||||
}
|
||||
if got[0].Rule != "data_skew" || got[0].Severity != SeverityCritical || got[0].Stage != "stage 1" {
|
||||
t.Fatalf("unexpected finding: %+v", got[0])
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("zero_min_partition_bytes", func(t *testing.T) {
|
||||
in := Input{
|
||||
StageMetrics: map[string]StageMetric{
|
||||
"stage 2": {MaxPartitionBytes: 1000, MinPartitionBytes: 0},
|
||||
},
|
||||
}
|
||||
if got := Analyze(in, defaultThresholds); len(got) != 0 {
|
||||
t.Fatalf("expected no findings for zero min, got %+v", got)
|
||||
}
|
||||
})
|
||||
})
|
||||
|
||||
t.Run("gc_pressure", func(t *testing.T) {
|
||||
t.Run("normal", func(t *testing.T) {
|
||||
in := Input{
|
||||
ExecutorMetrics: map[string]ExecutorMetric{
|
||||
"1": {GCTimeMS: 100, CPUTimeMS: 10000},
|
||||
},
|
||||
}
|
||||
if got := Analyze(in, defaultThresholds); len(got) != 0 {
|
||||
t.Fatalf("expected no findings, got %+v", got)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("warning", func(t *testing.T) {
|
||||
in := Input{
|
||||
ExecutorMetrics: map[string]ExecutorMetric{
|
||||
"1": {GCTimeMS: 2000, CPUTimeMS: 10000},
|
||||
},
|
||||
}
|
||||
got := Analyze(in, defaultThresholds)
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("expected 1 finding, got %+v", got)
|
||||
}
|
||||
if got[0].Rule != "gc_pressure" || got[0].Severity != SeverityWarning {
|
||||
t.Fatalf("unexpected finding: %+v", got[0])
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("critical", func(t *testing.T) {
|
||||
in := Input{
|
||||
ExecutorMetrics: map[string]ExecutorMetric{
|
||||
"1": {GCTimeMS: 5000, CPUTimeMS: 10000},
|
||||
},
|
||||
}
|
||||
got := Analyze(in, defaultThresholds)
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("expected 1 finding, got %+v", got)
|
||||
}
|
||||
if got[0].Rule != "gc_pressure" || got[0].Severity != SeverityCritical {
|
||||
t.Fatalf("unexpected finding: %+v", got[0])
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("zero_cpu_time", func(t *testing.T) {
|
||||
in := Input{
|
||||
ExecutorMetrics: map[string]ExecutorMetric{
|
||||
"1": {GCTimeMS: 5000, CPUTimeMS: 0},
|
||||
},
|
||||
}
|
||||
if got := Analyze(in, defaultThresholds); len(got) != 0 {
|
||||
t.Fatalf("expected no findings for zero cpu, got %+v", got)
|
||||
}
|
||||
})
|
||||
})
|
||||
|
||||
t.Run("bottleneck", func(t *testing.T) {
|
||||
t.Run("normal", func(t *testing.T) {
|
||||
in := Input{
|
||||
StageMetrics: map[string]StageMetric{
|
||||
"stage 0": {ShuffleReadBytes: 1 << 30, ShuffleWriteBytes: 0},
|
||||
},
|
||||
}
|
||||
if got := Analyze(in, defaultThresholds); len(got) != 0 {
|
||||
t.Fatalf("expected no findings, got %+v", got)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("warning", func(t *testing.T) {
|
||||
in := Input{
|
||||
StageMetrics: map[string]StageMetric{
|
||||
"stage 1": {ShuffleReadBytes: 120 << 30, ShuffleWriteBytes: 0},
|
||||
},
|
||||
}
|
||||
got := Analyze(in, defaultThresholds)
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("expected 1 finding, got %+v", got)
|
||||
}
|
||||
if got[0].Rule != "bottleneck" || got[0].Severity != SeverityWarning {
|
||||
t.Fatalf("unexpected finding: %+v", got[0])
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("critical", func(t *testing.T) {
|
||||
in := Input{
|
||||
StageMetrics: map[string]StageMetric{
|
||||
"stage 2": {ShuffleReadBytes: 300 << 30, ShuffleWriteBytes: 0},
|
||||
},
|
||||
}
|
||||
got := Analyze(in, defaultThresholds)
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("expected 1 finding, got %+v", got)
|
||||
}
|
||||
if got[0].Rule != "bottleneck" || got[0].Severity != SeverityCritical {
|
||||
t.Fatalf("unexpected finding: %+v", got[0])
|
||||
}
|
||||
})
|
||||
})
|
||||
}
|
||||
@@ -0,0 +1,47 @@
|
||||
package analyzer
|
||||
|
||||
// Severity is the heuristic finding severity.
|
||||
type Severity string
|
||||
|
||||
const (
|
||||
SeverityInfo Severity = "info"
|
||||
SeverityWarning Severity = "warning"
|
||||
SeverityCritical Severity = "critical"
|
||||
)
|
||||
|
||||
// Finding is a single triggered heuristic rule.
|
||||
type Finding struct {
|
||||
Rule string `json:"rule"`
|
||||
Severity Severity `json:"severity"`
|
||||
Stage string `json:"stage,omitempty"`
|
||||
Evidence map[string]any `json:"evidence"`
|
||||
Message string `json:"message"`
|
||||
}
|
||||
|
||||
// Thresholds holds analyzer thresholds injected at runtime.
|
||||
type Thresholds struct {
|
||||
DataSkewRatio float64
|
||||
GCPressureRatio float64
|
||||
BottleneckShuffleGB float64
|
||||
}
|
||||
|
||||
// Input is the data passed to the heuristic rules.
|
||||
type Input struct {
|
||||
StageMetrics map[string]StageMetric `json:"stage_metrics"`
|
||||
ExecutorMetrics map[string]ExecutorMetric `json:"executor_metrics"`
|
||||
}
|
||||
|
||||
// StageMetric contains per-stage measurements.
|
||||
type StageMetric struct {
|
||||
DurationMS int64 `json:"duration_ms"`
|
||||
ShuffleReadBytes int64 `json:"shuffle_read_bytes"`
|
||||
ShuffleWriteBytes int64 `json:"shuffle_write_bytes"`
|
||||
MaxPartitionBytes int64 `json:"max_partition_bytes"`
|
||||
MinPartitionBytes int64 `json:"min_partition_bytes"`
|
||||
}
|
||||
|
||||
// ExecutorMetric contains per-executor measurements.
|
||||
type ExecutorMetric struct {
|
||||
GCTimeMS int64 `json:"gc_time_ms"`
|
||||
CPUTimeMS int64 `json:"cpu_time_ms"`
|
||||
}
|
||||
Reference in New Issue
Block a user