mirror of
https://github.com/Awuqing/BackupX.git
synced 2026-09-06 16:06:45 +08:00
功能: v2.1 可观测性与流控 (#47)
* 功能: v2.1 可观测性与流控 — Prometheus + 节点带宽 + 审计 Webhook 核心能力: - Prometheus /metrics 端点:11 类指标(任务/存储/节点/SLA/验证/恢复/复制) - 节点级带宽限速生效:model.Node.BandwidthLimit 覆盖全局默认 - 审计日志 Webhook 外输:HMAC-SHA256 签名,配合 SIEM 合规留档 实现: - server/internal/metrics/ 独立 Registry + 异步 Gauge Collector(30s) - backup/restore/verify/replication 服务注入 metrics 钩子,nil 安全 - resolveProviderForNode() 按 task.NodeID 解析 BandwidthLimit - AuditService.SetWebhook + 动态 settings 推送,无需重启 测试: - metrics/registry_test.go: 注册/采集/nil safety/HTTP handler - service/audit_service_webhook_test.go: 签名正确性/异步投递/禁用路径 - go test ./... 全部通过 * chore: 触发 CodeQL 扫描
This commit is contained in:
@@ -0,0 +1,152 @@
|
||||
package metrics
|
||||
|
||||
import (
|
||||
"context"
|
||||
"time"
|
||||
|
||||
"backupx/server/internal/model"
|
||||
"backupx/server/internal/repository"
|
||||
)
|
||||
|
||||
// SampleSource 抽象 Collector 需要的仓储访问,便于单测替换。
|
||||
type SampleSource interface {
|
||||
ListStorageTargets(ctx context.Context) ([]model.StorageTarget, error)
|
||||
StorageUsage(ctx context.Context) ([]repository.BackupStorageUsageItem, error)
|
||||
ListNodes(ctx context.Context) ([]model.Node, error)
|
||||
CountSLABreach(ctx context.Context) (int, error)
|
||||
}
|
||||
|
||||
// repoSource 把 repository 适配到 SampleSource。
|
||||
type repoSource struct {
|
||||
targets repository.StorageTargetRepository
|
||||
records repository.BackupRecordRepository
|
||||
nodes repository.NodeRepository
|
||||
tasks repository.BackupTaskRepository
|
||||
now func() time.Time
|
||||
}
|
||||
|
||||
// NewRepoSource 用仓储实例构造 SampleSource。
|
||||
func NewRepoSource(
|
||||
targets repository.StorageTargetRepository,
|
||||
records repository.BackupRecordRepository,
|
||||
nodes repository.NodeRepository,
|
||||
tasks repository.BackupTaskRepository,
|
||||
) SampleSource {
|
||||
return &repoSource{
|
||||
targets: targets,
|
||||
records: records,
|
||||
nodes: nodes,
|
||||
tasks: tasks,
|
||||
now: func() time.Time { return time.Now().UTC() },
|
||||
}
|
||||
}
|
||||
|
||||
func (s *repoSource) ListStorageTargets(ctx context.Context) ([]model.StorageTarget, error) {
|
||||
return s.targets.List(ctx)
|
||||
}
|
||||
|
||||
func (s *repoSource) StorageUsage(ctx context.Context) ([]repository.BackupStorageUsageItem, error) {
|
||||
return s.records.StorageUsage(ctx)
|
||||
}
|
||||
|
||||
func (s *repoSource) ListNodes(ctx context.Context) ([]model.Node, error) {
|
||||
return s.nodes.List(ctx)
|
||||
}
|
||||
|
||||
// CountSLABreach 统计当前违反 RPO 的任务:
|
||||
// - 任务启用且配置了 SLAHoursRPO > 0
|
||||
// - 最近一次成功备份距今超出 SLA 时间窗,或从未成功过
|
||||
func (s *repoSource) CountSLABreach(ctx context.Context) (int, error) {
|
||||
tasks, err := s.tasks.List(ctx, repository.BackupTaskListOptions{})
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
now := s.now()
|
||||
count := 0
|
||||
for i := range tasks {
|
||||
task := &tasks[i]
|
||||
if task.SLAHoursRPO <= 0 || !task.Enabled {
|
||||
continue
|
||||
}
|
||||
threshold := now.Add(-time.Duration(task.SLAHoursRPO) * time.Hour)
|
||||
if task.LastRunAt == nil || task.LastRunAt.Before(threshold) {
|
||||
count++
|
||||
}
|
||||
}
|
||||
return count, nil
|
||||
}
|
||||
|
||||
// Collector 周期性采集 gauge 类指标(存储用量、节点在线、SLA 违约)。
|
||||
// 用后台 goroutine 驱动,避免在 /metrics 请求路径做慢 IO。
|
||||
type Collector struct {
|
||||
metrics *Metrics
|
||||
source SampleSource
|
||||
interval time.Duration
|
||||
}
|
||||
|
||||
// NewCollector 创建周期采集器。interval=0 走默认 30s。
|
||||
func NewCollector(m *Metrics, source SampleSource, interval time.Duration) *Collector {
|
||||
if interval <= 0 {
|
||||
interval = 30 * time.Second
|
||||
}
|
||||
return &Collector{metrics: m, source: source, interval: interval}
|
||||
}
|
||||
|
||||
// Start 在后台运行采集循环;随 ctx 取消而终止。
|
||||
// 启动时立即采一次,之后按 interval 轮询。
|
||||
func (c *Collector) Start(ctx context.Context) {
|
||||
if c == nil || c.metrics == nil || c.source == nil {
|
||||
return
|
||||
}
|
||||
go func() {
|
||||
c.collect(ctx)
|
||||
ticker := time.NewTicker(c.interval)
|
||||
defer ticker.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-ticker.C:
|
||||
c.collect(ctx)
|
||||
}
|
||||
}
|
||||
}()
|
||||
}
|
||||
|
||||
// collect 执行一次采样;单轮失败不影响下次。
|
||||
func (c *Collector) collect(ctx context.Context) {
|
||||
// 存储用量:按 StorageTargetID 聚合 file_size,对应 target name/type
|
||||
if targets, err := c.source.ListStorageTargets(ctx); err == nil {
|
||||
nameByID := make(map[uint]string, len(targets))
|
||||
typeByID := make(map[uint]string, len(targets))
|
||||
for i := range targets {
|
||||
nameByID[targets[i].ID] = targets[i].Name
|
||||
typeByID[targets[i].ID] = targets[i].Type
|
||||
}
|
||||
if usage, uerr := c.source.StorageUsage(ctx); uerr == nil {
|
||||
c.metrics.ResetStorageUsed()
|
||||
for _, item := range usage {
|
||||
name := nameByID[item.StorageTargetID]
|
||||
if name == "" {
|
||||
continue
|
||||
}
|
||||
c.metrics.SetStorageUsed(name, typeByID[item.StorageTargetID], item.TotalSize)
|
||||
}
|
||||
}
|
||||
}
|
||||
// 节点在线状态:role 约定为 master / agent
|
||||
if nodes, err := c.source.ListNodes(ctx); err == nil {
|
||||
c.metrics.ResetNodeOnline()
|
||||
for i := range nodes {
|
||||
n := &nodes[i]
|
||||
role := "agent"
|
||||
if n.IsLocal {
|
||||
role = "master"
|
||||
}
|
||||
c.metrics.SetNodeOnline(n.Name, role, n.Status == model.NodeStatusOnline)
|
||||
}
|
||||
}
|
||||
if breach, err := c.source.CountSLABreach(ctx); err == nil {
|
||||
c.metrics.SetSLABreach(breach)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,225 @@
|
||||
// Package metrics 暴露 BackupX 的 Prometheus 采集器。
|
||||
//
|
||||
// 设计要点:
|
||||
// - 使用独立 Registry,避免与 default registry 中的 Go runtime metrics 混淆
|
||||
// - Counter/Gauge/Histogram 全部以 backupx_ 为前缀,遵循 Prometheus 命名规范
|
||||
// - 所有指标都支持零值:未注入时调用方法是 no-op,不会 panic
|
||||
// - 组件只依赖本包,不反向引用 service/repository,避免循环
|
||||
package metrics
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
|
||||
"github.com/prometheus/client_golang/prometheus"
|
||||
"github.com/prometheus/client_golang/prometheus/collectors"
|
||||
"github.com/prometheus/client_golang/prometheus/promhttp"
|
||||
)
|
||||
|
||||
// Metrics 聚合所有采集器,由 app 层组装一次并按需注入到 service。
|
||||
type Metrics struct {
|
||||
registry *prometheus.Registry
|
||||
|
||||
// 任务执行计数(labels: status, task_type)
|
||||
TaskRunTotal *prometheus.CounterVec
|
||||
// 任务耗时分布(labels: task_type)
|
||||
TaskRunDuration *prometheus.HistogramVec
|
||||
// 任务产出字节数(labels: task_type)
|
||||
TaskBytesTotal *prometheus.CounterVec
|
||||
// 正在运行的任务数
|
||||
TaskRunningGauge prometheus.Gauge
|
||||
// 存储目标用量(labels: target_name, target_type)
|
||||
StorageUsedBytes *prometheus.GaugeVec
|
||||
// 节点在线状态(labels: node_name, role;value: 0/1)
|
||||
NodeOnline *prometheus.GaugeVec
|
||||
// 验证演练结果(labels: status)
|
||||
VerifyRunTotal *prometheus.CounterVec
|
||||
// 恢复操作结果(labels: status)
|
||||
RestoreRunTotal *prometheus.CounterVec
|
||||
// 副本复制结果(labels: status)
|
||||
ReplicationRunTotal *prometheus.CounterVec
|
||||
// SLA 违约数(gauge)
|
||||
SLABreachGauge prometheus.Gauge
|
||||
// 应用信息(label: version)
|
||||
AppInfo *prometheus.GaugeVec
|
||||
}
|
||||
|
||||
// New 构造并注册所有采集器。
|
||||
// 失败时 panic:采集器注册失败属于启动期编程错误,没有合理 fallback。
|
||||
func New(version string) *Metrics {
|
||||
reg := prometheus.NewRegistry()
|
||||
// 注入标准 Go runtime + process 指标
|
||||
reg.MustRegister(collectors.NewGoCollector())
|
||||
reg.MustRegister(collectors.NewProcessCollector(collectors.ProcessCollectorOpts{}))
|
||||
|
||||
m := &Metrics{
|
||||
registry: reg,
|
||||
TaskRunTotal: prometheus.NewCounterVec(prometheus.CounterOpts{
|
||||
Name: "backupx_task_run_total",
|
||||
Help: "备份任务执行总数,按状态和任务类型细分",
|
||||
}, []string{"status", "task_type"}),
|
||||
TaskRunDuration: prometheus.NewHistogramVec(prometheus.HistogramOpts{
|
||||
Name: "backupx_task_run_duration_seconds",
|
||||
Help: "备份任务耗时分布",
|
||||
Buckets: []float64{1, 5, 15, 30, 60, 120, 300, 600, 1800, 3600, 7200},
|
||||
}, []string{"task_type"}),
|
||||
TaskBytesTotal: prometheus.NewCounterVec(prometheus.CounterOpts{
|
||||
Name: "backupx_task_bytes_total",
|
||||
Help: "备份任务累计产出字节数",
|
||||
}, []string{"task_type"}),
|
||||
TaskRunningGauge: prometheus.NewGauge(prometheus.GaugeOpts{
|
||||
Name: "backupx_task_running",
|
||||
Help: "当前正在执行的备份任务数",
|
||||
}),
|
||||
StorageUsedBytes: prometheus.NewGaugeVec(prometheus.GaugeOpts{
|
||||
Name: "backupx_storage_used_bytes",
|
||||
Help: "存储目标已用字节数",
|
||||
}, []string{"target_name", "target_type"}),
|
||||
NodeOnline: prometheus.NewGaugeVec(prometheus.GaugeOpts{
|
||||
Name: "backupx_node_online",
|
||||
Help: "集群节点在线状态(1 在线 / 0 离线)",
|
||||
}, []string{"node_name", "role"}),
|
||||
VerifyRunTotal: prometheus.NewCounterVec(prometheus.CounterOpts{
|
||||
Name: "backupx_verify_run_total",
|
||||
Help: "备份验证演练执行总数",
|
||||
}, []string{"status"}),
|
||||
RestoreRunTotal: prometheus.NewCounterVec(prometheus.CounterOpts{
|
||||
Name: "backupx_restore_run_total",
|
||||
Help: "恢复操作执行总数",
|
||||
}, []string{"status"}),
|
||||
ReplicationRunTotal: prometheus.NewCounterVec(prometheus.CounterOpts{
|
||||
Name: "backupx_replication_run_total",
|
||||
Help: "备份副本复制执行总数",
|
||||
}, []string{"status"}),
|
||||
SLABreachGauge: prometheus.NewGauge(prometheus.GaugeOpts{
|
||||
Name: "backupx_sla_breach_tasks",
|
||||
Help: "当前违反 SLA/RPO 的任务数",
|
||||
}),
|
||||
AppInfo: prometheus.NewGaugeVec(prometheus.GaugeOpts{
|
||||
Name: "backupx_app_info",
|
||||
Help: "BackupX 应用元信息(恒为 1,通过 label 暴露版本号)",
|
||||
}, []string{"version"}),
|
||||
}
|
||||
reg.MustRegister(
|
||||
m.TaskRunTotal,
|
||||
m.TaskRunDuration,
|
||||
m.TaskBytesTotal,
|
||||
m.TaskRunningGauge,
|
||||
m.StorageUsedBytes,
|
||||
m.NodeOnline,
|
||||
m.VerifyRunTotal,
|
||||
m.RestoreRunTotal,
|
||||
m.ReplicationRunTotal,
|
||||
m.SLABreachGauge,
|
||||
m.AppInfo,
|
||||
)
|
||||
m.AppInfo.WithLabelValues(version).Set(1)
|
||||
return m
|
||||
}
|
||||
|
||||
// Handler 返回 /metrics 的 HTTP handler。
|
||||
// 使用本包专属 registry,避免混入其他组件的默认 metrics。
|
||||
func (m *Metrics) Handler() http.Handler {
|
||||
if m == nil {
|
||||
return http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
|
||||
http.Error(w, "metrics disabled", http.StatusServiceUnavailable)
|
||||
})
|
||||
}
|
||||
return promhttp.HandlerFor(m.registry, promhttp.HandlerOpts{
|
||||
EnableOpenMetrics: false,
|
||||
})
|
||||
}
|
||||
|
||||
// ObserveTaskRun 记录一次任务执行结果。
|
||||
// status 常用值:success / failed / cancelled。nil 接收器安全。
|
||||
func (m *Metrics) ObserveTaskRun(taskType, status string, durationSec float64, bytes int64) {
|
||||
if m == nil {
|
||||
return
|
||||
}
|
||||
m.TaskRunTotal.WithLabelValues(status, taskType).Inc()
|
||||
m.TaskRunDuration.WithLabelValues(taskType).Observe(durationSec)
|
||||
if bytes > 0 {
|
||||
m.TaskBytesTotal.WithLabelValues(taskType).Add(float64(bytes))
|
||||
}
|
||||
}
|
||||
|
||||
// IncTaskRunning / DecTaskRunning 配套使用,反映并发中任务数。
|
||||
func (m *Metrics) IncTaskRunning() {
|
||||
if m == nil {
|
||||
return
|
||||
}
|
||||
m.TaskRunningGauge.Inc()
|
||||
}
|
||||
|
||||
func (m *Metrics) DecTaskRunning() {
|
||||
if m == nil {
|
||||
return
|
||||
}
|
||||
m.TaskRunningGauge.Dec()
|
||||
}
|
||||
|
||||
// ObserveRestore / ObserveVerify / ObserveReplication 记录子动作结果。
|
||||
// 所有方法对 nil 接收器安全:未注入 Metrics 时静默降级,不 panic。
|
||||
func (m *Metrics) ObserveRestore(status string) {
|
||||
if m == nil {
|
||||
return
|
||||
}
|
||||
m.RestoreRunTotal.WithLabelValues(status).Inc()
|
||||
}
|
||||
|
||||
func (m *Metrics) ObserveVerify(status string) {
|
||||
if m == nil {
|
||||
return
|
||||
}
|
||||
m.VerifyRunTotal.WithLabelValues(status).Inc()
|
||||
}
|
||||
|
||||
func (m *Metrics) ObserveReplication(status string) {
|
||||
if m == nil {
|
||||
return
|
||||
}
|
||||
m.ReplicationRunTotal.WithLabelValues(status).Inc()
|
||||
}
|
||||
|
||||
// SetStorageUsed 刷新某存储目标的用量。调用方负责周期采集。
|
||||
func (m *Metrics) SetStorageUsed(name, targetType string, bytes int64) {
|
||||
if m == nil {
|
||||
return
|
||||
}
|
||||
m.StorageUsedBytes.WithLabelValues(name, targetType).Set(float64(bytes))
|
||||
}
|
||||
|
||||
// SetNodeOnline 刷新节点在线状态。
|
||||
func (m *Metrics) SetNodeOnline(name, role string, online bool) {
|
||||
if m == nil {
|
||||
return
|
||||
}
|
||||
val := 0.0
|
||||
if online {
|
||||
val = 1
|
||||
}
|
||||
m.NodeOnline.WithLabelValues(name, role).Set(val)
|
||||
}
|
||||
|
||||
// ResetNodeOnline 清空节点 gauge(当节点被删除时避免残留指标)。
|
||||
func (m *Metrics) ResetNodeOnline() {
|
||||
if m == nil {
|
||||
return
|
||||
}
|
||||
m.NodeOnline.Reset()
|
||||
}
|
||||
|
||||
// ResetStorageUsed 清空存储目标 gauge。
|
||||
func (m *Metrics) ResetStorageUsed() {
|
||||
if m == nil {
|
||||
return
|
||||
}
|
||||
m.StorageUsedBytes.Reset()
|
||||
}
|
||||
|
||||
// SetSLABreach 刷新 SLA 违约任务数。
|
||||
func (m *Metrics) SetSLABreach(count int) {
|
||||
if m == nil {
|
||||
return
|
||||
}
|
||||
m.SLABreachGauge.Set(float64(count))
|
||||
}
|
||||
@@ -0,0 +1,76 @@
|
||||
package metrics
|
||||
|
||||
import (
|
||||
"io"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/prometheus/client_golang/prometheus/testutil"
|
||||
)
|
||||
|
||||
func TestNew_AppInfoVersionLabel(t *testing.T) {
|
||||
m := New("2.1.0")
|
||||
if got := testutil.ToFloat64(m.AppInfo.WithLabelValues("2.1.0")); got != 1 {
|
||||
t.Fatalf("app_info(version=2.1.0) expected 1, got %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestObserveTaskRun_IncrementsCounterAndHistogram(t *testing.T) {
|
||||
m := New("test")
|
||||
m.ObserveTaskRun("mysql", "success", 12.5, 1024)
|
||||
m.ObserveTaskRun("mysql", "failed", 3.0, 0)
|
||||
if got := testutil.ToFloat64(m.TaskRunTotal.WithLabelValues("success", "mysql")); got != 1 {
|
||||
t.Fatalf("task_run_total{status=success,task_type=mysql}: expected 1, got %v", got)
|
||||
}
|
||||
if got := testutil.ToFloat64(m.TaskRunTotal.WithLabelValues("failed", "mysql")); got != 1 {
|
||||
t.Fatalf("task_run_total{status=failed,task_type=mysql}: expected 1, got %v", got)
|
||||
}
|
||||
if got := testutil.ToFloat64(m.TaskBytesTotal.WithLabelValues("mysql")); got != 1024 {
|
||||
t.Fatalf("task_bytes_total{task_type=mysql}: expected 1024, got %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestObserveTaskRun_NilReceiverIsSafe(t *testing.T) {
|
||||
var m *Metrics // nil
|
||||
m.ObserveTaskRun("file", "success", 1, 1)
|
||||
m.ObserveRestore("success")
|
||||
m.ObserveVerify("failed")
|
||||
m.ObserveReplication("success")
|
||||
m.IncTaskRunning()
|
||||
m.DecTaskRunning()
|
||||
m.SetStorageUsed("a", "s3", 1)
|
||||
m.SetNodeOnline("n1", "master", true)
|
||||
m.SetSLABreach(3)
|
||||
m.ResetNodeOnline()
|
||||
m.ResetStorageUsed()
|
||||
// no panic -> pass
|
||||
}
|
||||
|
||||
func TestHandler_ExposesBackupxMetrics(t *testing.T) {
|
||||
m := New("0.0.0-test")
|
||||
m.ObserveTaskRun("file", "success", 1.0, 2048)
|
||||
m.SetNodeOnline("n1", "master", true)
|
||||
m.SetSLABreach(1)
|
||||
|
||||
recorder := httptest.NewRecorder()
|
||||
req := httptest.NewRequest("GET", "/metrics", nil)
|
||||
m.Handler().ServeHTTP(recorder, req)
|
||||
|
||||
body, err := io.ReadAll(recorder.Result().Body)
|
||||
if err != nil {
|
||||
t.Fatalf("read body: %v", err)
|
||||
}
|
||||
content := string(body)
|
||||
for _, keyword := range []string{
|
||||
"backupx_task_run_total",
|
||||
"backupx_task_run_duration_seconds",
|
||||
"backupx_node_online",
|
||||
"backupx_sla_breach_tasks",
|
||||
"backupx_app_info",
|
||||
} {
|
||||
if !strings.Contains(content, keyword) {
|
||||
t.Errorf("expected /metrics to contain %q", keyword)
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user