Skip to content

监控 - 幂等冲突日志 ​

概述 ​

监控幂等性系统的运行状态对于及时发现问题、优化性能至关重要。需要收集关键指标、设置告警规则,并建立可视化的监控面板。

关键指标 ​

1. 核心指标分类 ​

幂等性监控指标体系:
├── 请求指标
│   ├── 总请求数
│   ├── 新请求数
│   └── 重复请求数
├── 性能指标
│   ├── 平均响应时间
│   ├── P95/P99 响应时间
│   └── 锁等待时间
├── 存储指标
│   ├── Token 数量
│   ├── 缓存命中率
│   └── 存储空间使用率
└── 错误指标
    ├── 冲突率
    ├── 超时率
    └── 失败率

OpenTelemetry 实现 ​

1. 配置 OpenTelemetry ​

csharp
using OpenTelemetry;
using OpenTelemetry.Metrics;
using OpenTelemetry.Trace;

public static class OpenTelemetryConfig
{
    public static IServiceCollection AddCustomOpenTelemetry(this IServiceCollection services)
    {
        services.AddOpenTelemetry()
            .WithTracing(tracing =>
            {
                tracing.AddAspNetCoreInstrumentation()
                    .AddHttpClientInstrumentation()
                    .AddSource("Idempotency")
                    .AddOtlpExporter(options =>
                    {
                        options.Endpoint = new Uri("http://localhost:4317");
                    });
            })
            .WithMetrics(metrics =>
            {
                metrics.AddAspNetCoreInstrumentation()
                    .AddHttpClientInstrumentation()
                    .AddMeter("Idempotency")
                    .AddPrometheusExporter();
            });
        
        return services;
    }
}

// Program.cs
builder.Services.AddCustomOpenTelemetry();

2. 定义指标 ​

csharp
using System.Diagnostics.Metrics;

public class IdempotencyMetrics : IDisposable
{
    private readonly Meter _meter;
    
    // 计数器
    private readonly Counter<long> _requestsTotal;
    private readonly Counter<long> _newRequests;
    private readonly Counter<long> _duplicateRequests;
    private readonly Counter<long> _cacheHits;
    private readonly Counter<long> _cacheMisses;
    private readonly Counter<long> _conflicts;
    private readonly Counter<long> _errors;
    
    // 直方图
    private readonly Histogram<double> _requestDuration;
    private readonly Histogram<double> _lockWaitTime;
    private readonly Histogram<int> _retryCount;
    
    // 可观察量
    private readonly ObservableGauge<int> _activeTokens;
    private readonly ObservableGauge<double> _cacheHitRate;
    
    private long _cacheHitsCount = 0;
    private long _cacheMissesCount = 0;
    private int _currentTokenCount = 0;
    
    public IdempotencyMetrics(IMeterFactory meterFactory)
    {
        _meter = meterFactory.Create("Idempotency", "1.0.0");
        
        // 计数器
        _requestsTotal = _meter.CreateCounter<long>(
            "idempotency.requests.total",
            description: "Total number of idempotent requests");
        
        _newRequests = _meter.CreateCounter<long>(
            "idempotency.requests.new",
            description: "Number of new (non-duplicate) requests");
        
        _duplicateRequests = _meter.CreateCounter<long>(
            "idempotency.requests.duplicate",
            description: "Number of duplicate requests");
        
        _cacheHits = _meter.CreateCounter<long>(
            "idempotency.cache.hits",
            description: "Number of cache hits");
        
        _cacheMisses = _meter.CreateCounter<long>(
            "idempotency.cache.misses",
            description: "Number of cache misses");
        
        _conflicts = _meter.CreateCounter<long>(
            "idempotency.conflicts.total",
            description: "Number of idempotency conflicts");
        
        _errors = _meter.CreateCounter<long>(
            "idempotency.errors.total",
            description: "Number of idempotency errors");
        
        // 直方图
        _requestDuration = _meter.CreateHistogram<double>(
            "idempotency.request.duration",
            unit: "ms",
            description: "Request processing duration");
        
        _lockWaitTime = _meter.CreateHistogram<double>(
            "idempotency.lock.wait.time",
            unit: "ms",
            description: "Time spent waiting for lock");
        
        _retryCount = _meter.CreateHistogram<int>(
            "idempotency.retry.count",
            description: "Number of retries per request");
        
        // 可观察量
        _activeTokens = _meter.CreateObservableGauge(
            "idempotency.tokens.active",
            () => _currentTokenCount,
            description: "Number of active idempotency tokens");
        
        _cacheHitRate = _meter.CreateObservableGauge(
            "idempotency.cache.hit_rate",
            () => GetCacheHitRate(),
            description: "Cache hit rate (0-1)");
    }
    
    public void RecordRequest(bool isDuplicate, double durationMs)
    {
        _requestsTotal.Add(1);
        
        if (isDuplicate)
        {
            _duplicateRequests.Add(1);
        }
        else
        {
            _newRequests.Add(1);
        }
        
        _requestDuration.Record(durationMs);
    }
    
    public void RecordCacheHit()
    {
        Interlocked.Increment(ref _cacheHitsCount);
        _cacheHits.Add(1);
    }
    
    public void RecordCacheMiss()
    {
        Interlocked.Increment(ref _cacheMissesCount);
        _cacheMisses.Add(1);
    }
    
    public void RecordConflict()
    {
        _conflicts.Add(1);
    }
    
    public void RecordError(string errorType)
    {
        _errors.Add(1, new KeyValuePair<string, object?>("error_type", errorType));
    }
    
    public void RecordLockWait(double waitTimeMs)
    {
        _lockWaitTime.Record(waitTimeMs);
    }
    
    public void RecordRetry(int retryCount)
    {
        _retryCount.Record(retryCount);
    }
    
    public void UpdateTokenCount(int count)
    {
        _currentTokenCount = count;
    }
    
    private double GetCacheHitRate()
    {
        var total = Volatile.Read(ref _cacheHitsCount) + Volatile.Read(ref _cacheMissesCount);
        return total > 0 ? (double)Volatile.Read(ref _cacheHitsCount) / total : 0;
    }
    
    public void Dispose()
    {
        _meter?.Dispose();
    }
}

3. 在中间件中使用 ​

csharp
public class IdempotencyMiddlewareWithMetrics
{
    private readonly RequestDelegate _next;
    private readonly IdempotencyMetrics _metrics;
    private readonly ILogger<IdempotencyMiddlewareWithMetrics> _logger;
    
    public async Task InvokeAsync(HttpContext context)
    {
        var stopwatch = Stopwatch.StartNew();
        var idempotencyKey = context.Request.Headers["Idempotency-Key"].FirstOrDefault();
        
        try
        {
            // ... 幂等性处理逻辑
            
            if (IsDuplicateRequest(idempotencyKey))
            {
                _metrics.RecordCacheHit();
                _metrics.RecordRequest(true, stopwatch.ElapsedMilliseconds);
                
                // 返回缓存结果
                await ReturnCachedResponse(context, idempotencyKey);
                return;
            }
            
            _metrics.RecordCacheMiss();
            
            // 处理新请求
            await _next(context);
            
            _metrics.RecordRequest(false, stopwatch.ElapsedMilliseconds);
        }
        catch (Exception ex)
        {
            stopwatch.Stop();
            _metrics.RecordError(ex.GetType().Name);
            _metrics.RecordRequest(false, stopwatch.ElapsedMilliseconds);
            
            throw;
        }
    }
}

Prometheus + Grafana ​

1. Prometheus 配置 ​

yaml
# prometheus.yml
global:
  scrape_interval: 15s

scrape_configs:
  - job_name: 'idempotency-service'
    static_configs:
      - targets: ['localhost:5000']
    metrics_path: '/metrics'

2. Grafana 仪表板 JSON ​

json
{
  "dashboard": {
    "title": "Idempotency Monitoring",
    "panels": [
      {
        "title": "Request Rate",
        "type": "graph",
        "targets": [
          {
            "expr": "rate(idempotency_requests_total[5m])",
            "legendFormat": "Total"
          },
          {
            "expr": "rate(idempotency_requests_duplicate[5m])",
            "legendFormat": "Duplicate"
          }
        ]
      },
      {
        "title": "Cache Hit Rate",
        "type": "gauge",
        "targets": [
          {
            "expr": "idempotency_cache_hit_rate",
            "legendFormat": "Hit Rate"
          }
        ]
      },
      {
        "title": "Request Duration (P95)",
        "type": "graph",
        "targets": [
          {
            "expr": "histogram_quantile(0.95, rate(idempotency_request_duration_bucket[5m]))",
            "legendFormat": "P95"
          }
        ]
      },
      {
        "title": "Conflict Rate",
        "type": "graph",
        "targets": [
          {
            "expr": "rate(idempotency_conflicts_total[5m]) / rate(idempotency_requests_total[5m])",
            "legendFormat": "Conflict Rate"
          }
        ]
      }
    ]
  }
}

告警规则 ​

Prometheus AlertManager ​

yaml
# alert_rules.yml
groups:
  - name: idempotency_alerts
    rules:
      # 高重复请求率
      - alert: HighDuplicateRequestRate
        expr: |
          rate(idempotency_requests_duplicate[5m]) / 
          rate(idempotency_requests_total[5m]) > 0.3
        for: 5m
        labels:
          severity: warning
        annotations:
          summary: "High duplicate request rate"
          description: "More than 30% of requests are duplicates"
      
      # 低缓存命中率
      - alert: LowCacheHitRate
        expr: idempotency_cache_hit_rate < 0.5
        for: 10m
        labels:
          severity: warning
        annotations:
          summary: "Low cache hit rate"
          description: "Cache hit rate is below 50%"
      
      # 高冲突率
      - alert: HighConflictRate
        expr: |
          rate(idempotency_conflicts_total[5m]) / 
          rate(idempotency_requests_total[5m]) > 0.05
        for: 5m
        labels:
          severity: critical
        annotations:
          summary: "High idempotency conflict rate"
          description: "More than 5% of requests have conflicts"
      
      # 慢请求
      - alert: SlowIdempotencyRequests
        expr: |
          histogram_quantile(0.99, rate(idempotency_request_duration_bucket[5m])) > 1000
        for: 5m
        labels:
          severity: warning
        annotations:
          summary: "Slow idempotency requests"
          description: "99th percentile request duration exceeds 1 second"
      
      # 高错误率
      - alert: HighIdempotencyErrorRate
        expr: |
          rate(idempotency_errors_total[5m]) / 
          rate(idempotency_requests_total[5m]) > 0.01
        for: 5m
        labels:
          severity: critical
        annotations:
          summary: "High idempotency error rate"
          description: "More than 1% of requests are failing"

日志记录 ​

结构化日志 ​

csharp
public class IdempotencyLogger
{
    private readonly ILogger<IdempotencyLogger> _logger;
    
    public void LogDuplicateRequest(string idempotencyKey, string endpoint)
    {
        _logger.LogInformation(
            "Duplicate request detected: Key={Key}, Endpoint={Endpoint}",
            idempotencyKey,
            endpoint);
    }
    
    public void LogCacheHit(string idempotencyKey, TimeSpan cachedAge)
    {
        _logger.LogDebug(
            "Cache hit: Key={Key}, Age={Age}s",
            idempotencyKey,
            cachedAge.TotalSeconds);
    }
    
    public void LogConflict(string idempotencyKey, string reason)
    {
        _logger.LogWarning(
            "Idempotency conflict: Key={Key}, Reason={Reason}",
            idempotencyKey,
            reason);
    }
    
    public void LogTokenExpired(string idempotencyKey)
    {
        _logger.LogWarning(
            "Token expired: Key={Key}",
            idempotencyKey);
    }
}

ELK Stack 集成 ​

1. Serilog 配置 ​

csharp
using Serilog;
using Serilog.Sinks.Elasticsearch;

Log.Logger = new LoggerConfiguration()
    .Enrich.FromLogContext()
    .Enrich.WithProperty("Application", "IdempotencyService")
    .WriteTo.Console()
    .WriteTo.Elasticsearch(new ElasticsearchSinkOptions(new Uri("http://localhost:9200"))
    {
        AutoRegisterTemplate = true,
        IndexFormat = "idempotency-logs-{0:yyyy.MM.dd}"
    })
    .CreateLogger();

2. Kibana 查询示例 ​

kql
# 查找所有重复请求
event.category: idempotency AND idempotency.type: duplicate

# 查找高延迟请求
event.category: idempotency AND response.duration > 1000

# 统计每小时的重复请求数
event.category: idempotency AND idempotency.type: duplicate

最佳实践总结 ​

✅ DO ​

  1. 收集全面指标:请求、性能、存储、错误
  2. 设置合理告警:避免告警疲劳
  3. 结构化日志:便于搜索和分析
  4. 可视化面板:实时监控
  5. 定期review:优化指标和告警

❌ DON'T ​

  1. 不要忽略指标:数据驱动优化
  2. 不要设置过多告警:关注关键指标
  3. 不要记录敏感信息:脱敏幂等键
  4. 不要忘记清理旧数据:控制存储成本

总结 ​

完善的监控体系是幂等性系统稳定运行的保障:

✅ 指标收集:OpenTelemetry 标准化
✅ 可视化:Grafana 仪表板
✅ 告警:Prometheus AlertManager
✅ 日志:ELK Stack 集中管理

通过全面的监控,可以及时发现问题、优化性能、保证系统可靠性。

Released under the MIT License.