监控 - 异常重试告警
概述
在幂等性系统中,异常重试是常见现象。但过度的重试可能导致:
- 资源浪费:大量无效请求消耗系统资源
- 雪崩效应:重试风暴导致系统崩溃
- 用户体验差:请求长时间无响应
- 数据不一致:重试可能导致状态混乱
需要建立完善的监控和告警机制,及时发现和处理异常重试。
关键指标
1. 重试相关指标
重试监控指标体系:
├── 重试频率
│ ├── 总重试次数
│ ├── 重试率(重试/总请求)
│ └── 每用户重试次数
├── 重试原因
│ ├── 超时重试
│ ├── 网络错误重试
│ ├── 业务冲突重试
│ └── Token 过期重试
├── 重试效果
│ ├── 重试成功率
│ ├── 平均重试次数
│ └── 最大重试次数
└── 系统影响
├── CPU 使用率
├── 内存使用率
├── 数据库连接数
└── Redis QPSOpenTelemetry 实现
1. 定义重试指标
csharp
using System.Diagnostics.Metrics;
public class RetryMetrics : IDisposable
{
private readonly Meter _meter;
// 计数器
private readonly Counter<long> _retryTotal;
private readonly Counter<long> _retryByReason;
private readonly Counter<long> _retryExhausted;
// 直方图
private readonly Histogram<int> _retryCountPerRequest;
private readonly Histogram<double> _retryDelay;
private readonly Histogram<double> _totalRetryTime;
// 可观察量
private readonly ObservableGauge<double> _retryRate;
private long _totalRequests = 0;
private long _totalRetries = 0;
public RetryMetrics(IMeterFactory meterFactory)
{
_meter = meterFactory.Create("Idempotency.Retry", "1.0.0");
_retryTotal = _meter.CreateCounter<long>(
"retry.total",
description: "Total number of retries");
_retryByReason = _meter.CreateCounter<long>(
"retry.by_reason",
description: "Retries categorized by reason");
_retryExhausted = _meter.CreateCounter<long>(
"retry.exhausted",
description: "Requests that exhausted all retry attempts");
_retryCountPerRequest = _meter.CreateHistogram<int>(
"retry.count_per_request",
description: "Number of retries per request");
_retryDelay = _meter.CreateHistogram<double>(
"retry.delay",
unit: "ms",
description: "Delay between retries");
_totalRetryTime = _meter.CreateHistogram<double>(
"retry.total_time",
unit: "ms",
description: "Total time spent on retries");
_retryRate = _meter.CreateObservableGauge(
"retry.rate",
() => GetRetryRate(),
description: "Current retry rate (retries/total requests)");
}
public void RecordRetry(string reason, int retryCount, double delayMs, double totalTimeMs)
{
Interlocked.Increment(ref _totalRetries);
_retryTotal.Add(1);
_retryByReason.Add(1, new KeyValuePair<string, object?>("reason", reason));
_retryCountPerRequest.Record(retryCount);
_retryDelay.Record(delayMs);
_totalRetryTime.Record(totalTimeMs);
}
public void RecordRetryExhausted(string reason)
{
_retryExhausted.Add(1, new KeyValuePair<string, object?>("reason", reason));
}
public void RecordRequest()
{
Interlocked.Increment(ref _totalRequests);
}
private double GetRetryRate()
{
var total = Volatile.Read(ref _totalRequests);
var retries = Volatile.Read(ref _totalRetries);
return total > 0 ? (double)retries / total : 0;
}
public void Dispose()
{
_meter?.Dispose();
}
}2. 重试中间件
csharp
public class RetryMonitoringMiddleware
{
private readonly RequestDelegate _next;
private readonly RetryMetrics _metrics;
private readonly ILogger<RetryMonitoringMiddleware> _logger;
public async Task InvokeAsync(HttpContext context)
{
var stopwatch = Stopwatch.StartNew();
var requestId = context.TraceIdentifier;
var retryCount = 0;
try
{
await _next(context);
stopwatch.Stop();
_metrics.RecordRequest();
// 记录高延迟请求
if (stopwatch.ElapsedMilliseconds > 1000)
{
_logger.LogWarning("Slow request: {RequestId}, Duration: {Duration}ms",
requestId, stopwatch.ElapsedMilliseconds);
}
}
catch (Exception ex)
{
stopwatch.Stop();
// 判断是否为可重试错误
if (IsRetryableError(ex))
{
_metrics.RecordRetry(
GetRetryReason(ex),
retryCount,
0,
stopwatch.ElapsedMilliseconds);
_logger.LogWarning(ex,
"Retryable error for request {RequestId}: {Error}",
requestId, ex.Message);
}
throw;
}
}
private bool IsRetryableError(Exception ex)
{
return ex is TimeoutException ||
ex is HttpRequestException httpEx &&
(httpEx.StatusCode >= 500 || httpEx.StatusCode == 408 || httpEx.StatusCode == 429);
}
private string GetRetryReason(Exception ex)
{
return ex switch
{
TimeoutException => "timeout",
HttpRequestException when ((HttpRequestException)ex).StatusCode == 429 => "rate_limit",
HttpRequestException when ((HttpRequestException)ex).StatusCode >= 500 => "server_error",
_ => "unknown"
};
}
}Prometheus 告警规则
1. 基础告警
yaml
groups:
- name: retry_alerts
rules:
# 重试率过高
- alert: HighRetryRate
expr: rate(retry_total[5m]) / rate(request_total[5m]) > 0.2
for: 5m
labels:
severity: warning
annotations:
summary: "High retry rate detected"
description: "More than 20% of requests are being retried"
# 重试耗尽过多
- alert: HighRetryExhaustion
expr: rate(retry_exhausted_total[5m]) > 10
for: 5m
labels:
severity: critical
annotations:
summary: "Many requests exhausting retries"
description: "More than 10 requests per second exhausted all retry attempts"
# 平均重试次数过高
- alert: HighAverageRetryCount
expr: histogram_quantile(0.5, rate(retry_count_per_request_bucket[5m])) > 3
for: 10m
labels:
severity: warning
annotations:
summary: "High average retry count"
description: "Median retry count per request exceeds 3"2. 原因分类告警
yaml
# 超时重试过多
- alert: HighTimeoutRetries
expr: rate(retry_by_reason_total{reason="timeout"}[5m]) > 50
for: 5m
labels:
severity: warning
annotations:
summary: "High timeout retry rate"
description: "More than 50 timeout retries per second"
# 服务器错误重试过多
- alert: HighServerErrorRetries
expr: rate(retry_by_reason_total{reason="server_error"}[5m]) > 30
for: 5m
labels:
severity: critical
annotations:
summary: "High server error retry rate"
description: "More than 30 server error retries per second"
# 限流重试过多
- alert: HighRateLimitRetries
expr: rate(retry_by_reason_total{reason="rate_limit"}[5m]) > 100
for: 5m
labels:
severity: warning
annotations:
summary: "High rate limit retry rate"
description: "Clients are hitting rate limits frequently"3. 系统影响告警
yaml
# 重试导致的 CPU 使用率过高
- alert: HighCPUFromRetries
expr: cpu_usage_percent > 80 and rate(retry_total[5m]) > 100
for: 5m
labels:
severity: critical
annotations:
summary: "High CPU usage due to retries"
description: "CPU usage exceeds 80% with high retry rate"
# 数据库连接池耗尽
- alert: DatabaseConnectionPoolExhaustion
expr: db_connection_pool_used / db_connection_pool_max > 0.9
for: 2m
labels:
severity: critical
annotations:
summary: "Database connection pool nearly exhausted"
description: "More than 90% of database connections are in use"
# Redis QPS 过高
- alert: HighRedisQPS
expr: rate(redis_commands_processed_total[5m]) > 10000
for: 5m
labels:
severity: warning
annotations:
summary: "High Redis QPS"
description: "Redis is processing more than 10000 commands per second"Grafana 仪表板
1. 重试概览面板
json
{
"dashboard": {
"title": "Retry Monitoring Dashboard",
"panels": [
{
"title": "Retry Rate Over Time",
"type": "graph",
"targets": [
{
"expr": "rate(retry_total[5m]) / rate(request_total[5m])",
"legendFormat": "Retry Rate"
}
],
"thresholds": [
{
"value": 0.2,
"colorMode": "critical"
}
]
},
{
"title": "Retries by Reason",
"type": "piechart",
"targets": [
{
"expr": "sum by (reason) (rate(retry_by_reason_total[5m]))",
"legendFormat": "{{reason}}"
}
]
},
{
"title": "Retry Count Distribution",
"type": "heatmap",
"targets": [
{
"expr": "rate(retry_count_per_request_bucket[5m])",
"legendFormat": "{{le}}"
}
]
},
{
"title": "Retry Exhaustion Rate",
"type": "stat",
"targets": [
{
"expr": "rate(retry_exhausted_total[5m])",
"legendFormat": "Exhausted/sec"
}
],
"thresholds": [
{
"value": 10,
"colorMode": "critical"
}
]
}
]
}
}自动降级策略
1. 基于重试率的自动降级
csharp
public class AdaptiveRetryPolicy
{
private readonly RetryMetrics _metrics;
private readonly ILogger<AdaptiveRetryPolicy> _logger;
private int _maxRetries = 3;
private TimeSpan _baseDelay = TimeSpan.FromSeconds(1);
public int MaxRetries => _maxRetries;
public TimeSpan BaseDelay => _baseDelay;
public void UpdatePolicy()
{
var retryRate = GetCurrentRetryRate();
if (retryRate > 0.5)
{
// 重试率超过 50%,大幅降低重试次数
_maxRetries = 1;
_baseDelay = TimeSpan.FromSeconds(5);
_logger.LogWarning("High retry rate ({Rate}), reducing max retries to {MaxRetries}",
retryRate, _maxRetries);
}
else if (retryRate > 0.3)
{
// 重试率超过 30%,适度降低
_maxRetries = 2;
_baseDelay = TimeSpan.FromSeconds(2);
_logger.LogWarning("Moderate retry rate ({Rate}), adjusting policy", retryRate);
}
else if (retryRate < 0.1)
{
// 重试率低于 10%,恢复正常策略
_maxRetries = 3;
_baseDelay = TimeSpan.FromSeconds(1);
}
}
private double GetCurrentRetryRate()
{
// 从指标系统获取当前重试率
return _metrics.GetCurrentRetryRate();
}
}2. 熔断器集成
csharp
public class RetryAwareCircuitBreaker
{
private readonly CircuitBreaker _circuitBreaker;
private readonly RetryMetrics _metrics;
public async Task<T> ExecuteAsync<T>(Func<Task<T>> action)
{
// 检查熔断器状态
if (_circuitBreaker.State == CircuitState.Open)
{
_metrics.RecordRetryExhausted("circuit_breaker_open");
throw new CircuitBreakerOpenException("Circuit breaker is open");
}
try
{
return await action();
}
catch (Exception ex)
{
_circuitBreaker.RecordFailure();
throw;
}
}
}日志记录
结构化日志
csharp
public class RetryLogger
{
private readonly ILogger<RetryLogger> _logger;
public void LogRetryAttempt(
string requestId,
string endpoint,
int attemptNumber,
string reason,
TimeSpan delay)
{
_logger.LogWarning(
"Retry attempt {Attempt} for request {RequestId} on {Endpoint}, Reason: {Reason}, Delay: {Delay}ms",
attemptNumber,
requestId,
endpoint,
reason,
delay.TotalMilliseconds);
}
public void LogRetryExhausted(
string requestId,
string endpoint,
int totalAttempts,
string finalError)
{
_logger.LogError(
"Request {RequestId} on {Endpoint} exhausted all {Attempts} retry attempts. Final error: {Error}",
requestId,
endpoint,
totalAttempts,
finalError);
}
}最佳实践总结
✅ DO
- 全面监控:收集重试频率、原因、效果等指标
- 分级告警:根据严重程度设置不同告警级别
- 自动降级:重试率过高时自动调整策略
- 详细日志:记录每次重试的详细信息
- 定期分析:分析重试模式,优化系统
❌ DON'T
- 不要无限重试:设置最大重试次数
- 不要忽略告警:及时处理重试异常
- 不要固定重试间隔:使用指数退避
- 不要忘记清理:定期清理旧的重试记录
总结
异常重试告警是保证系统稳定性的关键:
✅ 全面监控:OpenTelemetry 指标收集
✅ 智能告警:Prometheus 分级告警规则
✅ 可视化:Grafana 实时监控面板
✅ 自动降级:基于重试率的自适应策略
通过完善的监控和告警机制,可以及时发现和处理异常重试,保证系统的稳定性和可用性。