feat(localdb): replication health check + meter export
LocalDbReplicationHealthCheck reports Healthy when replication is default-OFF or LocalDb is absent (admin-only graphs) - so a plain node is never degraded by it - Degraded when configured-but-disconnected, connected-with-unknown-backlog (a failed oplog poll must not read as zero), or connected-with-backlog past the threshold. Never Unhealthy: a replication fault does not stop the node serving its address space. Pure decision core is table-tested; registered unconditionally via a factory so AddOtOpcUaHealth keeps its no-arg signature and resolves ISyncStatus optionally. Adds LocalDbMetrics.MeterName to the observability allowlist. o.Meters is a strict allowlist with no wildcard, so the localdb.sync.* / localdb.oplog.depth series were otherwise silently absent from /metrics - the omission a ScadaBridge live gate caught.
This commit is contained in:
@@ -1,10 +1,15 @@
|
||||
using Microsoft.AspNetCore.Routing;
|
||||
using Microsoft.EntityFrameworkCore;
|
||||
using Microsoft.Extensions.Configuration;
|
||||
using Microsoft.Extensions.DependencyInjection;
|
||||
using Microsoft.Extensions.Diagnostics.HealthChecks;
|
||||
using Microsoft.Extensions.Options;
|
||||
using ZB.MOM.WW.Health;
|
||||
using ZB.MOM.WW.Health.Akka;
|
||||
using ZB.MOM.WW.Health.EntityFrameworkCore;
|
||||
using ZB.MOM.WW.LocalDb.Replication;
|
||||
using ZB.MOM.WW.OtOpcUa.Configuration;
|
||||
using ZB.MOM.WW.OtOpcUa.Host.Configuration;
|
||||
|
||||
namespace ZB.MOM.WW.OtOpcUa.Host.Health;
|
||||
|
||||
@@ -36,7 +41,20 @@ public static class HealthEndpoints
|
||||
"admin-leader",
|
||||
failureStatus: null,
|
||||
tags: new[] { ZbHealthTags.Active },
|
||||
args: "admin");
|
||||
args: "admin")
|
||||
// Registered unconditionally, not driver-gated. AddOtOpcUaHealth takes no role argument
|
||||
// and runs on every node; the check itself resolves ISyncStatus optionally and reports
|
||||
// Healthy when LocalDb is absent (admin-only graphs) or replication is default-OFF, so a
|
||||
// plain node is never degraded by it. A factory registration keeps this no-arg signature
|
||||
// while still reading ISyncStatus + options + the sync port from the container.
|
||||
.Add(new HealthCheckRegistration(
|
||||
"localdb-replication",
|
||||
sp => new LocalDbReplicationHealthCheck(
|
||||
sp.GetService<ISyncStatus>(),
|
||||
sp.GetRequiredService<IOptions<ReplicationOptions>>(),
|
||||
LocalDbRegistration.SyncListenPort(sp.GetRequiredService<IConfiguration>())),
|
||||
failureStatus: null,
|
||||
tags: new[] { ZbHealthTags.Active }));
|
||||
return services;
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,126 @@
|
||||
using Microsoft.Extensions.Diagnostics.HealthChecks;
|
||||
using Microsoft.Extensions.Options;
|
||||
using ZB.MOM.WW.LocalDb.Replication;
|
||||
|
||||
namespace ZB.MOM.WW.OtOpcUa.Host.Health;
|
||||
|
||||
/// <summary>
|
||||
/// Pure decision function for the LocalDb replication probe, factored out of
|
||||
/// <see cref="LocalDbReplicationHealthCheck"/> so the whole matrix is table-testable without
|
||||
/// standing up a replication engine.
|
||||
/// </summary>
|
||||
public static class LocalDbReplicationDecision
|
||||
{
|
||||
/// <summary>
|
||||
/// Maps the resolved replication facts to a health status.
|
||||
/// </summary>
|
||||
/// <param name="peerConfigured">
|
||||
/// True when this node participates in replication at all — it has a peer address to dial, or
|
||||
/// a sync listener bound. False means default-OFF, and default-OFF must never degrade a node.
|
||||
/// </param>
|
||||
/// <param name="connected">Whether a sync session is currently running.</param>
|
||||
/// <param name="oplogBacklog">
|
||||
/// The unacked oplog backlog, or <see langword="null"/> when it could not be read. Null is
|
||||
/// deliberately not treated as zero: a failed poll is "unknown", not "caught up".
|
||||
/// </param>
|
||||
/// <param name="degradedThreshold">
|
||||
/// Backlog at or above which a connected pair is judged to be falling behind.
|
||||
/// </param>
|
||||
/// <returns>The status plus a human-readable reason.</returns>
|
||||
public static (HealthStatus Status, string Description) Evaluate(
|
||||
bool peerConfigured, bool connected, long? oplogBacklog, long degradedThreshold)
|
||||
{
|
||||
if (!peerConfigured)
|
||||
return (HealthStatus.Healthy, "LocalDb replication is not configured on this node (default-OFF).");
|
||||
|
||||
if (!connected)
|
||||
return (HealthStatus.Degraded, "LocalDb replication is configured but no sync session is connected.");
|
||||
|
||||
if (oplogBacklog is null)
|
||||
return (HealthStatus.Degraded, "LocalDb replication is connected but its oplog backlog is unknown (the poll failed).");
|
||||
|
||||
if (oplogBacklog.Value >= degradedThreshold)
|
||||
return (HealthStatus.Degraded,
|
||||
$"LocalDb replication is connected but its oplog backlog ({oplogBacklog.Value}) is at or above the degraded threshold ({degradedThreshold}).");
|
||||
|
||||
return (HealthStatus.Healthy, $"LocalDb replication is connected; oplog backlog {oplogBacklog.Value}.");
|
||||
}
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Reports the health of this node's LocalDb replication link. Registered unconditionally with
|
||||
/// the shared health pipeline; when LocalDb is not present at all (admin-only graphs) it reports
|
||||
/// <see cref="HealthStatus.Healthy"/> rather than degrading — matching the
|
||||
/// <c>ActiveNodeHealthCheck</c> precedent of "not applicable ⇒ Healthy".
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// A node running LocalDb with replication switched off (no peer, no listener) is also Healthy:
|
||||
/// that is the default posture for most of the fleet, and it must not look degraded. Only a node
|
||||
/// that is <i>meant</i> to be replicating and is not — or is connected but cannot confirm it is
|
||||
/// draining — is surfaced as Degraded. It never reports Unhealthy: a replication problem does not
|
||||
/// stop the node serving its address space, so it must not fail a readiness gate.
|
||||
/// </remarks>
|
||||
public sealed class LocalDbReplicationHealthCheck : IHealthCheck
|
||||
{
|
||||
/// <summary>
|
||||
/// Backlog at or above which a connected pair is reported Degraded. Well below the library's
|
||||
/// <c>MaxOplogRows</c> default (1,000,000, where a snapshot resync is forced), so the probe
|
||||
/// flags a pair that is falling behind long before the engine itself intervenes.
|
||||
/// </summary>
|
||||
public const long DefaultBacklogDegradedThreshold = 100_000;
|
||||
|
||||
private readonly ISyncStatus? _syncStatus;
|
||||
private readonly ReplicationOptions _options;
|
||||
private readonly int _syncListenPort;
|
||||
private readonly long _degradedThreshold;
|
||||
|
||||
/// <summary>Creates the health check.</summary>
|
||||
/// <param name="syncStatus">
|
||||
/// The replication status singleton, or <see langword="null"/> when the replication engine is
|
||||
/// not registered (admin-only nodes) — in which case the check is a Healthy no-op.
|
||||
/// </param>
|
||||
/// <param name="options">The bound replication options (peer address, etc.).</param>
|
||||
/// <param name="syncListenPort">
|
||||
/// The configured sync listener port; a value greater than zero counts as "replication
|
||||
/// configured" even when this node is the passive side with no peer address.
|
||||
/// </param>
|
||||
/// <param name="degradedThreshold">Backlog degraded threshold; defaults to <see cref="DefaultBacklogDegradedThreshold"/>.</param>
|
||||
public LocalDbReplicationHealthCheck(
|
||||
ISyncStatus? syncStatus,
|
||||
IOptions<ReplicationOptions> options,
|
||||
int syncListenPort,
|
||||
long degradedThreshold = DefaultBacklogDegradedThreshold)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(options);
|
||||
|
||||
_syncStatus = syncStatus;
|
||||
_options = options.Value;
|
||||
_syncListenPort = syncListenPort;
|
||||
_degradedThreshold = degradedThreshold;
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public Task<HealthCheckResult> CheckHealthAsync(
|
||||
HealthCheckContext context, CancellationToken cancellationToken = default)
|
||||
{
|
||||
// No replication engine registered at all → not applicable → Healthy. Admin-only nodes and
|
||||
// any graph that did not call AddOtOpcUaLocalDb land here.
|
||||
if (_syncStatus is null)
|
||||
return Task.FromResult(HealthCheckResult.Healthy(
|
||||
"LocalDb replication is not present on this node."));
|
||||
|
||||
var peerConfigured =
|
||||
!string.IsNullOrWhiteSpace(_options.PeerAddress) || _syncListenPort > 0;
|
||||
|
||||
var (status, description) = LocalDbReplicationDecision.Evaluate(
|
||||
peerConfigured, _syncStatus.Connected, _syncStatus.OplogBacklog, _degradedThreshold);
|
||||
|
||||
var result = status switch
|
||||
{
|
||||
HealthStatus.Healthy => HealthCheckResult.Healthy(description),
|
||||
_ => HealthCheckResult.Degraded(description),
|
||||
};
|
||||
|
||||
return Task.FromResult(result);
|
||||
}
|
||||
}
|
||||
@@ -1,4 +1,5 @@
|
||||
using Microsoft.Extensions.Configuration;
|
||||
using ZB.MOM.WW.LocalDb.Replication;
|
||||
using ZB.MOM.WW.OtOpcUa.Commons.Observability;
|
||||
using ZB.MOM.WW.Telemetry;
|
||||
|
||||
@@ -27,7 +28,11 @@ public static class ObservabilityExtensions
|
||||
return services.AddZbTelemetry(o =>
|
||||
{
|
||||
o.ServiceName = "otopcua";
|
||||
o.Meters = [OtOpcUaTelemetry.MeterName];
|
||||
// o.Meters is a STRICT allowlist — ZbTelemetry adds only the named meters, with no
|
||||
// wildcard. The LocalDb replication engine publishes under its own meter name, so
|
||||
// without this entry its localdb.sync.* / localdb.oplog.depth series are silently absent
|
||||
// from /metrics on driver nodes (the exact omission a ScadaBridge live gate caught).
|
||||
o.Meters = [OtOpcUaTelemetry.MeterName, LocalDbMetrics.MeterName];
|
||||
o.ActivitySources = [OtOpcUaTelemetry.ActivitySourceName];
|
||||
if (Enum.TryParse<ZbExporter>(configuration["OtOpcUa:Telemetry:Exporter"], ignoreCase: true, out var exporter))
|
||||
o.Exporter = exporter;
|
||||
|
||||
+122
@@ -0,0 +1,122 @@
|
||||
using Microsoft.Extensions.Diagnostics.HealthChecks;
|
||||
using Microsoft.Extensions.Options;
|
||||
using Shouldly;
|
||||
using Xunit;
|
||||
using ZB.MOM.WW.LocalDb.Replication;
|
||||
using ZB.MOM.WW.OtOpcUa.Host.Health;
|
||||
|
||||
namespace ZB.MOM.WW.OtOpcUa.Host.IntegrationTests;
|
||||
|
||||
/// <summary>
|
||||
/// LocalDb Phase 1 (Task 10) — the replication health check.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The load-bearing case is the default-OFF one: a plain driver node with no peer and no
|
||||
/// listener must report <b>Healthy</b>, or every unreplicated node in the fleet degrades the
|
||||
/// moment this check ships. Beyond that, "connected" is not sufficient for healthy — an unknown
|
||||
/// backlog (a failed oplog poll) must not read as zero.
|
||||
/// </remarks>
|
||||
public sealed class LocalDbReplicationHealthCheckTests
|
||||
{
|
||||
// ---- Pure decision core (table-tested) --------------------------------------------------
|
||||
|
||||
[Fact]
|
||||
public void Unconfigured_IsHealthy_SoDefaultOffNodesDoNotDegrade()
|
||||
{
|
||||
LocalDbReplicationDecision.Evaluate(
|
||||
peerConfigured: false, connected: false, oplogBacklog: null, degradedThreshold: 100_000)
|
||||
.Status.ShouldBe(HealthStatus.Healthy);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void ConfiguredButNotConnected_IsDegraded()
|
||||
{
|
||||
LocalDbReplicationDecision.Evaluate(
|
||||
peerConfigured: true, connected: false, oplogBacklog: null, degradedThreshold: 100_000)
|
||||
.Status.ShouldBe(HealthStatus.Degraded);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void ConnectedButBacklogUnknown_IsDegraded()
|
||||
{
|
||||
// null backlog = the oplog poll failed. Reporting Healthy here would call a pair that cannot
|
||||
// read its own oplog perfectly fine.
|
||||
LocalDbReplicationDecision.Evaluate(
|
||||
peerConfigured: true, connected: true, oplogBacklog: null, degradedThreshold: 100_000)
|
||||
.Status.ShouldBe(HealthStatus.Degraded);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void ConnectedWithSmallBacklog_IsHealthy()
|
||||
{
|
||||
LocalDbReplicationDecision.Evaluate(
|
||||
peerConfigured: true, connected: true, oplogBacklog: 12, degradedThreshold: 100_000)
|
||||
.Status.ShouldBe(HealthStatus.Healthy);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void ConnectedWithBacklogOverThreshold_IsDegraded()
|
||||
{
|
||||
// A backlog past the threshold means the peer is falling behind faster than it drains —
|
||||
// connected, but not keeping up.
|
||||
LocalDbReplicationDecision.Evaluate(
|
||||
peerConfigured: true, connected: true, oplogBacklog: 100_001, degradedThreshold: 100_000)
|
||||
.Status.ShouldBe(HealthStatus.Degraded);
|
||||
}
|
||||
|
||||
// ---- IHealthCheck wrapper ---------------------------------------------------------------
|
||||
|
||||
[Fact]
|
||||
public async Task Wrapper_WithNoSyncStatusRegistered_IsHealthy()
|
||||
{
|
||||
// Admin-only graphs never register the replication engine. The check is registered
|
||||
// unconditionally (AddOtOpcUaHealth takes no role argument), so it must no-op to Healthy
|
||||
// rather than throw or degrade when LocalDb is simply absent.
|
||||
var check = new LocalDbReplicationHealthCheck(
|
||||
syncStatus: null,
|
||||
options: Options.Create(new ReplicationOptions()),
|
||||
syncListenPort: 0);
|
||||
|
||||
var result = await check.CheckHealthAsync(new HealthCheckContext(), TestContext.Current.CancellationToken);
|
||||
|
||||
result.Status.ShouldBe(HealthStatus.Healthy);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task Wrapper_ListenerConfiguredButNotConnected_IsDegraded()
|
||||
{
|
||||
// A passive node (no PeerAddress) still counts as "replication configured" when it is
|
||||
// listening — otherwise a passive node that never gets dialled would look healthy while
|
||||
// silently accepting nothing.
|
||||
var check = new LocalDbReplicationHealthCheck(
|
||||
syncStatus: new StubSyncStatus { Connected = false },
|
||||
options: Options.Create(new ReplicationOptions { PeerAddress = "" }),
|
||||
syncListenPort: 9001);
|
||||
|
||||
var result = await check.CheckHealthAsync(new HealthCheckContext(), TestContext.Current.CancellationToken);
|
||||
|
||||
result.Status.ShouldBe(HealthStatus.Degraded);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task Wrapper_PeerConfiguredAndConnectedAndDraining_IsHealthy()
|
||||
{
|
||||
var check = new LocalDbReplicationHealthCheck(
|
||||
syncStatus: new StubSyncStatus { Connected = true, OplogBacklog = 3 },
|
||||
options: Options.Create(new ReplicationOptions { PeerAddress = "https://peer:9001" }),
|
||||
syncListenPort: 9001);
|
||||
|
||||
var result = await check.CheckHealthAsync(new HealthCheckContext(), TestContext.Current.CancellationToken);
|
||||
|
||||
result.Status.ShouldBe(HealthStatus.Healthy);
|
||||
}
|
||||
|
||||
private sealed class StubSyncStatus : ISyncStatus
|
||||
{
|
||||
public bool Connected { get; init; }
|
||||
public string? PeerNodeId { get; init; }
|
||||
public DateTimeOffset? LastSyncUtc { get; init; }
|
||||
public long? OplogBacklog { get; init; }
|
||||
public long ConnectionAttempts { get; init; }
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user