feat(localdb): replication health check + meter export
LocalDbReplicationHealthCheck reports Healthy when replication is default-OFF or LocalDb is absent (admin-only graphs) - so a plain node is never degraded by it - Degraded when configured-but-disconnected, connected-with-unknown-backlog (a failed oplog poll must not read as zero), or connected-with-backlog past the threshold. Never Unhealthy: a replication fault does not stop the node serving its address space. Pure decision core is table-tested; registered unconditionally via a factory so AddOtOpcUaHealth keeps its no-arg signature and resolves ISyncStatus optionally. Adds LocalDbMetrics.MeterName to the observability allowlist. o.Meters is a strict allowlist with no wildcard, so the localdb.sync.* / localdb.oplog.depth series were otherwise silently absent from /metrics - the omission a ScadaBridge live gate caught.
This commit is contained in:
@@ -1,10 +1,15 @@
|
|||||||
using Microsoft.AspNetCore.Routing;
|
using Microsoft.AspNetCore.Routing;
|
||||||
using Microsoft.EntityFrameworkCore;
|
using Microsoft.EntityFrameworkCore;
|
||||||
|
using Microsoft.Extensions.Configuration;
|
||||||
using Microsoft.Extensions.DependencyInjection;
|
using Microsoft.Extensions.DependencyInjection;
|
||||||
|
using Microsoft.Extensions.Diagnostics.HealthChecks;
|
||||||
|
using Microsoft.Extensions.Options;
|
||||||
using ZB.MOM.WW.Health;
|
using ZB.MOM.WW.Health;
|
||||||
using ZB.MOM.WW.Health.Akka;
|
using ZB.MOM.WW.Health.Akka;
|
||||||
using ZB.MOM.WW.Health.EntityFrameworkCore;
|
using ZB.MOM.WW.Health.EntityFrameworkCore;
|
||||||
|
using ZB.MOM.WW.LocalDb.Replication;
|
||||||
using ZB.MOM.WW.OtOpcUa.Configuration;
|
using ZB.MOM.WW.OtOpcUa.Configuration;
|
||||||
|
using ZB.MOM.WW.OtOpcUa.Host.Configuration;
|
||||||
|
|
||||||
namespace ZB.MOM.WW.OtOpcUa.Host.Health;
|
namespace ZB.MOM.WW.OtOpcUa.Host.Health;
|
||||||
|
|
||||||
@@ -36,7 +41,20 @@ public static class HealthEndpoints
|
|||||||
"admin-leader",
|
"admin-leader",
|
||||||
failureStatus: null,
|
failureStatus: null,
|
||||||
tags: new[] { ZbHealthTags.Active },
|
tags: new[] { ZbHealthTags.Active },
|
||||||
args: "admin");
|
args: "admin")
|
||||||
|
// Registered unconditionally, not driver-gated. AddOtOpcUaHealth takes no role argument
|
||||||
|
// and runs on every node; the check itself resolves ISyncStatus optionally and reports
|
||||||
|
// Healthy when LocalDb is absent (admin-only graphs) or replication is default-OFF, so a
|
||||||
|
// plain node is never degraded by it. A factory registration keeps this no-arg signature
|
||||||
|
// while still reading ISyncStatus + options + the sync port from the container.
|
||||||
|
.Add(new HealthCheckRegistration(
|
||||||
|
"localdb-replication",
|
||||||
|
sp => new LocalDbReplicationHealthCheck(
|
||||||
|
sp.GetService<ISyncStatus>(),
|
||||||
|
sp.GetRequiredService<IOptions<ReplicationOptions>>(),
|
||||||
|
LocalDbRegistration.SyncListenPort(sp.GetRequiredService<IConfiguration>())),
|
||||||
|
failureStatus: null,
|
||||||
|
tags: new[] { ZbHealthTags.Active }));
|
||||||
return services;
|
return services;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,126 @@
|
|||||||
|
using Microsoft.Extensions.Diagnostics.HealthChecks;
|
||||||
|
using Microsoft.Extensions.Options;
|
||||||
|
using ZB.MOM.WW.LocalDb.Replication;
|
||||||
|
|
||||||
|
namespace ZB.MOM.WW.OtOpcUa.Host.Health;
|
||||||
|
|
||||||
|
/// <summary>
|
||||||
|
/// Pure decision function for the LocalDb replication probe, factored out of
|
||||||
|
/// <see cref="LocalDbReplicationHealthCheck"/> so the whole matrix is table-testable without
|
||||||
|
/// standing up a replication engine.
|
||||||
|
/// </summary>
|
||||||
|
public static class LocalDbReplicationDecision
|
||||||
|
{
|
||||||
|
/// <summary>
|
||||||
|
/// Maps the resolved replication facts to a health status.
|
||||||
|
/// </summary>
|
||||||
|
/// <param name="peerConfigured">
|
||||||
|
/// True when this node participates in replication at all — it has a peer address to dial, or
|
||||||
|
/// a sync listener bound. False means default-OFF, and default-OFF must never degrade a node.
|
||||||
|
/// </param>
|
||||||
|
/// <param name="connected">Whether a sync session is currently running.</param>
|
||||||
|
/// <param name="oplogBacklog">
|
||||||
|
/// The unacked oplog backlog, or <see langword="null"/> when it could not be read. Null is
|
||||||
|
/// deliberately not treated as zero: a failed poll is "unknown", not "caught up".
|
||||||
|
/// </param>
|
||||||
|
/// <param name="degradedThreshold">
|
||||||
|
/// Backlog at or above which a connected pair is judged to be falling behind.
|
||||||
|
/// </param>
|
||||||
|
/// <returns>The status plus a human-readable reason.</returns>
|
||||||
|
public static (HealthStatus Status, string Description) Evaluate(
|
||||||
|
bool peerConfigured, bool connected, long? oplogBacklog, long degradedThreshold)
|
||||||
|
{
|
||||||
|
if (!peerConfigured)
|
||||||
|
return (HealthStatus.Healthy, "LocalDb replication is not configured on this node (default-OFF).");
|
||||||
|
|
||||||
|
if (!connected)
|
||||||
|
return (HealthStatus.Degraded, "LocalDb replication is configured but no sync session is connected.");
|
||||||
|
|
||||||
|
if (oplogBacklog is null)
|
||||||
|
return (HealthStatus.Degraded, "LocalDb replication is connected but its oplog backlog is unknown (the poll failed).");
|
||||||
|
|
||||||
|
if (oplogBacklog.Value >= degradedThreshold)
|
||||||
|
return (HealthStatus.Degraded,
|
||||||
|
$"LocalDb replication is connected but its oplog backlog ({oplogBacklog.Value}) is at or above the degraded threshold ({degradedThreshold}).");
|
||||||
|
|
||||||
|
return (HealthStatus.Healthy, $"LocalDb replication is connected; oplog backlog {oplogBacklog.Value}.");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// <summary>
|
||||||
|
/// Reports the health of this node's LocalDb replication link. Registered unconditionally with
|
||||||
|
/// the shared health pipeline; when LocalDb is not present at all (admin-only graphs) it reports
|
||||||
|
/// <see cref="HealthStatus.Healthy"/> rather than degrading — matching the
|
||||||
|
/// <c>ActiveNodeHealthCheck</c> precedent of "not applicable ⇒ Healthy".
|
||||||
|
/// </summary>
|
||||||
|
/// <remarks>
|
||||||
|
/// A node running LocalDb with replication switched off (no peer, no listener) is also Healthy:
|
||||||
|
/// that is the default posture for most of the fleet, and it must not look degraded. Only a node
|
||||||
|
/// that is <i>meant</i> to be replicating and is not — or is connected but cannot confirm it is
|
||||||
|
/// draining — is surfaced as Degraded. It never reports Unhealthy: a replication problem does not
|
||||||
|
/// stop the node serving its address space, so it must not fail a readiness gate.
|
||||||
|
/// </remarks>
|
||||||
|
public sealed class LocalDbReplicationHealthCheck : IHealthCheck
|
||||||
|
{
|
||||||
|
/// <summary>
|
||||||
|
/// Backlog at or above which a connected pair is reported Degraded. Well below the library's
|
||||||
|
/// <c>MaxOplogRows</c> default (1,000,000, where a snapshot resync is forced), so the probe
|
||||||
|
/// flags a pair that is falling behind long before the engine itself intervenes.
|
||||||
|
/// </summary>
|
||||||
|
public const long DefaultBacklogDegradedThreshold = 100_000;
|
||||||
|
|
||||||
|
private readonly ISyncStatus? _syncStatus;
|
||||||
|
private readonly ReplicationOptions _options;
|
||||||
|
private readonly int _syncListenPort;
|
||||||
|
private readonly long _degradedThreshold;
|
||||||
|
|
||||||
|
/// <summary>Creates the health check.</summary>
|
||||||
|
/// <param name="syncStatus">
|
||||||
|
/// The replication status singleton, or <see langword="null"/> when the replication engine is
|
||||||
|
/// not registered (admin-only nodes) — in which case the check is a Healthy no-op.
|
||||||
|
/// </param>
|
||||||
|
/// <param name="options">The bound replication options (peer address, etc.).</param>
|
||||||
|
/// <param name="syncListenPort">
|
||||||
|
/// The configured sync listener port; a value greater than zero counts as "replication
|
||||||
|
/// configured" even when this node is the passive side with no peer address.
|
||||||
|
/// </param>
|
||||||
|
/// <param name="degradedThreshold">Backlog degraded threshold; defaults to <see cref="DefaultBacklogDegradedThreshold"/>.</param>
|
||||||
|
public LocalDbReplicationHealthCheck(
|
||||||
|
ISyncStatus? syncStatus,
|
||||||
|
IOptions<ReplicationOptions> options,
|
||||||
|
int syncListenPort,
|
||||||
|
long degradedThreshold = DefaultBacklogDegradedThreshold)
|
||||||
|
{
|
||||||
|
ArgumentNullException.ThrowIfNull(options);
|
||||||
|
|
||||||
|
_syncStatus = syncStatus;
|
||||||
|
_options = options.Value;
|
||||||
|
_syncListenPort = syncListenPort;
|
||||||
|
_degradedThreshold = degradedThreshold;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// <inheritdoc />
|
||||||
|
public Task<HealthCheckResult> CheckHealthAsync(
|
||||||
|
HealthCheckContext context, CancellationToken cancellationToken = default)
|
||||||
|
{
|
||||||
|
// No replication engine registered at all → not applicable → Healthy. Admin-only nodes and
|
||||||
|
// any graph that did not call AddOtOpcUaLocalDb land here.
|
||||||
|
if (_syncStatus is null)
|
||||||
|
return Task.FromResult(HealthCheckResult.Healthy(
|
||||||
|
"LocalDb replication is not present on this node."));
|
||||||
|
|
||||||
|
var peerConfigured =
|
||||||
|
!string.IsNullOrWhiteSpace(_options.PeerAddress) || _syncListenPort > 0;
|
||||||
|
|
||||||
|
var (status, description) = LocalDbReplicationDecision.Evaluate(
|
||||||
|
peerConfigured, _syncStatus.Connected, _syncStatus.OplogBacklog, _degradedThreshold);
|
||||||
|
|
||||||
|
var result = status switch
|
||||||
|
{
|
||||||
|
HealthStatus.Healthy => HealthCheckResult.Healthy(description),
|
||||||
|
_ => HealthCheckResult.Degraded(description),
|
||||||
|
};
|
||||||
|
|
||||||
|
return Task.FromResult(result);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,4 +1,5 @@
|
|||||||
using Microsoft.Extensions.Configuration;
|
using Microsoft.Extensions.Configuration;
|
||||||
|
using ZB.MOM.WW.LocalDb.Replication;
|
||||||
using ZB.MOM.WW.OtOpcUa.Commons.Observability;
|
using ZB.MOM.WW.OtOpcUa.Commons.Observability;
|
||||||
using ZB.MOM.WW.Telemetry;
|
using ZB.MOM.WW.Telemetry;
|
||||||
|
|
||||||
@@ -27,7 +28,11 @@ public static class ObservabilityExtensions
|
|||||||
return services.AddZbTelemetry(o =>
|
return services.AddZbTelemetry(o =>
|
||||||
{
|
{
|
||||||
o.ServiceName = "otopcua";
|
o.ServiceName = "otopcua";
|
||||||
o.Meters = [OtOpcUaTelemetry.MeterName];
|
// o.Meters is a STRICT allowlist — ZbTelemetry adds only the named meters, with no
|
||||||
|
// wildcard. The LocalDb replication engine publishes under its own meter name, so
|
||||||
|
// without this entry its localdb.sync.* / localdb.oplog.depth series are silently absent
|
||||||
|
// from /metrics on driver nodes (the exact omission a ScadaBridge live gate caught).
|
||||||
|
o.Meters = [OtOpcUaTelemetry.MeterName, LocalDbMetrics.MeterName];
|
||||||
o.ActivitySources = [OtOpcUaTelemetry.ActivitySourceName];
|
o.ActivitySources = [OtOpcUaTelemetry.ActivitySourceName];
|
||||||
if (Enum.TryParse<ZbExporter>(configuration["OtOpcUa:Telemetry:Exporter"], ignoreCase: true, out var exporter))
|
if (Enum.TryParse<ZbExporter>(configuration["OtOpcUa:Telemetry:Exporter"], ignoreCase: true, out var exporter))
|
||||||
o.Exporter = exporter;
|
o.Exporter = exporter;
|
||||||
|
|||||||
+122
@@ -0,0 +1,122 @@
|
|||||||
|
using Microsoft.Extensions.Diagnostics.HealthChecks;
|
||||||
|
using Microsoft.Extensions.Options;
|
||||||
|
using Shouldly;
|
||||||
|
using Xunit;
|
||||||
|
using ZB.MOM.WW.LocalDb.Replication;
|
||||||
|
using ZB.MOM.WW.OtOpcUa.Host.Health;
|
||||||
|
|
||||||
|
namespace ZB.MOM.WW.OtOpcUa.Host.IntegrationTests;
|
||||||
|
|
||||||
|
/// <summary>
|
||||||
|
/// LocalDb Phase 1 (Task 10) — the replication health check.
|
||||||
|
/// </summary>
|
||||||
|
/// <remarks>
|
||||||
|
/// The load-bearing case is the default-OFF one: a plain driver node with no peer and no
|
||||||
|
/// listener must report <b>Healthy</b>, or every unreplicated node in the fleet degrades the
|
||||||
|
/// moment this check ships. Beyond that, "connected" is not sufficient for healthy — an unknown
|
||||||
|
/// backlog (a failed oplog poll) must not read as zero.
|
||||||
|
/// </remarks>
|
||||||
|
public sealed class LocalDbReplicationHealthCheckTests
|
||||||
|
{
|
||||||
|
// ---- Pure decision core (table-tested) --------------------------------------------------
|
||||||
|
|
||||||
|
[Fact]
|
||||||
|
public void Unconfigured_IsHealthy_SoDefaultOffNodesDoNotDegrade()
|
||||||
|
{
|
||||||
|
LocalDbReplicationDecision.Evaluate(
|
||||||
|
peerConfigured: false, connected: false, oplogBacklog: null, degradedThreshold: 100_000)
|
||||||
|
.Status.ShouldBe(HealthStatus.Healthy);
|
||||||
|
}
|
||||||
|
|
||||||
|
[Fact]
|
||||||
|
public void ConfiguredButNotConnected_IsDegraded()
|
||||||
|
{
|
||||||
|
LocalDbReplicationDecision.Evaluate(
|
||||||
|
peerConfigured: true, connected: false, oplogBacklog: null, degradedThreshold: 100_000)
|
||||||
|
.Status.ShouldBe(HealthStatus.Degraded);
|
||||||
|
}
|
||||||
|
|
||||||
|
[Fact]
|
||||||
|
public void ConnectedButBacklogUnknown_IsDegraded()
|
||||||
|
{
|
||||||
|
// null backlog = the oplog poll failed. Reporting Healthy here would call a pair that cannot
|
||||||
|
// read its own oplog perfectly fine.
|
||||||
|
LocalDbReplicationDecision.Evaluate(
|
||||||
|
peerConfigured: true, connected: true, oplogBacklog: null, degradedThreshold: 100_000)
|
||||||
|
.Status.ShouldBe(HealthStatus.Degraded);
|
||||||
|
}
|
||||||
|
|
||||||
|
[Fact]
|
||||||
|
public void ConnectedWithSmallBacklog_IsHealthy()
|
||||||
|
{
|
||||||
|
LocalDbReplicationDecision.Evaluate(
|
||||||
|
peerConfigured: true, connected: true, oplogBacklog: 12, degradedThreshold: 100_000)
|
||||||
|
.Status.ShouldBe(HealthStatus.Healthy);
|
||||||
|
}
|
||||||
|
|
||||||
|
[Fact]
|
||||||
|
public void ConnectedWithBacklogOverThreshold_IsDegraded()
|
||||||
|
{
|
||||||
|
// A backlog past the threshold means the peer is falling behind faster than it drains —
|
||||||
|
// connected, but not keeping up.
|
||||||
|
LocalDbReplicationDecision.Evaluate(
|
||||||
|
peerConfigured: true, connected: true, oplogBacklog: 100_001, degradedThreshold: 100_000)
|
||||||
|
.Status.ShouldBe(HealthStatus.Degraded);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- IHealthCheck wrapper ---------------------------------------------------------------
|
||||||
|
|
||||||
|
[Fact]
|
||||||
|
public async Task Wrapper_WithNoSyncStatusRegistered_IsHealthy()
|
||||||
|
{
|
||||||
|
// Admin-only graphs never register the replication engine. The check is registered
|
||||||
|
// unconditionally (AddOtOpcUaHealth takes no role argument), so it must no-op to Healthy
|
||||||
|
// rather than throw or degrade when LocalDb is simply absent.
|
||||||
|
var check = new LocalDbReplicationHealthCheck(
|
||||||
|
syncStatus: null,
|
||||||
|
options: Options.Create(new ReplicationOptions()),
|
||||||
|
syncListenPort: 0);
|
||||||
|
|
||||||
|
var result = await check.CheckHealthAsync(new HealthCheckContext(), TestContext.Current.CancellationToken);
|
||||||
|
|
||||||
|
result.Status.ShouldBe(HealthStatus.Healthy);
|
||||||
|
}
|
||||||
|
|
||||||
|
[Fact]
|
||||||
|
public async Task Wrapper_ListenerConfiguredButNotConnected_IsDegraded()
|
||||||
|
{
|
||||||
|
// A passive node (no PeerAddress) still counts as "replication configured" when it is
|
||||||
|
// listening — otherwise a passive node that never gets dialled would look healthy while
|
||||||
|
// silently accepting nothing.
|
||||||
|
var check = new LocalDbReplicationHealthCheck(
|
||||||
|
syncStatus: new StubSyncStatus { Connected = false },
|
||||||
|
options: Options.Create(new ReplicationOptions { PeerAddress = "" }),
|
||||||
|
syncListenPort: 9001);
|
||||||
|
|
||||||
|
var result = await check.CheckHealthAsync(new HealthCheckContext(), TestContext.Current.CancellationToken);
|
||||||
|
|
||||||
|
result.Status.ShouldBe(HealthStatus.Degraded);
|
||||||
|
}
|
||||||
|
|
||||||
|
[Fact]
|
||||||
|
public async Task Wrapper_PeerConfiguredAndConnectedAndDraining_IsHealthy()
|
||||||
|
{
|
||||||
|
var check = new LocalDbReplicationHealthCheck(
|
||||||
|
syncStatus: new StubSyncStatus { Connected = true, OplogBacklog = 3 },
|
||||||
|
options: Options.Create(new ReplicationOptions { PeerAddress = "https://peer:9001" }),
|
||||||
|
syncListenPort: 9001);
|
||||||
|
|
||||||
|
var result = await check.CheckHealthAsync(new HealthCheckContext(), TestContext.Current.CancellationToken);
|
||||||
|
|
||||||
|
result.Status.ShouldBe(HealthStatus.Healthy);
|
||||||
|
}
|
||||||
|
|
||||||
|
private sealed class StubSyncStatus : ISyncStatus
|
||||||
|
{
|
||||||
|
public bool Connected { get; init; }
|
||||||
|
public string? PeerNodeId { get; init; }
|
||||||
|
public DateTimeOffset? LastSyncUtc { get; init; }
|
||||||
|
public long? OplogBacklog { get; init; }
|
||||||
|
public long ConnectionAttempts { get; init; }
|
||||||
|
}
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user