feat(health): serve MapZbHealth on site nodes
Site nodes served no health surface at all — gRPC on the HTTP/2-only listener and
/metrics on the HTTP/1.1 one — so nothing outside the cluster could ask a site
node whether it was ready or which half of the pair was active. The family
overview dashboard probes every instance the same way, and this is the one gap.
Three checks, registered in SiteServiceRegistration.Configure (not Program.cs, so
the composition-root tests that build this graph actually cover them) and mapped
by app.MapZbHealth() on the site's HTTP/1.1 listener (default :8084) alongside
/metrics:
akka-cluster [Ready] the shared AkkaClusterHealthCheck. Also carries the
cluster-view data (leader/memberCount/...) the dashboard
reads, free with ZB.MOM.WW.Health 0.2.0.
localdb [Ready] NEW SiteLocalDbHealthCheck — SELECT 1 through the
registered ILocalDb. Site has no EF context; central's
DatabaseHealthCheck<ScadaBridgeDbContext> is central-only.
Replication state rides along as data ENRICHMENT only:
replication is default-OFF, so failing on it would mark
every correctly-configured node unready.
active-node [Active] NEW SitePairActiveNodeHealthCheck, delegating to
IClusterNodeProvider.SelfIsPrimary. Deliberately NOT
central's OldestNodeActiveHealthCheck: that one calls
SelfIsOldest(cluster) with no role argument and would
compute "oldest" across the wrong member set on a mesh
carrying more than one site.
Anonymous, as central's are: the site pipeline runs no authentication middleware
and has no FallbackPolicy, so nothing extra was needed.
Also bumps ZB.MOM.WW.Health* 0.1.0 -> 0.2.0 (central gains data.leader for free).
Tests: SiteHealthCheckTests builds the REAL site container and activates every
registration through its factory (exact-set names, one tier tag each, resolved
types, central-only checks absent behind a positive control) plus behaviour for
both new checks. SiteHealthEndpointTests boots the real site Program over
WebApplicationFactory and proves the endpoints are MAPPED — without it, deleting
MapZbHealth would leave every registration test green while site nodes 404'd.
Prereq for scadaproj docs/plans/2026-07-22-overview-dashboard-impl-plan.md Phase 1.
This commit is contained in:
@@ -90,9 +90,9 @@
|
||||
to mark tests as Skipped (not silently Passed) when MSSQL is unreachable.
|
||||
-->
|
||||
<PackageVersion Include="Xunit.SkippableFact" Version="1.5.61" />
|
||||
<PackageVersion Include="ZB.MOM.WW.Health" Version="0.1.0" />
|
||||
<PackageVersion Include="ZB.MOM.WW.Health.Akka" Version="0.1.0" />
|
||||
<PackageVersion Include="ZB.MOM.WW.Health.EntityFrameworkCore" Version="0.1.0" />
|
||||
<PackageVersion Include="ZB.MOM.WW.Health" Version="0.2.0" />
|
||||
<PackageVersion Include="ZB.MOM.WW.Health.Akka" Version="0.2.0" />
|
||||
<PackageVersion Include="ZB.MOM.WW.Health.EntityFrameworkCore" Version="0.2.0" />
|
||||
<PackageVersion Include="ZB.MOM.WW.Telemetry" Version="0.1.0" />
|
||||
<PackageVersion Include="ZB.MOM.WW.Telemetry.Serilog" Version="0.1.0" />
|
||||
<PackageVersion Include="ZB.MOM.WW.MxGateway.Client" Version="0.1.1" />
|
||||
|
||||
@@ -0,0 +1,77 @@
|
||||
using Microsoft.Extensions.Diagnostics.HealthChecks;
|
||||
using ZB.MOM.WW.LocalDb;
|
||||
using ZB.MOM.WW.LocalDb.Replication;
|
||||
|
||||
namespace ZB.MOM.WW.ScadaBridge.Host.Health;
|
||||
|
||||
/// <summary>
|
||||
/// /health/ready check for the consolidated site database (LocalDb). Site nodes have no EF
|
||||
/// context — central's <c>DatabaseHealthCheck<ScadaBridgeDbContext></c> is central-only — so
|
||||
/// the site store is probed directly through the registered <see cref="ILocalDb"/> with a trivial
|
||||
/// <c>SELECT 1</c>. That is the honest readiness question for a site node: everything it persists
|
||||
/// (OperationTracking, site events, store-and-forward, the replicated config tables) lives here.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Replication state is reported as <c>data</c> ENRICHMENT ONLY and never affects the status.
|
||||
/// Replication is default-OFF by design, so a node with no peer reports <c>connected=false</c> with
|
||||
/// a real 0 backlog — the healthy steady state of an unreplicated node — and failing the check on
|
||||
/// it would mark every correctly-configured single node unready.
|
||||
/// </remarks>
|
||||
public sealed class SiteLocalDbHealthCheck : IHealthCheck
|
||||
{
|
||||
private readonly ILocalDb _localDb;
|
||||
private readonly ISyncStatus? _syncStatus;
|
||||
|
||||
/// <summary>Initializes a new <see cref="SiteLocalDbHealthCheck"/>.</summary>
|
||||
/// <param name="localDb">The consolidated site database registered by <c>AddZbLocalDb</c>.</param>
|
||||
/// <param name="syncStatus">
|
||||
/// The replication status view, when the replication engine is registered. Optional so the check
|
||||
/// still works in a composition that wires storage without replication.
|
||||
/// </param>
|
||||
public SiteLocalDbHealthCheck(ILocalDb localDb, ISyncStatus? syncStatus = null)
|
||||
{
|
||||
_localDb = localDb ?? throw new ArgumentNullException(nameof(localDb));
|
||||
_syncStatus = syncStatus;
|
||||
}
|
||||
|
||||
/// <summary>Probes the site database and reports its reachability.</summary>
|
||||
/// <param name="context">The health check context (unused; the verdict depends only on the probe).</param>
|
||||
/// <param name="cancellationToken">Cancellation token.</param>
|
||||
/// <returns>Healthy when the probe query succeeds; Unhealthy with the fault otherwise.</returns>
|
||||
public async Task<HealthCheckResult> CheckHealthAsync(
|
||||
HealthCheckContext context,
|
||||
CancellationToken cancellationToken = default)
|
||||
{
|
||||
try
|
||||
{
|
||||
await _localDb.QueryAsync("SELECT 1;", static r => r.GetInt32(0), ct: cancellationToken)
|
||||
.ConfigureAwait(false);
|
||||
}
|
||||
catch (Exception ex) when (ex is not OperationCanceledException)
|
||||
{
|
||||
return HealthCheckResult.Unhealthy("Site LocalDb is not reachable.", ex);
|
||||
}
|
||||
|
||||
return HealthCheckResult.Healthy("Site LocalDb reachable.", BuildData());
|
||||
}
|
||||
|
||||
private IReadOnlyDictionary<string, object> BuildData()
|
||||
{
|
||||
var data = new Dictionary<string, object>(StringComparer.Ordinal);
|
||||
if (_syncStatus is null)
|
||||
return data;
|
||||
|
||||
data["replicationConnected"] = _syncStatus.Connected;
|
||||
|
||||
// Nullable on purpose upstream: null means "backlog unknown" (no provider wired or the poll
|
||||
// failed), which must not be flattened to 0 — that would render a pair which cannot read its
|
||||
// own oplog as perfectly caught up. Omit the key instead.
|
||||
if (_syncStatus.OplogBacklog is { } backlog)
|
||||
data["replicationOplogBacklog"] = backlog;
|
||||
|
||||
if (_syncStatus.PeerNodeId is { } peer)
|
||||
data["replicationPeerNodeId"] = peer;
|
||||
|
||||
return data;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,39 @@
|
||||
using Microsoft.Extensions.Diagnostics.HealthChecks;
|
||||
using ZB.MOM.WW.ScadaBridge.HealthMonitoring;
|
||||
|
||||
namespace ZB.MOM.WW.ScadaBridge.Host.Health;
|
||||
|
||||
/// <summary>
|
||||
/// /health/active check for a SITE pair: Healthy only on the pair's primary — the oldest Up member
|
||||
/// carrying the <c>site-{SiteId}</c> role. Same active/standby semantics central publishes, so a
|
||||
/// consumer reads one contract fleet-wide (200 = Active, 503 = Standby).
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Deliberately NOT central's <see cref="OldestNodeActiveHealthCheck"/>: that one calls
|
||||
/// <c>ClusterActivityEvaluator.SelfIsOldest(cluster)</c> with no role argument, so on a mesh that
|
||||
/// carries more than one site's members it computes "oldest" across the wrong member set and a site
|
||||
/// node could report Active because it happens to be the oldest member overall. Delegating to
|
||||
/// <see cref="IClusterNodeProvider.SelfIsPrimary"/> reuses the role-scoped evaluation the site
|
||||
/// already trusts for singleton placement and event-log purge — one primary definition per site,
|
||||
/// not two that can disagree.
|
||||
/// </remarks>
|
||||
public sealed class SitePairActiveNodeHealthCheck : IHealthCheck
|
||||
{
|
||||
private readonly IClusterNodeProvider _nodeProvider;
|
||||
|
||||
/// <summary>Initializes a new <see cref="SitePairActiveNodeHealthCheck"/>.</summary>
|
||||
/// <param name="nodeProvider">The role-scoped site cluster node provider.</param>
|
||||
public SitePairActiveNodeHealthCheck(IClusterNodeProvider nodeProvider) =>
|
||||
_nodeProvider = nodeProvider ?? throw new ArgumentNullException(nameof(nodeProvider));
|
||||
|
||||
/// <summary>Reports Healthy on the site pair's primary and Unhealthy on the standby.</summary>
|
||||
/// <param name="context">The health check context (unused; the verdict depends only on cluster membership).</param>
|
||||
/// <param name="cancellationToken">Cancellation token.</param>
|
||||
/// <returns>Healthy on the primary; Unhealthy (503 = Standby) otherwise.</returns>
|
||||
public Task<HealthCheckResult> CheckHealthAsync(
|
||||
HealthCheckContext context,
|
||||
CancellationToken cancellationToken = default) =>
|
||||
Task.FromResult(_nodeProvider.SelfIsPrimary
|
||||
? HealthCheckResult.Healthy("site pair primary (oldest Up member of the site role)")
|
||||
: HealthCheckResult.Unhealthy("not the site pair primary (standby)"));
|
||||
}
|
||||
@@ -634,6 +634,20 @@ try
|
||||
// must be present before the Map* calls on the site role.
|
||||
app.UseRouting();
|
||||
|
||||
// Map the canonical three-tier health endpoints on the site node:
|
||||
// /health/ready — akka-cluster (site-pair membership, and the cluster-view data the
|
||||
// family overview dashboard reads) + localdb (the consolidated site
|
||||
// store — the only database a site node has).
|
||||
// /health/active — active-node = this site pair's primary; 200 on the primary, 503 on
|
||||
// the standby, the same contract central publishes.
|
||||
// /healthz — bare process liveness.
|
||||
// Checks are registered in SiteServiceRegistration.Configure. All three endpoints are
|
||||
// anonymous and use the canonical ZbHealthWriter JSON; the site pipeline runs no
|
||||
// authentication middleware and has no FallbackPolicy, so nothing extra is needed to keep
|
||||
// them reachable. They are served on the HTTP/1.1 listener (metricsPort, default :8084)
|
||||
// alongside /metrics — the gRPC listener is HTTP/2-only and unusable by a plain HTTP client.
|
||||
app.MapZbHealth();
|
||||
|
||||
// Observability — mount the always-on Prometheus /metrics scrape endpoint.
|
||||
// AddZbTelemetry (in SiteServiceRegistration.Configure → BindSharedOptions)
|
||||
// wires the OTel Resource + standard instrumentation + Prometheus exporter;
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
using Microsoft.Extensions.DependencyInjection.Extensions;
|
||||
using Microsoft.Extensions.Options;
|
||||
using ZB.MOM.WW.Health;
|
||||
using ZB.MOM.WW.Health.Akka;
|
||||
using ZB.MOM.WW.ScadaBridge.AuditLog;
|
||||
using ZB.MOM.WW.ScadaBridge.ClusterInfrastructure;
|
||||
using ZB.MOM.WW.ScadaBridge.Communication;
|
||||
@@ -190,6 +192,39 @@ public static class SiteServiceRegistration
|
||||
return () => nodeProvider.SelfIsPrimary;
|
||||
});
|
||||
|
||||
// Health checks — the shared ZB.MOM.WW.Health probes, mapped by MapZbHealth on the site's
|
||||
// HTTP/1.1 listener (Program.cs). Site nodes served no health endpoints before this; the
|
||||
// family overview dashboard probes every node the same way, so a site node has to answer
|
||||
// the same two questions central does.
|
||||
//
|
||||
// Ready (/health/ready) akka-cluster — this node's site-pair membership. It also carries
|
||||
// the cluster-view data (leader/memberCount/...) that the
|
||||
// dashboard reads, free with ZB.MOM.WW.Health 0.2.0.
|
||||
// localdb — the consolidated site store, which is the only
|
||||
// database a site node has (central's EF DatabaseHealthCheck is
|
||||
// central-only and does not apply here).
|
||||
// Active (/health/active) active-node — site-pair primary vs standby.
|
||||
//
|
||||
// Registered HERE rather than in Program.cs, like the LocalDb registrations above, so the
|
||||
// composition-root tests that build this graph actually cover them. The Akka check resolves
|
||||
// ActorSystem from DI per probe via the singleton bridge registered above.
|
||||
services.AddHealthChecks()
|
||||
.AddTypeActivatedCheck<AkkaClusterHealthCheck>(
|
||||
"akka-cluster",
|
||||
failureStatus: null,
|
||||
tags: new[] { ZbHealthTags.Ready },
|
||||
args: AkkaClusterStatusPolicy.Default)
|
||||
.AddTypeActivatedCheck<SiteLocalDbHealthCheck>(
|
||||
"localdb",
|
||||
failureStatus: null,
|
||||
tags: new[] { ZbHealthTags.Ready })
|
||||
// NOT central's OldestNodeActiveHealthCheck — that one is role-less; see
|
||||
// SitePairActiveNodeHealthCheck for why a site pair needs the role-scoped answer.
|
||||
.AddTypeActivatedCheck<SitePairActiveNodeHealthCheck>(
|
||||
"active-node",
|
||||
failureStatus: null,
|
||||
tags: new[] { ZbHealthTags.Active });
|
||||
|
||||
// Options binding
|
||||
BindSharedOptions(services, config);
|
||||
|
||||
|
||||
@@ -0,0 +1,235 @@
|
||||
using Microsoft.AspNetCore.Builder;
|
||||
using Microsoft.Extensions.Configuration;
|
||||
using Microsoft.Extensions.DependencyInjection;
|
||||
using Microsoft.Extensions.Diagnostics.HealthChecks;
|
||||
using Microsoft.Extensions.Options;
|
||||
using ZB.MOM.WW.Health;
|
||||
using ZB.MOM.WW.Health.Akka;
|
||||
using ZB.MOM.WW.LocalDb;
|
||||
using ZB.MOM.WW.LocalDb.Registration;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Health;
|
||||
using ZB.MOM.WW.ScadaBridge.HealthMonitoring;
|
||||
using ZB.MOM.WW.ScadaBridge.Host.Health;
|
||||
|
||||
namespace ZB.MOM.WW.ScadaBridge.Host.Tests;
|
||||
|
||||
/// <summary>
|
||||
/// Site-node health endpoints. Site nodes previously served no health surface at all — only gRPC on
|
||||
/// the HTTP/2 listener and <c>/metrics</c> on the HTTP/1.1 one — so the family overview dashboard
|
||||
/// had no uniform way to probe them. These tests pin the three checks the site now registers and
|
||||
/// the tier each one lands in.
|
||||
/// <para>
|
||||
/// The registration assertions build the REAL container from
|
||||
/// <see cref="SiteServiceRegistration.Configure"/> rather than a hand-assembled
|
||||
/// <c>ServiceCollection</c>, for the reason spelled out in <see cref="SiteLocalDbWiringTests"/>:
|
||||
/// this family's wiring defects have consistently been ones that only a container-built graph
|
||||
/// catches. Each registration is additionally ACTIVATED through its factory, so a check that is
|
||||
/// registered but cannot resolve its dependencies fails here instead of at the first live probe.
|
||||
/// </para>
|
||||
/// </summary>
|
||||
public class SiteHealthCheckTests : IDisposable
|
||||
{
|
||||
private readonly WebApplication _host;
|
||||
private readonly string _tempDbPath;
|
||||
private readonly string _tempSiteDbPath;
|
||||
private readonly string _tempTrackingDbPath;
|
||||
|
||||
public SiteHealthCheckTests()
|
||||
{
|
||||
var stamp = Guid.NewGuid();
|
||||
_tempDbPath = Path.Combine(Path.GetTempPath(), $"scadabridge_health_localdb_{stamp}.db");
|
||||
_tempSiteDbPath = Path.Combine(Path.GetTempPath(), $"scadabridge_health_site_{stamp}.db");
|
||||
_tempTrackingDbPath = Path.Combine(Path.GetTempPath(), $"scadabridge_health_track_{stamp}.db");
|
||||
|
||||
var builder = WebApplication.CreateBuilder();
|
||||
builder.Configuration.Sources.Clear();
|
||||
builder.Configuration.AddInMemoryCollection(new Dictionary<string, string?>
|
||||
{
|
||||
["ScadaBridge:Node:Role"] = "Site",
|
||||
["ScadaBridge:Node:NodeName"] = "node-a",
|
||||
["ScadaBridge:Node:NodeHostname"] = "test-site",
|
||||
["ScadaBridge:Node:SiteId"] = "TestSite",
|
||||
["ScadaBridge:Node:RemotingPort"] = "0",
|
||||
["ScadaBridge:Node:GrpcPort"] = "0",
|
||||
["ScadaBridge:Database:SiteDbPath"] = _tempSiteDbPath,
|
||||
["ScadaBridge:OperationTracking:ConnectionString"] = $"Data Source={_tempTrackingDbPath}",
|
||||
["ScadaBridge:Cluster:SeedNodes:0"] = "akka.tcp://scadabridge@localhost:2551",
|
||||
["ScadaBridge:Cluster:SeedNodes:1"] = "akka.tcp://scadabridge@localhost:2552",
|
||||
["LocalDb:Path"] = _tempDbPath,
|
||||
});
|
||||
|
||||
builder.Services.AddGrpc();
|
||||
builder.Services.AddSingleton<ZB.MOM.WW.ScadaBridge.Communication.Grpc.SiteStreamGrpcServer>();
|
||||
|
||||
SiteServiceRegistration.Configure(builder.Services, builder.Configuration);
|
||||
|
||||
AkkaHostedServiceRemover.RemoveAkkaHostedServiceOnly(builder.Services);
|
||||
|
||||
_host = builder.Build();
|
||||
}
|
||||
|
||||
public void Dispose()
|
||||
{
|
||||
(_host as IDisposable)?.Dispose();
|
||||
foreach (var path in new[] { _tempDbPath, _tempSiteDbPath, _tempTrackingDbPath })
|
||||
{
|
||||
try { File.Delete(path); } catch { /* best effort */ }
|
||||
}
|
||||
|
||||
GC.SuppressFinalize(this);
|
||||
}
|
||||
|
||||
private IReadOnlyList<HealthCheckRegistration> Registrations =>
|
||||
_host.Services.GetRequiredService<IOptions<HealthCheckServiceOptions>>().Value.Registrations.ToList();
|
||||
|
||||
[Fact]
|
||||
public void Site_RegistersExactlyTheThreeSiteChecks()
|
||||
{
|
||||
// Exact-set, not "contains": an extra check silently changes what /health/ready means for
|
||||
// every consumer, and a site node inheriting a central-only probe is precisely the mistake
|
||||
// this asserts against.
|
||||
Assert.Equal(
|
||||
new[] { "active-node", "akka-cluster", "localdb" },
|
||||
Registrations.Select(r => r.Name).OrderBy(n => n, StringComparer.Ordinal).ToArray());
|
||||
}
|
||||
|
||||
[Theory]
|
||||
[InlineData("akka-cluster", ZbHealthTags.Ready)]
|
||||
[InlineData("localdb", ZbHealthTags.Ready)]
|
||||
[InlineData("active-node", ZbHealthTags.Active)]
|
||||
public void Site_ChecksCarryTheirIntendedTier(string name, string expectedTag)
|
||||
{
|
||||
var registration = Assert.Single(Registrations, r => r.Name == name);
|
||||
|
||||
// Exactly one tier tag each: MapZbHealth selects by tag, so a check tagged both Ready and
|
||||
// Active would put standby-ness into readiness and drop the standby out of readiness too.
|
||||
Assert.Equal(new[] { expectedTag }, registration.Tags.ToArray());
|
||||
}
|
||||
|
||||
[Theory]
|
||||
[InlineData("akka-cluster", typeof(AkkaClusterHealthCheck))]
|
||||
[InlineData("localdb", typeof(SiteLocalDbHealthCheck))]
|
||||
[InlineData("active-node", typeof(SitePairActiveNodeHealthCheck))]
|
||||
public void Site_ChecksActivateToTheIntendedType(string name, Type expected)
|
||||
{
|
||||
var registration = Assert.Single(Registrations, r => r.Name == name);
|
||||
|
||||
// Running the factory is the point: a type-activated check whose constructor dependencies
|
||||
// are not registered on site would throw here rather than on the first probe in production.
|
||||
var check = registration.Factory(_host.Services);
|
||||
|
||||
Assert.IsType(expected, check);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void Site_ActiveNodeCheck_IsNotCentralsRoleLessOldestCheck()
|
||||
{
|
||||
// Central's OldestNodeActiveHealthCheck calls SelfIsOldest(cluster) with no role argument.
|
||||
// On a mesh carrying more than one site's members that computes "oldest" across the wrong
|
||||
// member set, so a site node could report Active purely for being the oldest member overall.
|
||||
var check = Assert.Single(Registrations, r => r.Name == "active-node").Factory(_host.Services);
|
||||
|
||||
Assert.IsNotType<OldestNodeActiveHealthCheck>(check);
|
||||
}
|
||||
|
||||
[Theory]
|
||||
[InlineData("database")]
|
||||
[InlineData("required-singletons")]
|
||||
public void Site_DoesNotRegisterCentralOnlyChecks(string centralOnlyName)
|
||||
{
|
||||
// Positive control first: if the harness ever stopped registering health checks at all,
|
||||
// every "is absent" assertion below would pass vacuously and this test would be worthless.
|
||||
Assert.Contains(Registrations, r => r.Name == "akka-cluster");
|
||||
|
||||
Assert.DoesNotContain(Registrations, r => r.Name == centralOnlyName);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task Site_LocalDbCheck_ReportsHealthyAgainstTheRealSiteDatabase()
|
||||
{
|
||||
// Behaviour, not wiring: the probe runs against the consolidated site database this
|
||||
// composition actually built, so a check that resolves but cannot query fails here.
|
||||
var check = (SiteLocalDbHealthCheck)Assert.Single(Registrations, r => r.Name == "localdb")
|
||||
.Factory(_host.Services);
|
||||
|
||||
var result = await check.CheckHealthAsync(NewContext("localdb", check));
|
||||
|
||||
Assert.Equal(HealthStatus.Healthy, result.Status);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task Site_LocalDbCheck_EnrichesWithReplicationStateWithoutFailingOnDefaultOff()
|
||||
{
|
||||
// Replication ships default-OFF, so the unreplicated steady state must stay Healthy while
|
||||
// still surfacing the state as data — enrichment must never become a verdict.
|
||||
var check = (SiteLocalDbHealthCheck)Assert.Single(Registrations, r => r.Name == "localdb")
|
||||
.Factory(_host.Services);
|
||||
|
||||
var result = await check.CheckHealthAsync(NewContext("localdb", check));
|
||||
|
||||
Assert.Equal(HealthStatus.Healthy, result.Status);
|
||||
Assert.False(Assert.IsType<bool>(result.Data["replicationConnected"]));
|
||||
Assert.Equal(0L, Assert.IsType<long>(result.Data["replicationOplogBacklog"]));
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task Site_LocalDbCheck_ReportsUnhealthyWhenTheStoreIsUnreachable()
|
||||
{
|
||||
var check = new SiteLocalDbHealthCheck(new ThrowingLocalDb());
|
||||
|
||||
var result = await check.CheckHealthAsync(NewContext("localdb", check));
|
||||
|
||||
Assert.Equal(HealthStatus.Unhealthy, result.Status);
|
||||
}
|
||||
|
||||
[Theory]
|
||||
[InlineData(true, HealthStatus.Healthy)]
|
||||
[InlineData(false, HealthStatus.Unhealthy)]
|
||||
public async Task Site_ActiveNodeCheck_SplitsPrimaryFromStandby(bool selfIsPrimary, HealthStatus expected)
|
||||
{
|
||||
// 200 on the primary / 503 on the standby is the same contract central publishes, so a
|
||||
// consumer reads one active-node meaning fleet-wide.
|
||||
var check = new SitePairActiveNodeHealthCheck(new StubNodeProvider(selfIsPrimary));
|
||||
|
||||
var result = await check.CheckHealthAsync(NewContext("active-node", check));
|
||||
|
||||
Assert.Equal(expected, result.Status);
|
||||
}
|
||||
|
||||
private static HealthCheckContext NewContext(string name, IHealthCheck check) => new()
|
||||
{
|
||||
Registration = new HealthCheckRegistration(name, check, HealthStatus.Unhealthy, tags: null),
|
||||
};
|
||||
|
||||
private sealed class StubNodeProvider(bool selfIsPrimary) : IClusterNodeProvider
|
||||
{
|
||||
public bool SelfIsPrimary { get; } = selfIsPrimary;
|
||||
|
||||
public IReadOnlyList<NodeStatus> GetClusterNodes() => [];
|
||||
}
|
||||
|
||||
private sealed class ThrowingLocalDb : ILocalDb
|
||||
{
|
||||
public Task<int> ExecuteAsync(string sql, object? parameters = null, CancellationToken ct = default) =>
|
||||
throw new InvalidOperationException("store unreachable");
|
||||
|
||||
public Task<IReadOnlyList<T>> QueryAsync<T>(
|
||||
string sql,
|
||||
Func<Microsoft.Data.Sqlite.SqliteDataReader, T> map,
|
||||
object? parameters = null,
|
||||
CancellationToken ct = default) =>
|
||||
throw new InvalidOperationException("store unreachable");
|
||||
|
||||
public Task<ILocalDbTransaction> BeginTransactionAsync(CancellationToken ct = default) =>
|
||||
throw new InvalidOperationException("store unreachable");
|
||||
|
||||
public Microsoft.Data.Sqlite.SqliteConnection CreateConnection() =>
|
||||
throw new InvalidOperationException("store unreachable");
|
||||
|
||||
public ReplicatedTable RegisterReplicated(string tableName) =>
|
||||
throw new InvalidOperationException("store unreachable");
|
||||
|
||||
public IReadOnlyDictionary<string, ReplicatedTable> ReplicatedTables =>
|
||||
throw new InvalidOperationException("store unreachable");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,156 @@
|
||||
using System.Net;
|
||||
using System.Text.Json;
|
||||
using Microsoft.AspNetCore.Hosting;
|
||||
using Microsoft.AspNetCore.Mvc.Testing;
|
||||
using Microsoft.AspNetCore.TestHost;
|
||||
using Microsoft.Extensions.DependencyInjection;
|
||||
|
||||
namespace ZB.MOM.WW.ScadaBridge.Host.Tests;
|
||||
|
||||
/// <summary>
|
||||
/// Proves the site pipeline actually MAPS the health endpoints — the half
|
||||
/// <see cref="SiteHealthCheckTests"/> cannot cover, since that fixture builds the site container
|
||||
/// from <see cref="SiteServiceRegistration"/> and never runs <c>Program</c>'s endpoint wiring.
|
||||
/// Without this, deleting <c>app.MapZbHealth()</c> from the site branch would leave every
|
||||
/// registration test green while site nodes served 404 to the overview dashboard.
|
||||
/// <para>
|
||||
/// Boots the real <c>Program</c> in the Site role. <c>Program</c> builds its own
|
||||
/// <c>ConfigurationBuilder</c> before the web host exists, so its role and settings come from
|
||||
/// PROCESS-WIDE environment variables (<c>SCADABRIDGE_CONFIG</c> selects appsettings.Site.json)
|
||||
/// rather than from <c>WithWebHostBuilder</c> — which is why this fixture joins
|
||||
/// <see cref="HostBootCollection"/> and restores every var it sets.
|
||||
/// </para>
|
||||
/// </summary>
|
||||
[Collection(HostBootCollection.Name)]
|
||||
public class SiteHealthEndpointTests : IDisposable
|
||||
{
|
||||
private readonly List<IDisposable> _disposables = new();
|
||||
private readonly Dictionary<string, string?> _previousEnv = new(StringComparer.Ordinal);
|
||||
private readonly string _tempDbPath;
|
||||
|
||||
public SiteHealthEndpointTests()
|
||||
{
|
||||
_tempDbPath = Path.Combine(Path.GetTempPath(), $"scadabridge_health_ep_{Guid.NewGuid()}.db");
|
||||
|
||||
// Whole-key env overrides, the sanctioned path: supplying GrpcPsk concretely makes the
|
||||
// pre-host secrets expander skip appsettings.Site.json's ${secret:SB-GRPC-PSK-site-1}
|
||||
// token, so this test needs no seeded secrets store.
|
||||
//
|
||||
// The remoting port is moved off the site default (8082) so this fixture cannot collide
|
||||
// with a locally running node, but it must be a REAL port and seed-nodes[0] must be this
|
||||
// node itself — StartupValidator enforces both before the host is built, and it runs
|
||||
// whether or not the Akka hosted service is later removed. Nothing binds it: the hosted
|
||||
// service is removed below.
|
||||
SetEnv("SCADABRIDGE_CONFIG", "Site");
|
||||
SetEnv("ScadaBridge__Node__Role", "Site");
|
||||
SetEnv("ScadaBridge__Node__SiteId", "TestSite");
|
||||
SetEnv("ScadaBridge__Node__NodeHostname", "localhost");
|
||||
SetEnv("ScadaBridge__Node__RemotingPort", "18082");
|
||||
SetEnv("ScadaBridge__Cluster__SeedNodes__0", "akka.tcp://scadabridge@localhost:18082");
|
||||
SetEnv("ScadaBridge__Cluster__SeedNodes__1", "akka.tcp://scadabridge@localhost:18085");
|
||||
SetEnv("ScadaBridge__Communication__GrpcPsk", "test-psk-0123456789");
|
||||
SetEnv("LocalDb__Path", _tempDbPath);
|
||||
}
|
||||
|
||||
private void SetEnv(string key, string? value)
|
||||
{
|
||||
_previousEnv[key] = Environment.GetEnvironmentVariable(key);
|
||||
Environment.SetEnvironmentVariable(key, value);
|
||||
}
|
||||
|
||||
public void Dispose()
|
||||
{
|
||||
foreach (var d in _disposables)
|
||||
{
|
||||
try { d.Dispose(); } catch { /* best effort */ }
|
||||
}
|
||||
|
||||
foreach (var (key, value) in _previousEnv)
|
||||
{
|
||||
Environment.SetEnvironmentVariable(key, value);
|
||||
}
|
||||
|
||||
try { File.Delete(_tempDbPath); } catch { /* best effort */ }
|
||||
|
||||
GC.SuppressFinalize(this);
|
||||
}
|
||||
|
||||
private HttpClient CreateSiteClient()
|
||||
{
|
||||
var factory = new WebApplicationFactory<Program>()
|
||||
.WithWebHostBuilder(builder =>
|
||||
{
|
||||
// WebApplicationFactory defaults the environment to Development, which turns on
|
||||
// ValidateOnBuild/ValidateScopes. The site graph carries scoped registrations whose
|
||||
// dependencies are central-only (ReconcileService → IDeploymentManagerRepository,
|
||||
// AuditLogKpiSampleSource → IAuditLogRepository) and are never resolved on a site
|
||||
// node, so eager validation fails on a composition that runs fine in production.
|
||||
// That is a pre-existing property of the site container, not of the health wiring
|
||||
// under test — run this fixture the way a site node actually runs.
|
||||
builder.UseEnvironment("Production");
|
||||
|
||||
// ConfigureTestServices runs after the app's own registrations, so this drops the
|
||||
// hosted service that would form a real cluster while leaving the AkkaHostedService
|
||||
// singleton resolvable (the site pipeline resolves it for shutdown ordering).
|
||||
builder.ConfigureTestServices(AkkaHostedServiceRemover.RemoveAkkaHostedServiceOnly);
|
||||
});
|
||||
_disposables.Add(factory);
|
||||
|
||||
var client = factory.CreateClient();
|
||||
_disposables.Add(client);
|
||||
return client;
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task Site_MapsHealthReady_WithTheCanonicalJsonBody()
|
||||
{
|
||||
var response = await CreateSiteClient().GetAsync("/health/ready");
|
||||
|
||||
// Never 404. 200 vs 503 depends on cluster state this fixture does not form, so both are
|
||||
// accepted — what is asserted is that the tier is mapped and answers in canonical shape.
|
||||
Assert.NotEqual(HttpStatusCode.NotFound, response.StatusCode);
|
||||
Assert.True(
|
||||
response.StatusCode is HttpStatusCode.OK or HttpStatusCode.ServiceUnavailable,
|
||||
$"Expected 200 or 503, got {(int)response.StatusCode}");
|
||||
|
||||
using var doc = JsonDocument.Parse(await response.Content.ReadAsStringAsync());
|
||||
var entries = doc.RootElement.GetProperty("entries");
|
||||
|
||||
Assert.True(entries.TryGetProperty("akka-cluster", out _), "ready tier must carry akka-cluster");
|
||||
Assert.True(entries.TryGetProperty("localdb", out _), "ready tier must carry localdb");
|
||||
// Readiness must NOT depend on being the primary, or a healthy standby would read unready.
|
||||
Assert.False(entries.TryGetProperty("active-node", out _), "active-node belongs to the Active tier");
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task Site_MapsHealthActive_CarryingOnlyTheActiveNodeCheck()
|
||||
{
|
||||
var response = await CreateSiteClient().GetAsync("/health/active");
|
||||
|
||||
Assert.NotEqual(HttpStatusCode.NotFound, response.StatusCode);
|
||||
|
||||
using var doc = JsonDocument.Parse(await response.Content.ReadAsStringAsync());
|
||||
var entries = doc.RootElement.GetProperty("entries");
|
||||
|
||||
Assert.Equal(
|
||||
new[] { "active-node" },
|
||||
entries.EnumerateObject().Select(p => p.Name).ToArray());
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task Site_HealthEndpointsAreAnonymous()
|
||||
{
|
||||
// The dashboard authenticates nowhere. The site pipeline runs no authentication middleware
|
||||
// today, so this is a regression pin: adding one later must not silently close these.
|
||||
var client = CreateSiteClient();
|
||||
|
||||
foreach (var path in new[] { "/health/ready", "/health/active", "/healthz" })
|
||||
{
|
||||
var response = await client.GetAsync(path);
|
||||
|
||||
Assert.NotEqual(HttpStatusCode.Unauthorized, response.StatusCode);
|
||||
Assert.NotEqual(HttpStatusCode.Forbidden, response.StatusCode);
|
||||
Assert.NotEqual(HttpStatusCode.NotFound, response.StatusCode);
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user