diff --git a/Directory.Packages.props b/Directory.Packages.props
index 71e1ff93..d111b772 100644
--- a/Directory.Packages.props
+++ b/Directory.Packages.props
@@ -90,9 +90,9 @@
to mark tests as Skipped (not silently Passed) when MSSQL is unreachable.
-->
-
-
-
+
+
+
diff --git a/src/ZB.MOM.WW.ScadaBridge.Host/Health/SiteLocalDbHealthCheck.cs b/src/ZB.MOM.WW.ScadaBridge.Host/Health/SiteLocalDbHealthCheck.cs
new file mode 100644
index 00000000..fb472298
--- /dev/null
+++ b/src/ZB.MOM.WW.ScadaBridge.Host/Health/SiteLocalDbHealthCheck.cs
@@ -0,0 +1,77 @@
+using Microsoft.Extensions.Diagnostics.HealthChecks;
+using ZB.MOM.WW.LocalDb;
+using ZB.MOM.WW.LocalDb.Replication;
+
+namespace ZB.MOM.WW.ScadaBridge.Host.Health;
+
+///
+/// /health/ready check for the consolidated site database (LocalDb). Site nodes have no EF
+/// context — central's DatabaseHealthCheck<ScadaBridgeDbContext> is central-only — so
+/// the site store is probed directly through the registered with a trivial
+/// SELECT 1. That is the honest readiness question for a site node: everything it persists
+/// (OperationTracking, site events, store-and-forward, the replicated config tables) lives here.
+///
+///
+/// Replication state is reported as data ENRICHMENT ONLY and never affects the status.
+/// Replication is default-OFF by design, so a node with no peer reports connected=false with
+/// a real 0 backlog — the healthy steady state of an unreplicated node — and failing the check on
+/// it would mark every correctly-configured single node unready.
+///
+public sealed class SiteLocalDbHealthCheck : IHealthCheck
+{
+ private readonly ILocalDb _localDb;
+ private readonly ISyncStatus? _syncStatus;
+
+ /// Initializes a new .
+ /// The consolidated site database registered by AddZbLocalDb.
+ ///
+ /// The replication status view, when the replication engine is registered. Optional so the check
+ /// still works in a composition that wires storage without replication.
+ ///
+ public SiteLocalDbHealthCheck(ILocalDb localDb, ISyncStatus? syncStatus = null)
+ {
+ _localDb = localDb ?? throw new ArgumentNullException(nameof(localDb));
+ _syncStatus = syncStatus;
+ }
+
+ /// Probes the site database and reports its reachability.
+ /// The health check context (unused; the verdict depends only on the probe).
+ /// Cancellation token.
+ /// Healthy when the probe query succeeds; Unhealthy with the fault otherwise.
+ public async Task CheckHealthAsync(
+ HealthCheckContext context,
+ CancellationToken cancellationToken = default)
+ {
+ try
+ {
+ await _localDb.QueryAsync("SELECT 1;", static r => r.GetInt32(0), ct: cancellationToken)
+ .ConfigureAwait(false);
+ }
+ catch (Exception ex) when (ex is not OperationCanceledException)
+ {
+ return HealthCheckResult.Unhealthy("Site LocalDb is not reachable.", ex);
+ }
+
+ return HealthCheckResult.Healthy("Site LocalDb reachable.", BuildData());
+ }
+
+ private IReadOnlyDictionary BuildData()
+ {
+ var data = new Dictionary(StringComparer.Ordinal);
+ if (_syncStatus is null)
+ return data;
+
+ data["replicationConnected"] = _syncStatus.Connected;
+
+ // Nullable on purpose upstream: null means "backlog unknown" (no provider wired or the poll
+ // failed), which must not be flattened to 0 — that would render a pair which cannot read its
+ // own oplog as perfectly caught up. Omit the key instead.
+ if (_syncStatus.OplogBacklog is { } backlog)
+ data["replicationOplogBacklog"] = backlog;
+
+ if (_syncStatus.PeerNodeId is { } peer)
+ data["replicationPeerNodeId"] = peer;
+
+ return data;
+ }
+}
diff --git a/src/ZB.MOM.WW.ScadaBridge.Host/Health/SitePairActiveNodeHealthCheck.cs b/src/ZB.MOM.WW.ScadaBridge.Host/Health/SitePairActiveNodeHealthCheck.cs
new file mode 100644
index 00000000..c17e0bfd
--- /dev/null
+++ b/src/ZB.MOM.WW.ScadaBridge.Host/Health/SitePairActiveNodeHealthCheck.cs
@@ -0,0 +1,39 @@
+using Microsoft.Extensions.Diagnostics.HealthChecks;
+using ZB.MOM.WW.ScadaBridge.HealthMonitoring;
+
+namespace ZB.MOM.WW.ScadaBridge.Host.Health;
+
+///
+/// /health/active check for a SITE pair: Healthy only on the pair's primary — the oldest Up member
+/// carrying the site-{SiteId} role. Same active/standby semantics central publishes, so a
+/// consumer reads one contract fleet-wide (200 = Active, 503 = Standby).
+///
+///
+/// Deliberately NOT central's : that one calls
+/// ClusterActivityEvaluator.SelfIsOldest(cluster) with no role argument, so on a mesh that
+/// carries more than one site's members it computes "oldest" across the wrong member set and a site
+/// node could report Active because it happens to be the oldest member overall. Delegating to
+/// reuses the role-scoped evaluation the site
+/// already trusts for singleton placement and event-log purge — one primary definition per site,
+/// not two that can disagree.
+///
+public sealed class SitePairActiveNodeHealthCheck : IHealthCheck
+{
+ private readonly IClusterNodeProvider _nodeProvider;
+
+ /// Initializes a new .
+ /// The role-scoped site cluster node provider.
+ public SitePairActiveNodeHealthCheck(IClusterNodeProvider nodeProvider) =>
+ _nodeProvider = nodeProvider ?? throw new ArgumentNullException(nameof(nodeProvider));
+
+ /// Reports Healthy on the site pair's primary and Unhealthy on the standby.
+ /// The health check context (unused; the verdict depends only on cluster membership).
+ /// Cancellation token.
+ /// Healthy on the primary; Unhealthy (503 = Standby) otherwise.
+ public Task CheckHealthAsync(
+ HealthCheckContext context,
+ CancellationToken cancellationToken = default) =>
+ Task.FromResult(_nodeProvider.SelfIsPrimary
+ ? HealthCheckResult.Healthy("site pair primary (oldest Up member of the site role)")
+ : HealthCheckResult.Unhealthy("not the site pair primary (standby)"));
+}
diff --git a/src/ZB.MOM.WW.ScadaBridge.Host/Program.cs b/src/ZB.MOM.WW.ScadaBridge.Host/Program.cs
index 433d2990..831506fe 100644
--- a/src/ZB.MOM.WW.ScadaBridge.Host/Program.cs
+++ b/src/ZB.MOM.WW.ScadaBridge.Host/Program.cs
@@ -634,6 +634,20 @@ try
// must be present before the Map* calls on the site role.
app.UseRouting();
+ // Map the canonical three-tier health endpoints on the site node:
+ // /health/ready — akka-cluster (site-pair membership, and the cluster-view data the
+ // family overview dashboard reads) + localdb (the consolidated site
+ // store — the only database a site node has).
+ // /health/active — active-node = this site pair's primary; 200 on the primary, 503 on
+ // the standby, the same contract central publishes.
+ // /healthz — bare process liveness.
+ // Checks are registered in SiteServiceRegistration.Configure. All three endpoints are
+ // anonymous and use the canonical ZbHealthWriter JSON; the site pipeline runs no
+ // authentication middleware and has no FallbackPolicy, so nothing extra is needed to keep
+ // them reachable. They are served on the HTTP/1.1 listener (metricsPort, default :8084)
+ // alongside /metrics — the gRPC listener is HTTP/2-only and unusable by a plain HTTP client.
+ app.MapZbHealth();
+
// Observability — mount the always-on Prometheus /metrics scrape endpoint.
// AddZbTelemetry (in SiteServiceRegistration.Configure → BindSharedOptions)
// wires the OTel Resource + standard instrumentation + Prometheus exporter;
diff --git a/src/ZB.MOM.WW.ScadaBridge.Host/SiteServiceRegistration.cs b/src/ZB.MOM.WW.ScadaBridge.Host/SiteServiceRegistration.cs
index f14aac2e..83794565 100644
--- a/src/ZB.MOM.WW.ScadaBridge.Host/SiteServiceRegistration.cs
+++ b/src/ZB.MOM.WW.ScadaBridge.Host/SiteServiceRegistration.cs
@@ -1,5 +1,7 @@
using Microsoft.Extensions.DependencyInjection.Extensions;
using Microsoft.Extensions.Options;
+using ZB.MOM.WW.Health;
+using ZB.MOM.WW.Health.Akka;
using ZB.MOM.WW.ScadaBridge.AuditLog;
using ZB.MOM.WW.ScadaBridge.ClusterInfrastructure;
using ZB.MOM.WW.ScadaBridge.Communication;
@@ -190,6 +192,39 @@ public static class SiteServiceRegistration
return () => nodeProvider.SelfIsPrimary;
});
+ // Health checks — the shared ZB.MOM.WW.Health probes, mapped by MapZbHealth on the site's
+ // HTTP/1.1 listener (Program.cs). Site nodes served no health endpoints before this; the
+ // family overview dashboard probes every node the same way, so a site node has to answer
+ // the same two questions central does.
+ //
+ // Ready (/health/ready) akka-cluster — this node's site-pair membership. It also carries
+ // the cluster-view data (leader/memberCount/...) that the
+ // dashboard reads, free with ZB.MOM.WW.Health 0.2.0.
+ // localdb — the consolidated site store, which is the only
+ // database a site node has (central's EF DatabaseHealthCheck is
+ // central-only and does not apply here).
+ // Active (/health/active) active-node — site-pair primary vs standby.
+ //
+ // Registered HERE rather than in Program.cs, like the LocalDb registrations above, so the
+ // composition-root tests that build this graph actually cover them. The Akka check resolves
+ // ActorSystem from DI per probe via the singleton bridge registered above.
+ services.AddHealthChecks()
+ .AddTypeActivatedCheck(
+ "akka-cluster",
+ failureStatus: null,
+ tags: new[] { ZbHealthTags.Ready },
+ args: AkkaClusterStatusPolicy.Default)
+ .AddTypeActivatedCheck(
+ "localdb",
+ failureStatus: null,
+ tags: new[] { ZbHealthTags.Ready })
+ // NOT central's OldestNodeActiveHealthCheck — that one is role-less; see
+ // SitePairActiveNodeHealthCheck for why a site pair needs the role-scoped answer.
+ .AddTypeActivatedCheck(
+ "active-node",
+ failureStatus: null,
+ tags: new[] { ZbHealthTags.Active });
+
// Options binding
BindSharedOptions(services, config);
diff --git a/tests/ZB.MOM.WW.ScadaBridge.Host.Tests/SiteHealthCheckTests.cs b/tests/ZB.MOM.WW.ScadaBridge.Host.Tests/SiteHealthCheckTests.cs
new file mode 100644
index 00000000..850902f3
--- /dev/null
+++ b/tests/ZB.MOM.WW.ScadaBridge.Host.Tests/SiteHealthCheckTests.cs
@@ -0,0 +1,235 @@
+using Microsoft.AspNetCore.Builder;
+using Microsoft.Extensions.Configuration;
+using Microsoft.Extensions.DependencyInjection;
+using Microsoft.Extensions.Diagnostics.HealthChecks;
+using Microsoft.Extensions.Options;
+using ZB.MOM.WW.Health;
+using ZB.MOM.WW.Health.Akka;
+using ZB.MOM.WW.LocalDb;
+using ZB.MOM.WW.LocalDb.Registration;
+using ZB.MOM.WW.ScadaBridge.Commons.Messages.Health;
+using ZB.MOM.WW.ScadaBridge.HealthMonitoring;
+using ZB.MOM.WW.ScadaBridge.Host.Health;
+
+namespace ZB.MOM.WW.ScadaBridge.Host.Tests;
+
+///
+/// Site-node health endpoints. Site nodes previously served no health surface at all — only gRPC on
+/// the HTTP/2 listener and /metrics on the HTTP/1.1 one — so the family overview dashboard
+/// had no uniform way to probe them. These tests pin the three checks the site now registers and
+/// the tier each one lands in.
+///
+/// The registration assertions build the REAL container from
+/// rather than a hand-assembled
+/// ServiceCollection, for the reason spelled out in :
+/// this family's wiring defects have consistently been ones that only a container-built graph
+/// catches. Each registration is additionally ACTIVATED through its factory, so a check that is
+/// registered but cannot resolve its dependencies fails here instead of at the first live probe.
+///
+///
+public class SiteHealthCheckTests : IDisposable
+{
+ private readonly WebApplication _host;
+ private readonly string _tempDbPath;
+ private readonly string _tempSiteDbPath;
+ private readonly string _tempTrackingDbPath;
+
+ public SiteHealthCheckTests()
+ {
+ var stamp = Guid.NewGuid();
+ _tempDbPath = Path.Combine(Path.GetTempPath(), $"scadabridge_health_localdb_{stamp}.db");
+ _tempSiteDbPath = Path.Combine(Path.GetTempPath(), $"scadabridge_health_site_{stamp}.db");
+ _tempTrackingDbPath = Path.Combine(Path.GetTempPath(), $"scadabridge_health_track_{stamp}.db");
+
+ var builder = WebApplication.CreateBuilder();
+ builder.Configuration.Sources.Clear();
+ builder.Configuration.AddInMemoryCollection(new Dictionary
+ {
+ ["ScadaBridge:Node:Role"] = "Site",
+ ["ScadaBridge:Node:NodeName"] = "node-a",
+ ["ScadaBridge:Node:NodeHostname"] = "test-site",
+ ["ScadaBridge:Node:SiteId"] = "TestSite",
+ ["ScadaBridge:Node:RemotingPort"] = "0",
+ ["ScadaBridge:Node:GrpcPort"] = "0",
+ ["ScadaBridge:Database:SiteDbPath"] = _tempSiteDbPath,
+ ["ScadaBridge:OperationTracking:ConnectionString"] = $"Data Source={_tempTrackingDbPath}",
+ ["ScadaBridge:Cluster:SeedNodes:0"] = "akka.tcp://scadabridge@localhost:2551",
+ ["ScadaBridge:Cluster:SeedNodes:1"] = "akka.tcp://scadabridge@localhost:2552",
+ ["LocalDb:Path"] = _tempDbPath,
+ });
+
+ builder.Services.AddGrpc();
+ builder.Services.AddSingleton();
+
+ SiteServiceRegistration.Configure(builder.Services, builder.Configuration);
+
+ AkkaHostedServiceRemover.RemoveAkkaHostedServiceOnly(builder.Services);
+
+ _host = builder.Build();
+ }
+
+ public void Dispose()
+ {
+ (_host as IDisposable)?.Dispose();
+ foreach (var path in new[] { _tempDbPath, _tempSiteDbPath, _tempTrackingDbPath })
+ {
+ try { File.Delete(path); } catch { /* best effort */ }
+ }
+
+ GC.SuppressFinalize(this);
+ }
+
+ private IReadOnlyList Registrations =>
+ _host.Services.GetRequiredService>().Value.Registrations.ToList();
+
+ [Fact]
+ public void Site_RegistersExactlyTheThreeSiteChecks()
+ {
+ // Exact-set, not "contains": an extra check silently changes what /health/ready means for
+ // every consumer, and a site node inheriting a central-only probe is precisely the mistake
+ // this asserts against.
+ Assert.Equal(
+ new[] { "active-node", "akka-cluster", "localdb" },
+ Registrations.Select(r => r.Name).OrderBy(n => n, StringComparer.Ordinal).ToArray());
+ }
+
+ [Theory]
+ [InlineData("akka-cluster", ZbHealthTags.Ready)]
+ [InlineData("localdb", ZbHealthTags.Ready)]
+ [InlineData("active-node", ZbHealthTags.Active)]
+ public void Site_ChecksCarryTheirIntendedTier(string name, string expectedTag)
+ {
+ var registration = Assert.Single(Registrations, r => r.Name == name);
+
+ // Exactly one tier tag each: MapZbHealth selects by tag, so a check tagged both Ready and
+ // Active would put standby-ness into readiness and drop the standby out of readiness too.
+ Assert.Equal(new[] { expectedTag }, registration.Tags.ToArray());
+ }
+
+ [Theory]
+ [InlineData("akka-cluster", typeof(AkkaClusterHealthCheck))]
+ [InlineData("localdb", typeof(SiteLocalDbHealthCheck))]
+ [InlineData("active-node", typeof(SitePairActiveNodeHealthCheck))]
+ public void Site_ChecksActivateToTheIntendedType(string name, Type expected)
+ {
+ var registration = Assert.Single(Registrations, r => r.Name == name);
+
+ // Running the factory is the point: a type-activated check whose constructor dependencies
+ // are not registered on site would throw here rather than on the first probe in production.
+ var check = registration.Factory(_host.Services);
+
+ Assert.IsType(expected, check);
+ }
+
+ [Fact]
+ public void Site_ActiveNodeCheck_IsNotCentralsRoleLessOldestCheck()
+ {
+ // Central's OldestNodeActiveHealthCheck calls SelfIsOldest(cluster) with no role argument.
+ // On a mesh carrying more than one site's members that computes "oldest" across the wrong
+ // member set, so a site node could report Active purely for being the oldest member overall.
+ var check = Assert.Single(Registrations, r => r.Name == "active-node").Factory(_host.Services);
+
+ Assert.IsNotType(check);
+ }
+
+ [Theory]
+ [InlineData("database")]
+ [InlineData("required-singletons")]
+ public void Site_DoesNotRegisterCentralOnlyChecks(string centralOnlyName)
+ {
+ // Positive control first: if the harness ever stopped registering health checks at all,
+ // every "is absent" assertion below would pass vacuously and this test would be worthless.
+ Assert.Contains(Registrations, r => r.Name == "akka-cluster");
+
+ Assert.DoesNotContain(Registrations, r => r.Name == centralOnlyName);
+ }
+
+ [Fact]
+ public async Task Site_LocalDbCheck_ReportsHealthyAgainstTheRealSiteDatabase()
+ {
+ // Behaviour, not wiring: the probe runs against the consolidated site database this
+ // composition actually built, so a check that resolves but cannot query fails here.
+ var check = (SiteLocalDbHealthCheck)Assert.Single(Registrations, r => r.Name == "localdb")
+ .Factory(_host.Services);
+
+ var result = await check.CheckHealthAsync(NewContext("localdb", check));
+
+ Assert.Equal(HealthStatus.Healthy, result.Status);
+ }
+
+ [Fact]
+ public async Task Site_LocalDbCheck_EnrichesWithReplicationStateWithoutFailingOnDefaultOff()
+ {
+ // Replication ships default-OFF, so the unreplicated steady state must stay Healthy while
+ // still surfacing the state as data — enrichment must never become a verdict.
+ var check = (SiteLocalDbHealthCheck)Assert.Single(Registrations, r => r.Name == "localdb")
+ .Factory(_host.Services);
+
+ var result = await check.CheckHealthAsync(NewContext("localdb", check));
+
+ Assert.Equal(HealthStatus.Healthy, result.Status);
+ Assert.False(Assert.IsType(result.Data["replicationConnected"]));
+ Assert.Equal(0L, Assert.IsType(result.Data["replicationOplogBacklog"]));
+ }
+
+ [Fact]
+ public async Task Site_LocalDbCheck_ReportsUnhealthyWhenTheStoreIsUnreachable()
+ {
+ var check = new SiteLocalDbHealthCheck(new ThrowingLocalDb());
+
+ var result = await check.CheckHealthAsync(NewContext("localdb", check));
+
+ Assert.Equal(HealthStatus.Unhealthy, result.Status);
+ }
+
+ [Theory]
+ [InlineData(true, HealthStatus.Healthy)]
+ [InlineData(false, HealthStatus.Unhealthy)]
+ public async Task Site_ActiveNodeCheck_SplitsPrimaryFromStandby(bool selfIsPrimary, HealthStatus expected)
+ {
+ // 200 on the primary / 503 on the standby is the same contract central publishes, so a
+ // consumer reads one active-node meaning fleet-wide.
+ var check = new SitePairActiveNodeHealthCheck(new StubNodeProvider(selfIsPrimary));
+
+ var result = await check.CheckHealthAsync(NewContext("active-node", check));
+
+ Assert.Equal(expected, result.Status);
+ }
+
+ private static HealthCheckContext NewContext(string name, IHealthCheck check) => new()
+ {
+ Registration = new HealthCheckRegistration(name, check, HealthStatus.Unhealthy, tags: null),
+ };
+
+ private sealed class StubNodeProvider(bool selfIsPrimary) : IClusterNodeProvider
+ {
+ public bool SelfIsPrimary { get; } = selfIsPrimary;
+
+ public IReadOnlyList GetClusterNodes() => [];
+ }
+
+ private sealed class ThrowingLocalDb : ILocalDb
+ {
+ public Task ExecuteAsync(string sql, object? parameters = null, CancellationToken ct = default) =>
+ throw new InvalidOperationException("store unreachable");
+
+ public Task> QueryAsync(
+ string sql,
+ Func map,
+ object? parameters = null,
+ CancellationToken ct = default) =>
+ throw new InvalidOperationException("store unreachable");
+
+ public Task BeginTransactionAsync(CancellationToken ct = default) =>
+ throw new InvalidOperationException("store unreachable");
+
+ public Microsoft.Data.Sqlite.SqliteConnection CreateConnection() =>
+ throw new InvalidOperationException("store unreachable");
+
+ public ReplicatedTable RegisterReplicated(string tableName) =>
+ throw new InvalidOperationException("store unreachable");
+
+ public IReadOnlyDictionary ReplicatedTables =>
+ throw new InvalidOperationException("store unreachable");
+ }
+}
diff --git a/tests/ZB.MOM.WW.ScadaBridge.Host.Tests/SiteHealthEndpointTests.cs b/tests/ZB.MOM.WW.ScadaBridge.Host.Tests/SiteHealthEndpointTests.cs
new file mode 100644
index 00000000..5be47b9e
--- /dev/null
+++ b/tests/ZB.MOM.WW.ScadaBridge.Host.Tests/SiteHealthEndpointTests.cs
@@ -0,0 +1,156 @@
+using System.Net;
+using System.Text.Json;
+using Microsoft.AspNetCore.Hosting;
+using Microsoft.AspNetCore.Mvc.Testing;
+using Microsoft.AspNetCore.TestHost;
+using Microsoft.Extensions.DependencyInjection;
+
+namespace ZB.MOM.WW.ScadaBridge.Host.Tests;
+
+///
+/// Proves the site pipeline actually MAPS the health endpoints — the half
+/// cannot cover, since that fixture builds the site container
+/// from and never runs Program's endpoint wiring.
+/// Without this, deleting app.MapZbHealth() from the site branch would leave every
+/// registration test green while site nodes served 404 to the overview dashboard.
+///
+/// Boots the real Program in the Site role. Program builds its own
+/// ConfigurationBuilder before the web host exists, so its role and settings come from
+/// PROCESS-WIDE environment variables (SCADABRIDGE_CONFIG selects appsettings.Site.json)
+/// rather than from WithWebHostBuilder — which is why this fixture joins
+/// and restores every var it sets.
+///
+///
+[Collection(HostBootCollection.Name)]
+public class SiteHealthEndpointTests : IDisposable
+{
+ private readonly List _disposables = new();
+ private readonly Dictionary _previousEnv = new(StringComparer.Ordinal);
+ private readonly string _tempDbPath;
+
+ public SiteHealthEndpointTests()
+ {
+ _tempDbPath = Path.Combine(Path.GetTempPath(), $"scadabridge_health_ep_{Guid.NewGuid()}.db");
+
+ // Whole-key env overrides, the sanctioned path: supplying GrpcPsk concretely makes the
+ // pre-host secrets expander skip appsettings.Site.json's ${secret:SB-GRPC-PSK-site-1}
+ // token, so this test needs no seeded secrets store.
+ //
+ // The remoting port is moved off the site default (8082) so this fixture cannot collide
+ // with a locally running node, but it must be a REAL port and seed-nodes[0] must be this
+ // node itself — StartupValidator enforces both before the host is built, and it runs
+ // whether or not the Akka hosted service is later removed. Nothing binds it: the hosted
+ // service is removed below.
+ SetEnv("SCADABRIDGE_CONFIG", "Site");
+ SetEnv("ScadaBridge__Node__Role", "Site");
+ SetEnv("ScadaBridge__Node__SiteId", "TestSite");
+ SetEnv("ScadaBridge__Node__NodeHostname", "localhost");
+ SetEnv("ScadaBridge__Node__RemotingPort", "18082");
+ SetEnv("ScadaBridge__Cluster__SeedNodes__0", "akka.tcp://scadabridge@localhost:18082");
+ SetEnv("ScadaBridge__Cluster__SeedNodes__1", "akka.tcp://scadabridge@localhost:18085");
+ SetEnv("ScadaBridge__Communication__GrpcPsk", "test-psk-0123456789");
+ SetEnv("LocalDb__Path", _tempDbPath);
+ }
+
+ private void SetEnv(string key, string? value)
+ {
+ _previousEnv[key] = Environment.GetEnvironmentVariable(key);
+ Environment.SetEnvironmentVariable(key, value);
+ }
+
+ public void Dispose()
+ {
+ foreach (var d in _disposables)
+ {
+ try { d.Dispose(); } catch { /* best effort */ }
+ }
+
+ foreach (var (key, value) in _previousEnv)
+ {
+ Environment.SetEnvironmentVariable(key, value);
+ }
+
+ try { File.Delete(_tempDbPath); } catch { /* best effort */ }
+
+ GC.SuppressFinalize(this);
+ }
+
+ private HttpClient CreateSiteClient()
+ {
+ var factory = new WebApplicationFactory()
+ .WithWebHostBuilder(builder =>
+ {
+ // WebApplicationFactory defaults the environment to Development, which turns on
+ // ValidateOnBuild/ValidateScopes. The site graph carries scoped registrations whose
+ // dependencies are central-only (ReconcileService → IDeploymentManagerRepository,
+ // AuditLogKpiSampleSource → IAuditLogRepository) and are never resolved on a site
+ // node, so eager validation fails on a composition that runs fine in production.
+ // That is a pre-existing property of the site container, not of the health wiring
+ // under test — run this fixture the way a site node actually runs.
+ builder.UseEnvironment("Production");
+
+ // ConfigureTestServices runs after the app's own registrations, so this drops the
+ // hosted service that would form a real cluster while leaving the AkkaHostedService
+ // singleton resolvable (the site pipeline resolves it for shutdown ordering).
+ builder.ConfigureTestServices(AkkaHostedServiceRemover.RemoveAkkaHostedServiceOnly);
+ });
+ _disposables.Add(factory);
+
+ var client = factory.CreateClient();
+ _disposables.Add(client);
+ return client;
+ }
+
+ [Fact]
+ public async Task Site_MapsHealthReady_WithTheCanonicalJsonBody()
+ {
+ var response = await CreateSiteClient().GetAsync("/health/ready");
+
+ // Never 404. 200 vs 503 depends on cluster state this fixture does not form, so both are
+ // accepted — what is asserted is that the tier is mapped and answers in canonical shape.
+ Assert.NotEqual(HttpStatusCode.NotFound, response.StatusCode);
+ Assert.True(
+ response.StatusCode is HttpStatusCode.OK or HttpStatusCode.ServiceUnavailable,
+ $"Expected 200 or 503, got {(int)response.StatusCode}");
+
+ using var doc = JsonDocument.Parse(await response.Content.ReadAsStringAsync());
+ var entries = doc.RootElement.GetProperty("entries");
+
+ Assert.True(entries.TryGetProperty("akka-cluster", out _), "ready tier must carry akka-cluster");
+ Assert.True(entries.TryGetProperty("localdb", out _), "ready tier must carry localdb");
+ // Readiness must NOT depend on being the primary, or a healthy standby would read unready.
+ Assert.False(entries.TryGetProperty("active-node", out _), "active-node belongs to the Active tier");
+ }
+
+ [Fact]
+ public async Task Site_MapsHealthActive_CarryingOnlyTheActiveNodeCheck()
+ {
+ var response = await CreateSiteClient().GetAsync("/health/active");
+
+ Assert.NotEqual(HttpStatusCode.NotFound, response.StatusCode);
+
+ using var doc = JsonDocument.Parse(await response.Content.ReadAsStringAsync());
+ var entries = doc.RootElement.GetProperty("entries");
+
+ Assert.Equal(
+ new[] { "active-node" },
+ entries.EnumerateObject().Select(p => p.Name).ToArray());
+ }
+
+ [Fact]
+ public async Task Site_HealthEndpointsAreAnonymous()
+ {
+ // The dashboard authenticates nowhere. The site pipeline runs no authentication middleware
+ // today, so this is a regression pin: adding one later must not silently close these.
+ var client = CreateSiteClient();
+
+ foreach (var path in new[] { "/health/ready", "/health/active", "/healthz" })
+ {
+ var response = await client.GetAsync(path);
+
+ Assert.NotEqual(HttpStatusCode.Unauthorized, response.StatusCode);
+ Assert.NotEqual(HttpStatusCode.Forbidden, response.StatusCode);
+ Assert.NotEqual(HttpStatusCode.NotFound, response.StatusCode);
+ }
+ }
+}