Files
scadaproj/ZB.MOM.WW.Overview/tests/ZB.MOM.WW.Overview.Tests/ZbHealthReportClientTests.cs
T
Joseph Doherty d0110cf406 feat(overview): scaffold the family dashboard + registry + health client
Phase 3 tasks 3.1-3.3 of docs/plans/2026-07-22-overview-dashboard-impl-plan.md.

Scaffold (3.1): a plain directory in scadaproj (NOT a nested git repo), slnx over
src/ + tests/, HistorianGateway-pattern Directory.Build.props (warnings-as-errors)
and nuget.config (nuget.org * + the Gitea feed scoped to ZB.MOM.WW.*). Inline
package pins, no CPM — the app convention here. NO Auth/Audit/Secrets packages:
the dashboard is anonymous and read-only by requirement.

Registry (3.2): OverviewOptions/ApplicationEntry/InstanceEntry bound from the
"Overview" section, plus EffectiveTimings resolving instance > application >
global overrides in ONE place so no caller re-implements the precedence.
OverviewOptionsValidator (OptionsValidatorBase + AddValidatedOptions) rejects an
empty registry, applications with no instances, duplicate names, non-absolute or
non-http(s) URLs, and non-positive intervals — and checks stale > poll on the
EFFECTIVE values, since an override at either level can invert that on one
instance while the globals look fine. A ConfigPreflight presence check runs before
the host is built so a missing registry fails with the key name, not with
"Applications must have at least 1 entry".

Health client (3.3): ZbHealthReport DTOs for the canonical ZbHealthWriter body,
with per-entry `data` kept as JsonElement — it is an open contract, and binding it
to concrete types would break the moment a check adds a field. 200 AND 503 are
both parseable answers (503 carries the body naming the failing check); only a
transport failure or a non-canonical body is a failed probe, which is why
Unreachable stays distinct from Down. Host shutdown propagates as cancellation
rather than being recorded as every instance timing out.

Status model: Up/Degraded/Down/Unreachable + Active/Standby/Unknown/NotApplicable,
with two-strike flap damping — a failing observation only replaces a good state
after 2 consecutive failures, while recovery is immediate (damping suppresses
false alarms; symmetric damping would just make them linger).

53 tests green, 0 warnings in Debug and Release. Tests for the validator, the
client golden payloads (with AND without `data`, so a partly-bumped fleet renders)
and the derivation table are front-loaded here from task 3.8 so this batch is
verified rather than pending.

Also pins AngleSharp 1.5.2 test-side: bunit 1.40.0 pulls 1.2.0, which trips NU1902
and therefore the warnings-as-errors gate (same fix HistorianGateway made in
6bc005d).
2026-07-24 06:42:21 -04:00

200 lines
7.6 KiB
C#

using System.Net;
using System.Text;
using ZB.MOM.WW.Overview.Polling;
namespace ZB.MOM.WW.Overview.Tests;
/// <summary>
/// Golden-payload tests for the health client. The fixtures are hand-written to the exact shape
/// <c>ZbHealthWriter</c> emits — including an entry WITH <c>data</c> (0.2.0+) and one WITHOUT
/// (every check that publishes none, and every pre-0.2.0 node), because tolerating both is what
/// lets the dashboard run against a fleet that is only partly bumped.
/// </summary>
public class ZbHealthReportClientTests
{
private const string ReadyHealthyWithData = """
{
"status": "Healthy",
"totalDurationMs": 12.34,
"entries": {
"akka-cluster": {
"status": "Healthy",
"description": "Akka cluster member status: Up",
"durationMs": 1.23,
"data": {
"selfAddress": "akka.tcp://scadabridge@site-a-a:8082",
"selfRoles": ["Site", "site-a"],
"memberCount": 2,
"unreachableCount": 0,
"leader": "akka.tcp://scadabridge@site-a-a:8082"
}
},
"localdb": {
"status": "Healthy",
"description": "Site LocalDb reachable.",
"durationMs": 0.45
}
}
}
""";
private const string ReadyUnhealthyNoData = """
{
"status": "Unhealthy",
"totalDurationMs": 3001.5,
"entries": {
"database": {
"status": "Unhealthy",
"description": null,
"durationMs": 3000.1
}
}
}
""";
private static ZbHealthReportClient ClientReturning(
HttpStatusCode statusCode,
string body,
string contentType = "application/json") =>
new(new HttpClient(new StubHandler((statusCode, body, contentType))));
private static readonly TimeSpan ProbeTimeout = TimeSpan.FromSeconds(3);
[Fact]
public async Task Probe_HealthyBody_ParsesEnvelopeAndEntries()
{
var client = ClientReturning(HttpStatusCode.OK, ReadyHealthyWithData);
var result = await client.ProbeAsync("http://node/health/ready", ProbeTimeout);
Assert.True(result.Reachable);
Assert.Equal(200, result.StatusCode);
Assert.Equal("Healthy", result.Report!.Status);
Assert.Equal(12.34, result.Report.TotalDurationMs);
Assert.Equal(new[] { "akka-cluster", "localdb" }, result.Report.Entries.Keys.OrderBy(k => k, StringComparer.Ordinal));
Assert.Equal("Site LocalDb reachable.", result.Report.Entries["localdb"].Description);
}
[Fact]
public async Task Probe_EntryWithData_ExposesLeaderAndCounts()
{
var client = ClientReturning(HttpStatusCode.OK, ReadyHealthyWithData);
var result = await client.ProbeAsync("http://node/health/ready", ProbeTimeout);
var entry = result.Report!.Entries["akka-cluster"];
Assert.Equal("akka.tcp://scadabridge@site-a-a:8082", entry.DataString("leader"));
Assert.Equal(2, entry.DataInt("memberCount"));
Assert.Equal(0, entry.DataInt("unreachableCount"));
}
[Fact]
public async Task Probe_EntryWithoutData_ReadsAsNullNotAsAnError()
{
// The 0.1.0-era / no-data case. It must degrade to "no leader shown", never to a parse
// failure — that tolerance is what keeps the dashboard usable mid-rollout.
var client = ClientReturning(HttpStatusCode.OK, ReadyHealthyWithData);
var result = await client.ProbeAsync("http://node/health/ready", ProbeTimeout);
var entry = result.Report!.Entries["localdb"];
Assert.True(result.Reachable);
Assert.Null(entry.Data);
Assert.Null(entry.DataString("leader"));
Assert.Null(entry.DataInt("memberCount"));
}
[Fact]
public async Task Probe_503_IsAParseableAnswer_NotAFailedProbe()
{
// 503 carries a full body naming the failing check — the single most useful payload the
// dashboard receives. Treating it as a transport failure would throw that away.
var client = ClientReturning(HttpStatusCode.ServiceUnavailable, ReadyUnhealthyNoData);
var result = await client.ProbeAsync("http://node/health/ready", ProbeTimeout);
Assert.True(result.Reachable);
Assert.Equal(503, result.StatusCode);
Assert.Equal("Unhealthy", result.Report!.Status);
Assert.Null(result.Report.Entries["database"].Description);
}
[Theory]
[InlineData("<html><body>502 Bad Gateway</body></html>")]
[InlineData("not json at all")]
public async Task Probe_NonJsonBody_IsUnreachable(string body)
{
var client = ClientReturning(HttpStatusCode.OK, body, contentType: "text/html");
var result = await client.ProbeAsync("http://node/health/ready", ProbeTimeout);
Assert.False(result.Reachable);
Assert.Null(result.Report);
Assert.NotNull(result.Error);
}
[Fact]
public async Task Probe_ConnectionRefused_IsUnreachableAndDoesNotThrow()
{
var client = new ZbHealthReportClient(
new HttpClient(new ThrowingHandler(new HttpRequestException("Connection refused"))));
var result = await client.ProbeAsync("http://node/health/ready", ProbeTimeout);
Assert.False(result.Reachable);
Assert.Contains("refused", result.Error, StringComparison.OrdinalIgnoreCase);
}
[Fact]
public async Task Probe_Timeout_IsUnreachableAndReportsTheTimeout()
{
var client = new ZbHealthReportClient(new HttpClient(new HangingHandler()));
var result = await client.ProbeAsync("http://node/health/ready", TimeSpan.FromMilliseconds(50));
Assert.False(result.Reachable);
Assert.Contains("timed out", result.Error, StringComparison.OrdinalIgnoreCase);
}
[Fact]
public async Task Probe_HostShutdown_PropagatesCancellation()
{
// A shutdown must NOT be recorded as a timed-out instance — the sweep is being abandoned,
// and writing "Unreachable" for every node on the way out would be a lie in the snapshot.
using var shutdown = new CancellationTokenSource();
await shutdown.CancelAsync();
var client = new ZbHealthReportClient(new HttpClient(new HangingHandler()));
await Assert.ThrowsAnyAsync<OperationCanceledException>(
() => client.ProbeAsync("http://node/health/ready", ProbeTimeout, shutdown.Token));
}
private sealed class StubHandler((HttpStatusCode Status, string Body, string ContentType) response)
: HttpMessageHandler
{
protected override Task<HttpResponseMessage> SendAsync(
HttpRequestMessage request, CancellationToken cancellationToken) =>
Task.FromResult(new HttpResponseMessage(response.Status)
{
Content = new StringContent(response.Body, Encoding.UTF8, response.ContentType),
});
}
private sealed class ThrowingHandler(Exception exception) : HttpMessageHandler
{
protected override Task<HttpResponseMessage> SendAsync(
HttpRequestMessage request, CancellationToken cancellationToken) =>
Task.FromException<HttpResponseMessage>(exception);
}
private sealed class HangingHandler : HttpMessageHandler
{
protected override async Task<HttpResponseMessage> SendAsync(
HttpRequestMessage request, CancellationToken cancellationToken)
{
await Task.Delay(Timeout.InfiniteTimeSpan, cancellationToken).ConfigureAwait(false);
return new HttpResponseMessage(HttpStatusCode.OK);
}
}
}