feat(overview): scaffold the family dashboard + registry + health client
Phase 3 tasks 3.1-3.3 of docs/plans/2026-07-22-overview-dashboard-impl-plan.md. Scaffold (3.1): a plain directory in scadaproj (NOT a nested git repo), slnx over src/ + tests/, HistorianGateway-pattern Directory.Build.props (warnings-as-errors) and nuget.config (nuget.org * + the Gitea feed scoped to ZB.MOM.WW.*). Inline package pins, no CPM — the app convention here. NO Auth/Audit/Secrets packages: the dashboard is anonymous and read-only by requirement. Registry (3.2): OverviewOptions/ApplicationEntry/InstanceEntry bound from the "Overview" section, plus EffectiveTimings resolving instance > application > global overrides in ONE place so no caller re-implements the precedence. OverviewOptionsValidator (OptionsValidatorBase + AddValidatedOptions) rejects an empty registry, applications with no instances, duplicate names, non-absolute or non-http(s) URLs, and non-positive intervals — and checks stale > poll on the EFFECTIVE values, since an override at either level can invert that on one instance while the globals look fine. A ConfigPreflight presence check runs before the host is built so a missing registry fails with the key name, not with "Applications must have at least 1 entry". Health client (3.3): ZbHealthReport DTOs for the canonical ZbHealthWriter body, with per-entry `data` kept as JsonElement — it is an open contract, and binding it to concrete types would break the moment a check adds a field. 200 AND 503 are both parseable answers (503 carries the body naming the failing check); only a transport failure or a non-canonical body is a failed probe, which is why Unreachable stays distinct from Down. Host shutdown propagates as cancellation rather than being recorded as every instance timing out. Status model: Up/Degraded/Down/Unreachable + Active/Standby/Unknown/NotApplicable, with two-strike flap damping — a failing observation only replaces a good state after 2 consecutive failures, while recovery is immediate (damping suppresses false alarms; symmetric damping would just make them linger). 53 tests green, 0 warnings in Debug and Release. Tests for the validator, the client golden payloads (with AND without `data`, so a partly-bumped fleet renders) and the derivation table are front-loaded here from task 3.8 so this batch is verified rather than pending. Also pins AngleSharp 1.5.2 test-side: bunit 1.40.0 pulls 1.2.0, which trips NU1902 and therefore the warnings-as-errors gate (same fix HistorianGateway made in 6bc005d).
This commit is contained in:
@@ -0,0 +1,199 @@
|
||||
using System.Net;
|
||||
using System.Text;
|
||||
using ZB.MOM.WW.Overview.Polling;
|
||||
|
||||
namespace ZB.MOM.WW.Overview.Tests;
|
||||
|
||||
/// <summary>
|
||||
/// Golden-payload tests for the health client. The fixtures are hand-written to the exact shape
|
||||
/// <c>ZbHealthWriter</c> emits — including an entry WITH <c>data</c> (0.2.0+) and one WITHOUT
|
||||
/// (every check that publishes none, and every pre-0.2.0 node), because tolerating both is what
|
||||
/// lets the dashboard run against a fleet that is only partly bumped.
|
||||
/// </summary>
|
||||
public class ZbHealthReportClientTests
|
||||
{
|
||||
private const string ReadyHealthyWithData = """
|
||||
{
|
||||
"status": "Healthy",
|
||||
"totalDurationMs": 12.34,
|
||||
"entries": {
|
||||
"akka-cluster": {
|
||||
"status": "Healthy",
|
||||
"description": "Akka cluster member status: Up",
|
||||
"durationMs": 1.23,
|
||||
"data": {
|
||||
"selfAddress": "akka.tcp://scadabridge@site-a-a:8082",
|
||||
"selfRoles": ["Site", "site-a"],
|
||||
"memberCount": 2,
|
||||
"unreachableCount": 0,
|
||||
"leader": "akka.tcp://scadabridge@site-a-a:8082"
|
||||
}
|
||||
},
|
||||
"localdb": {
|
||||
"status": "Healthy",
|
||||
"description": "Site LocalDb reachable.",
|
||||
"durationMs": 0.45
|
||||
}
|
||||
}
|
||||
}
|
||||
""";
|
||||
|
||||
private const string ReadyUnhealthyNoData = """
|
||||
{
|
||||
"status": "Unhealthy",
|
||||
"totalDurationMs": 3001.5,
|
||||
"entries": {
|
||||
"database": {
|
||||
"status": "Unhealthy",
|
||||
"description": null,
|
||||
"durationMs": 3000.1
|
||||
}
|
||||
}
|
||||
}
|
||||
""";
|
||||
|
||||
private static ZbHealthReportClient ClientReturning(
|
||||
HttpStatusCode statusCode,
|
||||
string body,
|
||||
string contentType = "application/json") =>
|
||||
new(new HttpClient(new StubHandler((statusCode, body, contentType))));
|
||||
|
||||
private static readonly TimeSpan ProbeTimeout = TimeSpan.FromSeconds(3);
|
||||
|
||||
[Fact]
|
||||
public async Task Probe_HealthyBody_ParsesEnvelopeAndEntries()
|
||||
{
|
||||
var client = ClientReturning(HttpStatusCode.OK, ReadyHealthyWithData);
|
||||
|
||||
var result = await client.ProbeAsync("http://node/health/ready", ProbeTimeout);
|
||||
|
||||
Assert.True(result.Reachable);
|
||||
Assert.Equal(200, result.StatusCode);
|
||||
Assert.Equal("Healthy", result.Report!.Status);
|
||||
Assert.Equal(12.34, result.Report.TotalDurationMs);
|
||||
Assert.Equal(new[] { "akka-cluster", "localdb" }, result.Report.Entries.Keys.OrderBy(k => k, StringComparer.Ordinal));
|
||||
Assert.Equal("Site LocalDb reachable.", result.Report.Entries["localdb"].Description);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task Probe_EntryWithData_ExposesLeaderAndCounts()
|
||||
{
|
||||
var client = ClientReturning(HttpStatusCode.OK, ReadyHealthyWithData);
|
||||
|
||||
var result = await client.ProbeAsync("http://node/health/ready", ProbeTimeout);
|
||||
var entry = result.Report!.Entries["akka-cluster"];
|
||||
|
||||
Assert.Equal("akka.tcp://scadabridge@site-a-a:8082", entry.DataString("leader"));
|
||||
Assert.Equal(2, entry.DataInt("memberCount"));
|
||||
Assert.Equal(0, entry.DataInt("unreachableCount"));
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task Probe_EntryWithoutData_ReadsAsNullNotAsAnError()
|
||||
{
|
||||
// The 0.1.0-era / no-data case. It must degrade to "no leader shown", never to a parse
|
||||
// failure — that tolerance is what keeps the dashboard usable mid-rollout.
|
||||
var client = ClientReturning(HttpStatusCode.OK, ReadyHealthyWithData);
|
||||
|
||||
var result = await client.ProbeAsync("http://node/health/ready", ProbeTimeout);
|
||||
var entry = result.Report!.Entries["localdb"];
|
||||
|
||||
Assert.True(result.Reachable);
|
||||
Assert.Null(entry.Data);
|
||||
Assert.Null(entry.DataString("leader"));
|
||||
Assert.Null(entry.DataInt("memberCount"));
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task Probe_503_IsAParseableAnswer_NotAFailedProbe()
|
||||
{
|
||||
// 503 carries a full body naming the failing check — the single most useful payload the
|
||||
// dashboard receives. Treating it as a transport failure would throw that away.
|
||||
var client = ClientReturning(HttpStatusCode.ServiceUnavailable, ReadyUnhealthyNoData);
|
||||
|
||||
var result = await client.ProbeAsync("http://node/health/ready", ProbeTimeout);
|
||||
|
||||
Assert.True(result.Reachable);
|
||||
Assert.Equal(503, result.StatusCode);
|
||||
Assert.Equal("Unhealthy", result.Report!.Status);
|
||||
Assert.Null(result.Report.Entries["database"].Description);
|
||||
}
|
||||
|
||||
[Theory]
|
||||
[InlineData("<html><body>502 Bad Gateway</body></html>")]
|
||||
[InlineData("not json at all")]
|
||||
public async Task Probe_NonJsonBody_IsUnreachable(string body)
|
||||
{
|
||||
var client = ClientReturning(HttpStatusCode.OK, body, contentType: "text/html");
|
||||
|
||||
var result = await client.ProbeAsync("http://node/health/ready", ProbeTimeout);
|
||||
|
||||
Assert.False(result.Reachable);
|
||||
Assert.Null(result.Report);
|
||||
Assert.NotNull(result.Error);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task Probe_ConnectionRefused_IsUnreachableAndDoesNotThrow()
|
||||
{
|
||||
var client = new ZbHealthReportClient(
|
||||
new HttpClient(new ThrowingHandler(new HttpRequestException("Connection refused"))));
|
||||
|
||||
var result = await client.ProbeAsync("http://node/health/ready", ProbeTimeout);
|
||||
|
||||
Assert.False(result.Reachable);
|
||||
Assert.Contains("refused", result.Error, StringComparison.OrdinalIgnoreCase);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task Probe_Timeout_IsUnreachableAndReportsTheTimeout()
|
||||
{
|
||||
var client = new ZbHealthReportClient(new HttpClient(new HangingHandler()));
|
||||
|
||||
var result = await client.ProbeAsync("http://node/health/ready", TimeSpan.FromMilliseconds(50));
|
||||
|
||||
Assert.False(result.Reachable);
|
||||
Assert.Contains("timed out", result.Error, StringComparison.OrdinalIgnoreCase);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task Probe_HostShutdown_PropagatesCancellation()
|
||||
{
|
||||
// A shutdown must NOT be recorded as a timed-out instance — the sweep is being abandoned,
|
||||
// and writing "Unreachable" for every node on the way out would be a lie in the snapshot.
|
||||
using var shutdown = new CancellationTokenSource();
|
||||
await shutdown.CancelAsync();
|
||||
var client = new ZbHealthReportClient(new HttpClient(new HangingHandler()));
|
||||
|
||||
await Assert.ThrowsAnyAsync<OperationCanceledException>(
|
||||
() => client.ProbeAsync("http://node/health/ready", ProbeTimeout, shutdown.Token));
|
||||
}
|
||||
|
||||
private sealed class StubHandler((HttpStatusCode Status, string Body, string ContentType) response)
|
||||
: HttpMessageHandler
|
||||
{
|
||||
protected override Task<HttpResponseMessage> SendAsync(
|
||||
HttpRequestMessage request, CancellationToken cancellationToken) =>
|
||||
Task.FromResult(new HttpResponseMessage(response.Status)
|
||||
{
|
||||
Content = new StringContent(response.Body, Encoding.UTF8, response.ContentType),
|
||||
});
|
||||
}
|
||||
|
||||
private sealed class ThrowingHandler(Exception exception) : HttpMessageHandler
|
||||
{
|
||||
protected override Task<HttpResponseMessage> SendAsync(
|
||||
HttpRequestMessage request, CancellationToken cancellationToken) =>
|
||||
Task.FromException<HttpResponseMessage>(exception);
|
||||
}
|
||||
|
||||
private sealed class HangingHandler : HttpMessageHandler
|
||||
{
|
||||
protected override async Task<HttpResponseMessage> SendAsync(
|
||||
HttpRequestMessage request, CancellationToken cancellationToken)
|
||||
{
|
||||
await Task.Delay(Timeout.InfiniteTimeSpan, cancellationToken).ConfigureAwait(false);
|
||||
return new HttpResponseMessage(HttpStatusCode.OK);
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user