feat(fleet): reconcile ClusterNode dial targets against live membership
ClusterNode.AkkaPort and the node's own Cluster:Port are the same fact in two places and nothing made them agree. Phase 2 dials the row instead of gossiping, so a node binding 4054 while its row says 4053 becomes unreachable from central — and the symptom is a silent absence of acks rather than an error. That is the shape of the Modbus/ModbusTcp and TwinCat/Focas drifts already in this repo. Since Phase 1 also made the rows the deploy path's expected-ack set, drift already costs a failed deployment today. Implemented as the plan's preferred option: an admin-role singleton comparing rows against the membership an admin node can already see, rather than each driver node asserting its own row — Phase 4 removes the driver nodes' ConfigDb connection, so a self-assertion written there would have to be deleted again. Three shapes, split by severity: a row whose dial target disagrees with its own NodeId and a running node with no row are Errors; an enabled row with no matching member is a Warning, because a node down for maintenance is a legitimate state. Findings are logged only when the set changes — a check that reprints the same warning every sweep trains operators to filter it out. Documented limitation: Phase 2 must revisit this. Once the fleet splits into one mesh per cluster an admin node cannot see site members, and every site row would report EnabledRowNotInCluster forever. Sabotage: removing the change-detection guard turns the repeat test red. ControlPlane.Tests 100/100. Claude-Session: https://claude.ai/code/session_01GASWkNEi68FSCtvr6rLoEW
This commit is contained in:
+98
@@ -0,0 +1,98 @@
|
||||
using Akka.Actor;
|
||||
using Microsoft.EntityFrameworkCore;
|
||||
using Xunit;
|
||||
using ZB.MOM.WW.OtOpcUa.Configuration;
|
||||
using ZB.MOM.WW.OtOpcUa.Configuration.Entities;
|
||||
using ZB.MOM.WW.OtOpcUa.Configuration.Enums;
|
||||
using ZB.MOM.WW.OtOpcUa.ControlPlane.Fleet;
|
||||
using ZB.MOM.WW.OtOpcUa.ControlPlane.Tests.Harness;
|
||||
|
||||
namespace ZB.MOM.WW.OtOpcUa.ControlPlane.Tests;
|
||||
|
||||
/// <summary>
|
||||
/// Proves the reconciler is actually wired to something — subscribes to membership, reads the DB,
|
||||
/// and emits. <see cref="ClusterNodeAddressReconcilerTests"/> covers the comparison logic, which a
|
||||
/// dormant actor would leave perfectly correct and completely inert.
|
||||
/// </summary>
|
||||
public sealed class ClusterNodeAddressReconcilerActorTests : ControlPlaneActorTestBase
|
||||
{
|
||||
/// <summary>
|
||||
/// An enabled row with no matching driver member is warned about. The harness ActorSystem
|
||||
/// joins as <c>admin</c> with no driver members, so the seeded row is unmatched by
|
||||
/// construction.
|
||||
/// </summary>
|
||||
[Fact]
|
||||
public void Enabled_row_with_no_matching_member_is_logged()
|
||||
{
|
||||
var dbFactory = NewInMemoryDbFactory();
|
||||
SeedNode(dbFactory, "site-a-1:4053", enabled: true);
|
||||
|
||||
EventFilter.Warning(contains: "site-a-1:4053").ExpectOne(TimeSpan.FromSeconds(10), () =>
|
||||
Sys.ActorOf(ClusterNodeAddressReconcilerActor.Props(dbFactory)));
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// The finding set is logged once, not once per sweep. A check that reprints the same warning
|
||||
/// every five minutes trains operators to filter it out, which costs more than it catches —
|
||||
/// so this runs a deliberately fast sweep and asserts the count stays at one.
|
||||
/// </summary>
|
||||
[Fact]
|
||||
public void Unchanged_findings_are_not_repeated_on_every_sweep()
|
||||
{
|
||||
var dbFactory = NewInMemoryDbFactory();
|
||||
SeedNode(dbFactory, "site-a-1:4053", enabled: true);
|
||||
|
||||
// ~6 sweeps inside the window; without the change-detection guard this logs 6 times.
|
||||
EventFilter.Warning(contains: "site-a-1:4053").Expect(1, TimeSpan.FromSeconds(3), () =>
|
||||
{
|
||||
Sys.ActorOf(ClusterNodeAddressReconcilerActor.Props(
|
||||
dbFactory, sweepInterval: TimeSpan.FromMilliseconds(400)));
|
||||
// Hold the filter open past several sweeps rather than returning immediately.
|
||||
ExpectNoMsg(TimeSpan.FromSeconds(2.5));
|
||||
});
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// A disabled row is silence — the maintenance hatch must not trade a failed deployment for a
|
||||
/// permanent warning. Positive control for the test above: same setup, one flag flipped.
|
||||
/// </summary>
|
||||
[Fact]
|
||||
public void Disabled_row_produces_no_warning()
|
||||
{
|
||||
var dbFactory = NewInMemoryDbFactory();
|
||||
SeedNode(dbFactory, "site-a-1:4053", enabled: false);
|
||||
|
||||
EventFilter.Warning(contains: "site-a-1:4053").Expect(0, TimeSpan.FromSeconds(2), () =>
|
||||
{
|
||||
Sys.ActorOf(ClusterNodeAddressReconcilerActor.Props(
|
||||
dbFactory, sweepInterval: TimeSpan.FromMilliseconds(400)));
|
||||
ExpectNoMsg(TimeSpan.FromSeconds(1.5));
|
||||
});
|
||||
}
|
||||
|
||||
private static void SeedNode(
|
||||
IDbContextFactory<OtOpcUaConfigDbContext> dbFactory, string nodeId, bool enabled)
|
||||
{
|
||||
using var db = dbFactory.CreateDbContext();
|
||||
db.ServerClusters.Add(new ServerCluster
|
||||
{
|
||||
ClusterId = "SITE-A",
|
||||
Name = "Site A",
|
||||
Enterprise = "zb",
|
||||
Site = "site-a",
|
||||
NodeCount = 2,
|
||||
RedundancyMode = RedundancyMode.Warm,
|
||||
CreatedBy = "test",
|
||||
});
|
||||
db.ClusterNodes.Add(new ClusterNode
|
||||
{
|
||||
NodeId = nodeId,
|
||||
ClusterId = "SITE-A",
|
||||
Host = nodeId.Split(':')[0],
|
||||
ApplicationUri = $"urn:OtOpcUa:{nodeId}",
|
||||
Enabled = enabled,
|
||||
CreatedBy = "test",
|
||||
});
|
||||
db.SaveChanges();
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user