using Akka.Actor; using Akka.Cluster.Tools.Singleton; using Xunit; namespace ZB.MOM.WW.ScadaBridge.IntegrationTests.Cluster; /// /// Behavioral proof that the SBR downing provider enabled in /// AkkaHostedService.BuildHocon (arch-review 01 Critical) is actually active: /// after a hard crash, the surviving node DOWNS and REMOVES the crashed member. Under /// the pre-fix Akka default (NoDowning) the crashed member lingers Unreachable /// forever, so the member-removal assertions here are impossible to satisfy without the /// fix — that is what gives the test teeth. /// /// IMPORTANT — keep-oldest two-node semantics (verified empirically, Akka 1.5.62): /// SBR downs the partition that does NOT contain the oldest member. So the crash that /// SBR can recover from in a two-node cluster is the crash of the YOUNGER node — the /// oldest survives and keeps its singletons. Crashing the OLDEST node instead makes the /// younger survivor down ITSELF (total cluster loss); down-if-alone=on does not /// change this on a hard crash because the alone-oldest is no longer running to down /// itself. That asymmetry (active/oldest-node crash is NOT covered by two-node /// keep-oldest) is a design-level gap tracked separately, not something this test can /// assert as a success path. /// public class SbrFailoverTests { private sealed class EchoActor : ReceiveActor { public EchoActor() => ReceiveAny(msg => Sender.Tell(msg)); } private static (IActorRef manager, IActorRef proxy) StartSingleton(ActorSystem sys) { var manager = sys.ActorOf(ClusterSingletonManager.Props( Props.Create(() => new EchoActor()), PoisonPill.Instance, ClusterSingletonManagerSettings.Create(sys).WithSingletonName("failover-probe")), "failover-probe-singleton"); var proxy = sys.ActorOf(ClusterSingletonProxy.Props( "/user/failover-probe-singleton", ClusterSingletonProxySettings.Create(sys).WithSingletonName("failover-probe")), "failover-probe-proxy"); return (manager, proxy); } [Fact] public async Task HardCrashOfYoungerNode_SbrDownsIt_AndOldestKeepsSingleton() { await using var cluster = await TwoNodeClusterFixture.StartAsync(); var (_, proxyA) = StartSingleton(cluster.NodeA); // oldest hosts the singleton StartSingleton(cluster.NodeB); // Singleton is reachable from A (the oldest / singleton host). var echo = await proxyA.Ask("ping", TimeSpan.FromSeconds(20)); Assert.Equal("ping", echo); var victimAddress = Akka.Cluster.Cluster.Get(cluster.NodeB).SelfAddress; await TwoNodeClusterFixture.CrashNode(cluster.NodeB); // 1) SBR must DOWN and REMOVE the crashed younger member (NoDowning => this // times out; the member stays unreachable forever). // Budget: failure detection (~2s) + stable-after (3s) + gossip margin. await TwoNodeClusterFixture.WaitForMemberRemoved( cluster.NodeA, victimAddress, TimeSpan.FromSeconds(30)); // 2) The oldest survivor stays up and its singleton keeps answering. var deadline = DateTime.UtcNow + TimeSpan.FromSeconds(30); Exception? last = null; while (DateTime.UtcNow < deadline) { try { var echo2 = await proxyA.Ask("ping-after-crash", TimeSpan.FromSeconds(3)); Assert.Equal("ping-after-crash", echo2); return; } catch (Exception ex) { last = ex; } } throw new Xunit.Sdk.XunitException($"Singleton stopped answering on the surviving oldest node after SBR downing: {last}"); } }