using Akka.Actor;
using Akka.Cluster.Tools.Singleton;
using Xunit;
namespace ZB.MOM.WW.ScadaBridge.IntegrationTests.Cluster;
///
/// Behavioral proof that the SBR downing provider enabled in
/// AkkaHostedService.BuildHocon (arch-review 01 Critical) is actually active:
/// after a hard crash, the surviving node DOWNS and REMOVES the crashed member. Under
/// the pre-fix Akka default (NoDowning) the crashed member lingers Unreachable
/// forever, so the member-removal assertions here are impossible to satisfy without the
/// fix — that is what gives the test teeth.
///
/// IMPORTANT — keep-oldest two-node semantics (verified empirically, Akka 1.5.62):
/// SBR downs the partition that does NOT contain the oldest member. So the crash that
/// SBR can recover from in a two-node cluster is the crash of the YOUNGER node — the
/// oldest survives and keeps its singletons. Crashing the OLDEST node instead makes the
/// younger survivor down ITSELF (total cluster loss); down-if-alone=on does not
/// change this on a hard crash because the alone-oldest is no longer running to down
/// itself. That asymmetry (active/oldest-node crash is NOT covered by two-node
/// keep-oldest) is a design-level gap tracked separately, not something this test can
/// assert as a success path.
///
public class SbrFailoverTests
{
private sealed class EchoActor : ReceiveActor
{
public EchoActor() => ReceiveAny(msg => Sender.Tell(msg));
}
private static (IActorRef manager, IActorRef proxy) StartSingleton(ActorSystem sys)
{
var manager = sys.ActorOf(ClusterSingletonManager.Props(
Props.Create(() => new EchoActor()),
PoisonPill.Instance,
ClusterSingletonManagerSettings.Create(sys).WithSingletonName("failover-probe")),
"failover-probe-singleton");
var proxy = sys.ActorOf(ClusterSingletonProxy.Props(
"/user/failover-probe-singleton",
ClusterSingletonProxySettings.Create(sys).WithSingletonName("failover-probe")),
"failover-probe-proxy");
return (manager, proxy);
}
[Fact]
public async Task HardCrashOfYoungerNode_SbrDownsIt_AndOldestKeepsSingleton()
{
await using var cluster = await TwoNodeClusterFixture.StartAsync();
var (_, proxyA) = StartSingleton(cluster.NodeA); // oldest hosts the singleton
StartSingleton(cluster.NodeB);
// Singleton is reachable from A (the oldest / singleton host).
var echo = await proxyA.Ask("ping", TimeSpan.FromSeconds(20));
Assert.Equal("ping", echo);
var victimAddress = Akka.Cluster.Cluster.Get(cluster.NodeB).SelfAddress;
await TwoNodeClusterFixture.CrashNode(cluster.NodeB);
// 1) SBR must DOWN and REMOVE the crashed younger member (NoDowning => this
// times out; the member stays unreachable forever).
// Budget: failure detection (~2s) + stable-after (3s) + gossip margin.
await TwoNodeClusterFixture.WaitForMemberRemoved(
cluster.NodeA, victimAddress, TimeSpan.FromSeconds(30));
// 2) The oldest survivor stays up and its singleton keeps answering.
var deadline = DateTime.UtcNow + TimeSpan.FromSeconds(30);
Exception? last = null;
while (DateTime.UtcNow < deadline)
{
try
{
var echo2 = await proxyA.Ask("ping-after-crash", TimeSpan.FromSeconds(3));
Assert.Equal("ping-after-crash", echo2);
return;
}
catch (Exception ex) { last = ex; }
}
throw new Xunit.Sdk.XunitException($"Singleton stopped answering on the surviving oldest node after SBR downing: {last}");
}
}