diff --git a/src/ZB.MOM.WW.ScadaBridge.ClusterInfrastructure/ClusterBootstrapGuard.cs b/src/ZB.MOM.WW.ScadaBridge.ClusterInfrastructure/ClusterBootstrapGuard.cs new file mode 100644 index 00000000..cd0fb8e1 --- /dev/null +++ b/src/ZB.MOM.WW.ScadaBridge.ClusterInfrastructure/ClusterBootstrapGuard.cs @@ -0,0 +1,155 @@ +using System.Text.RegularExpressions; + +namespace ZB.MOM.WW.ScadaBridge.ClusterInfrastructure; + +/// +/// Pure decision logic for the simultaneous-cold-start split-brain guard (Gitea #33). +/// +/// +/// +/// Both nodes of a 2-node pair are self-first seeds so either can cold-start alone when +/// its partner is dead (see ). The cost is that when +/// BOTH cold-start at the same instant, each runs Akka's FirstSeedNodeProcess, times out +/// waiting for the other, and forms its OWN single-node cluster — a split brain (two oldest +/// nodes, two singletons, dual-active). Two independent clusters do not auto-merge, so the +/// split persists until an operator restarts one side. It bites on any event that powers up +/// both site VMs together (site power restoration, hypervisor host reboot). +/// +/// +/// This guard breaks the symmetry deterministically without giving up cold-start-alone. The +/// node with the lexicographically lower canonical address is the preferred +/// founder: it always uses self-first order and forms immediately. The higher node +/// first probes whether its partner's Akka endpoint is reachable (see +/// ): reachable ⇒ peer-first order so it JOINS the +/// founder rather than racing it; unreachable after the probe window ⇒ the partner is genuinely +/// down, so it falls back to self-first and forms alone. Exactly one node founds when both are +/// present; the lower node always founds when it starts alone. +/// +/// +/// Residual trade-off (accepted). Once the higher node observes the partner reachable it +/// commits to peer-first — Akka's JoinSeedNodeProcess retries InitJoin forever and +/// never self-forms. If the founder dies in the small window between the probe succeeding and +/// the join completing, the higher node hangs unjoined until it is restarted (a restart re-runs +/// the guard, finds the founder unreachable, and self-forms). We deliberately do NOT re-decide +/// mid-handshake — that is exactly the retired SelfFormAfter failure mode. Instead +/// makes the hang operator-visible with a warning if +/// the node has not come Up within a bounded time after committing peer-first. +/// +/// +/// This decides the join order BEFORE issuing a single JoinSeedNodes, from an explicit +/// reachability signal — unlike a timer-based watchdog, which fires a Join(self) +/// mid-handshake on a bare timeout and could not tell "no seed answered" from "a join is in +/// flight", islanding a node a failover had just bounced (the rejected self-form-watchdog +/// design; see SelfFirstSeedBootstrapTests). +/// +/// +public static class ClusterBootstrapGuard +{ + private static readonly Regex SeedPattern = + new(@"^akka(?:\.[a-z0-9]+)?://[^@/]+@(?[^:/]+):(?\d+)", RegexOptions.IgnoreCase | RegexOptions.Compiled); + + /// The analyzed bootstrap role of this node within its seed list. + /// True only when the guard engages: exactly two seeds, one of them this node. + /// True when this node is the preferred founder (lower address) — form immediately, no probe. + /// The partner's advertised host (to probe when this node is the higher one); null when not applicable. + /// The partner's Akka port. + /// [self, partner] — self-first; forms a new cluster when no peer answers. + /// [partner, self] — peer-first; joins the partner's cluster, never self-forms while it is up. + public sealed record BootstrapRole( + bool Applies, + bool IsFounder, + string? PartnerHost, + int PartnerPort, + string[] SelfFirstOrder, + string[] PeerFirstOrder); + + /// + /// Analyzes the configured seed list to decide this node's bootstrap role. The guard engages only + /// for the pair shape (exactly two seeds, one of them this node); any other shape + /// (single-seed / single-node install, three+ seeds, self absent) returns + /// = false and the caller joins the configured seeds unchanged. + /// + /// The configured seed URIs (self-first by convention, but order is not relied on here). + /// This node's advertised host (NodeOptions.NodeHostname). + /// This node's Akka remoting port (NodeOptions.RemotingPort). + /// The analyzed role; never null. + public static BootstrapRole Analyze(IReadOnlyList seeds, string selfHost, int selfPort) + { + ArgumentNullException.ThrowIfNull(seeds); + ArgumentException.ThrowIfNullOrWhiteSpace(selfHost); + + var na = new BootstrapRole(false, false, null, 0, Array.Empty(), Array.Empty()); + if (seeds.Count != 2) return na; + + var selfKey = Key(selfHost, selfPort); + string? self = null, partner = null; + string? partnerHost = null; + var partnerPort = 0; + + foreach (var seed in seeds) + { + if (!TryParse(seed, out var host, out var port)) return na; + if (string.Equals(Key(host, port), selfKey, StringComparison.OrdinalIgnoreCase)) + self = seed; + else + { + partner = seed; + partnerHost = host; + partnerPort = port; + } + } + + // Self must be exactly one of the two, and the other must be a distinct partner. + if (self is null || partner is null || partnerHost is null) return na; + + var selfFirst = new[] { self, partner }; + var peerFirst = new[] { partner, self }; + + // Deterministic tie-break: the lower canonical address is the preferred founder. Both nodes + // compute the same comparison (same two seeds), so exactly one is the founder. Case-INSENSITIVE + // to match how self/partner are classified above (and StartupValidator.SeedNodeIsSelf): + // container/DNS hostnames are conventionally case-insensitive, and a casing difference between + // the two sides' seed config must NOT make both think they are the founder (which would reopen + // the very split this guard closes). + var isFounder = string.Compare( + Key(selfHost, selfPort), + Key(partnerHost, partnerPort), + StringComparison.OrdinalIgnoreCase) < 0; + + return new BootstrapRole(true, isFounder, partnerHost, partnerPort, selfFirst, peerFirst); + } + + /// + /// The order the higher (non-founder) node uses once it has learned whether its partner is + /// reachable: peer-first when the founder is up (join it), self-first when the founder is absent + /// (form alone — cold-start-alone preserved). + /// + /// The analyzed role (must be applicable and NOT the founder). + /// Whether the partner's Akka endpoint became reachable within the probe window. + /// The seed order to pass to Cluster.JoinSeedNodes. + public static string[] HigherNodeOrder(BootstrapRole role, bool partnerReachable) + { + ArgumentNullException.ThrowIfNull(role); + return partnerReachable ? role.PeerFirstOrder : role.SelfFirstOrder; + } + + /// Parses akka.tcp://system@host:port[/...] into host + port. + /// The seed URI. + /// The parsed advertised host. + /// The parsed Akka port. + /// True when the seed matched the expected shape. + public static bool TryParse(string? seed, out string host, out int port) + { + host = string.Empty; + port = 0; + if (string.IsNullOrWhiteSpace(seed)) return false; + + var m = SeedPattern.Match(seed.Trim()); + if (!m.Success) return false; + if (!int.TryParse(m.Groups["port"].Value, out port)) return false; + host = m.Groups["host"].Value; + return true; + } + + private static string Key(string host, int port) => $"{host}:{port}"; +} diff --git a/src/ZB.MOM.WW.ScadaBridge.ClusterInfrastructure/ClusterOptions.cs b/src/ZB.MOM.WW.ScadaBridge.ClusterInfrastructure/ClusterOptions.cs index daf95107..87b5952e 100644 --- a/src/ZB.MOM.WW.ScadaBridge.ClusterInfrastructure/ClusterOptions.cs +++ b/src/ZB.MOM.WW.ScadaBridge.ClusterInfrastructure/ClusterOptions.cs @@ -114,4 +114,50 @@ public class ClusterOptions /// dials forever — log noise and a defeated validation intent (review 01). /// public bool AllowSingleNodeCluster { get; set; } + + /// + /// The simultaneous-cold-start split-brain guard (ScadaBridge:Cluster:BootstrapGuard). + /// Default OFF — a dark switch, so existing deployments and tests keep Akka's config-driven + /// self-first auto-join byte-identical. See for the + /// decision logic and ClusterBootstrapCoordinator for the runtime (Gitea #33). + /// + public ClusterBootstrapGuardOptions BootstrapGuard { get; set; } = new(); +} + +/// +/// Configuration for the simultaneous-cold-start split-brain guard. Both nodes of a pair are +/// self-first seeds so either can cold-start alone when its partner is dead — but the cost +/// is that when BOTH cold-start at the same instant each runs Akka's FirstSeedNodeProcess, +/// times out waiting for the other, and forms its own single-node cluster (a split brain that does +/// not auto-merge). When , the node does NOT auto-join from its config seeds +/// (AkkaHostedService.BuildHocon emits an empty seed list); a coordinator picks the join +/// order — founder self-first / joiner peer-first — after a reachability probe. See +/// . +/// +public sealed class ClusterBootstrapGuardOptions +{ + /// + /// Whether the guard is active. Default — Akka auto-joins from the + /// config seed list exactly as before. Only meaningful on a node that is one of its own two + /// pair seeds; inert everywhere else (single-seed site node, self absent, 3+ seeds). + /// + public bool Enabled { get; set; } + + /// + /// How long the higher-address node probes its partner's Akka endpoint before concluding the + /// partner is dead and forming alone. Must comfortably exceed the partner's worst-case + /// process-start-to-Akka-bind time so a slow-but-alive partner is never mistaken for a dead one + /// (which would re-open the split). Default 25s. Validated > 0 at boot when the guard is on. + /// + public int PartnerProbeSeconds { get; set; } = 25; + + /// + /// The interval between partner reachability probes, in milliseconds. Default 500ms. + /// + public int PartnerProbeIntervalMs { get; set; } = 500; + + /// + /// The per-probe TCP connect timeout, in milliseconds. Default 1000ms. + /// + public int ProbeConnectTimeoutMs { get; set; } = 1000; } diff --git a/src/ZB.MOM.WW.ScadaBridge.ClusterInfrastructure/ClusterOptionsValidator.cs b/src/ZB.MOM.WW.ScadaBridge.ClusterInfrastructure/ClusterOptionsValidator.cs index 75e7e9ec..26f26e05 100644 --- a/src/ZB.MOM.WW.ScadaBridge.ClusterInfrastructure/ClusterOptionsValidator.cs +++ b/src/ZB.MOM.WW.ScadaBridge.ClusterInfrastructure/ClusterOptionsValidator.cs @@ -30,6 +30,8 @@ public sealed class ClusterOptionsValidator : OptionsValidatorBase protected override void Validate(ValidationBuilder builder, ClusterOptions options) { + ValidateBootstrapGuard(builder, options.BootstrapGuard); + // The design doc states "both nodes are seed nodes — each node lists // both itself and its partner" so a properly-configured deployment lists // two. Accepting a single-seed configuration silently defeats the @@ -78,4 +80,33 @@ public sealed class ClusterOptionsValidator : OptionsValidatorBase + /// When the bootstrap guard is enabled (Gitea #33), its timing knobs must be positive. A + /// zero/negative in particular + /// silently degrades the guard to "never wait, always conclude the partner is dead" — the + /// higher node would form alone immediately and re-open the very split the guard exists to + /// close. Fail fast at boot rather than producing that silent degradation. Nothing is checked + /// when the guard is off (the knobs are inert), so a disabled guard never blocks a boot. + /// + private static void ValidateBootstrapGuard(ValidationBuilder builder, ClusterBootstrapGuardOptions? guard) + { + if (guard is null || !guard.Enabled) + { + return; + } + + builder.RequireThat(guard.PartnerProbeSeconds > 0, + $"ClusterOptions.BootstrapGuard.PartnerProbeSeconds must be > 0 when the guard is enabled " + + $"(was {guard.PartnerProbeSeconds}); a non-positive value makes the higher node conclude its " + + "partner is dead without waiting and form alone, re-opening the split the guard prevents."); + + builder.RequireThat(guard.PartnerProbeIntervalMs > 0, + $"ClusterOptions.BootstrapGuard.PartnerProbeIntervalMs must be > 0 when the guard is enabled " + + $"(was {guard.PartnerProbeIntervalMs})."); + + builder.RequireThat(guard.ProbeConnectTimeoutMs > 0, + $"ClusterOptions.BootstrapGuard.ProbeConnectTimeoutMs must be > 0 when the guard is enabled " + + $"(was {guard.ProbeConnectTimeoutMs})."); + } } diff --git a/src/ZB.MOM.WW.ScadaBridge.Host/Actors/AkkaHostedService.cs b/src/ZB.MOM.WW.ScadaBridge.Host/Actors/AkkaHostedService.cs index ae848762..d2d8bfa5 100644 --- a/src/ZB.MOM.WW.ScadaBridge.Host/Actors/AkkaHostedService.cs +++ b/src/ZB.MOM.WW.ScadaBridge.Host/Actors/AkkaHostedService.cs @@ -261,8 +261,16 @@ public class AkkaHostedService : IHostedService TimeSpan transportHeartbeat, TimeSpan transportFailure) { - var seedNodesStr = string.Join(",", - clusterOptions.SeedNodes.Select(QuoteHocon)); + // Simultaneous-cold-start split-brain guard (Gitea #33): when enabled the node must NOT + // auto-join from its config seeds — Akka's FirstSeedNodeProcess is exactly what races the + // peer and forms two 1-node clusters. Emit an EMPTY seed list so the ActorSystem comes up + // unjoined, and ClusterBootstrapCoordinator issues the single reachability-gated + // JoinSeedNodes (founder self-first / joiner peer-first). Default OFF, so guard-off configs + // render byte-identical to before. + var effectiveSeeds = clusterOptions.BootstrapGuard.Enabled + ? Enumerable.Empty() + : clusterOptions.SeedNodes; + var seedNodesStr = string.Join(",", effectiveSeeds.Select(QuoteHocon)); var rolesStr = string.Join(",", roles.Select(QuoteHocon)); // auto-down (default): AutoDowning provider — the leader among the reachable diff --git a/src/ZB.MOM.WW.ScadaBridge.Host/Actors/ClusterBootstrapCoordinator.cs b/src/ZB.MOM.WW.ScadaBridge.Host/Actors/ClusterBootstrapCoordinator.cs new file mode 100644 index 00000000..93307610 --- /dev/null +++ b/src/ZB.MOM.WW.ScadaBridge.Host/Actors/ClusterBootstrapCoordinator.cs @@ -0,0 +1,211 @@ +using System.Diagnostics; +using System.Net.Sockets; +using Akka.Actor; +using Akka.Cluster; +using Microsoft.Extensions.Options; +using ZB.MOM.WW.ScadaBridge.ClusterInfrastructure; + +namespace ZB.MOM.WW.ScadaBridge.Host.Actors; + +/// +/// Runs the simultaneous-cold-start split-brain guard (Gitea #33): when +/// ScadaBridge:Cluster:BootstrapGuard:Enabled, the node starts with NO config seed nodes +/// (AkkaHostedService.BuildHocon emits an empty seed list), and this coordinator picks the join +/// order and issues the single once the ActorSystem is +/// up. See for the decision logic and the split-brain rationale. +/// +/// +/// +/// The founder (lower canonical address) joins its self-first order immediately. The higher +/// node probes its partner's Akka endpoint (a plain TCP connect — the partner has bound its +/// port well before it forms) up to : +/// reachable ⇒ peer-first (join the founder); unreachable ⇒ self-first (partner is dead, form +/// alone). The join runs on a background task so it never blocks host startup, and it is issued +/// exactly once — the coordinator never re-forms a node mid-handshake, the failure mode of the +/// rejected self-form-watchdog design (see SelfFirstSeedBootstrapTests). +/// +/// +/// Registered unconditionally in both the Central and Site composition roots but a no-op unless +/// the guard is enabled, so guard-off deployments keep Akka's config-driven self-first +/// auto-join byte-identical. +/// +/// +public sealed class ClusterBootstrapCoordinator : IHostedService +{ + private readonly Func _system; + private readonly ClusterOptions _clusterOptions; + private readonly NodeOptions _nodeOptions; + private readonly ILogger _log; + private readonly CancellationTokenSource _cts = new(); + private Task? _joinTask; + + /// Creates the coordinator. + /// Lazy accessor for the node's ActorSystem (resolved after the Akka hosted service creates it). + /// The bound cluster options (seed list + guard settings). + /// The bound node identity (advertised hostname + remoting port). + /// Logger for the bootstrap decision. + public ClusterBootstrapCoordinator( + Func system, + IOptions clusterOptions, + IOptions nodeOptions, + ILogger log) + { + _system = system ?? throw new ArgumentNullException(nameof(system)); + _clusterOptions = clusterOptions?.Value ?? throw new ArgumentNullException(nameof(clusterOptions)); + _nodeOptions = nodeOptions?.Value ?? throw new ArgumentNullException(nameof(nodeOptions)); + _log = log ?? throw new ArgumentNullException(nameof(log)); + } + + /// + public Task StartAsync(CancellationToken cancellationToken) + { + if (!_clusterOptions.BootstrapGuard.Enabled) + return Task.CompletedTask; // dark switch off — Akka auto-joined from config seeds already. + + // Fire-and-forget so a higher node with a dead partner (which waits the full probe window) + // never blocks host startup. Exceptions are logged; a failed join leaves the node unjoined, + // which is visible and recoverable, never a silent split. + _joinTask = Task.Run(() => RunAsync(_cts.Token), _cts.Token); + return Task.CompletedTask; + } + + /// + public async Task StopAsync(CancellationToken cancellationToken) + { + await _cts.CancelAsync().ConfigureAwait(false); + if (_joinTask is not null) + { + try { await _joinTask.ConfigureAwait(false); } + catch (OperationCanceledException) { /* expected on shutdown */ } + } + } + + private async Task RunAsync(CancellationToken ct) + { + try + { + var seeds = _clusterOptions.SeedNodes ?? new List(); + var role = ClusterBootstrapGuard.Analyze(seeds, _nodeOptions.NodeHostname, _nodeOptions.RemotingPort); + + string[] order; + var committedPeerFirst = false; + if (!role.Applies) + { + // Not a pair seed (single-node install, self absent, malformed, or 3+ seeds): join the + // configured seeds unchanged — the guard arbitrates only the 2-node pair race. + order = seeds.ToArray(); + _log.LogInformation( + "Bootstrap guard: not a pair seed ({SeedCount} seed(s)); joining configured seeds unchanged.", + seeds.Count); + } + else if (role.IsFounder) + { + order = role.SelfFirstOrder; + _log.LogInformation( + "Bootstrap guard: this node is the preferred founder (lower address); joining self-first, forming immediately if no peer answers."); + } + else + { + var reachable = await ProbePartnerAsync(role.PartnerHost!, role.PartnerPort, ct).ConfigureAwait(false); + order = ClusterBootstrapGuard.HigherNodeOrder(role, reachable); + committedPeerFirst = reachable; + _log.LogInformation( + "Bootstrap guard: this node is the higher address; partner {PartnerHost}:{PartnerPort} was {Reachability} within the probe window; joining {Order}.", + role.PartnerHost, role.PartnerPort, reachable ? "REACHABLE" : "UNREACHABLE", + reachable ? "peer-first (join the founder)" : "self-first (partner down, forming alone)"); + } + + if (order.Length == 0) + { + _log.LogWarning("Bootstrap guard: no seed nodes to join; node stays unjoined until a seed is configured."); + return; + } + + var cluster = Akka.Cluster.Cluster.Get(_system()); + var addresses = order.Select(Address.Parse).ToArray(); + cluster.JoinSeedNodes(addresses); + + // Peer-first is a one-way commitment: JoinSeedNodeProcess never self-forms. If the founder + // died in the probe→join window, this node hangs unjoined forever. We do NOT re-form here + // (the rejected mid-handshake failure mode) — we make the hang operator-visible so a + // restart, which re-runs the guard against the now-dead founder and self-forms, recovers + // it. Nothing is done on the founder / self-first paths, which always self-form on their own. + if (committedPeerFirst) + await WarnIfNotUpAsync(cluster, ct).ConfigureAwait(false); + } + catch (OperationCanceledException) + { + // Host shutting down before the join completed — nothing to do. + } + catch (Exception ex) + { + _log.LogError(ex, "Bootstrap guard: failed to issue JoinSeedNodes; node remains unjoined."); + } + } + + /// + /// After committing peer-first, waits a bounded time for this node to reach Up and logs a + /// clear warning if it does not — the founder must have died in the probe→join window, leaving this + /// node hung in JoinSeedNodeProcess. Read-only: it never re-forms or re-joins (that is the + /// rejected mid-handshake failure mode); it only surfaces the condition so an operator restarts the node. + /// + private async Task WarnIfNotUpAsync(Akka.Cluster.Cluster cluster, CancellationToken ct) + { + // Give the join a generous grace: the founder's own self-form (seed-node-timeout) plus this + // node's join round-trip. Two probe windows is comfortably beyond both. + var grace = TimeSpan.FromSeconds(Math.Max(10, _clusterOptions.BootstrapGuard.PartnerProbeSeconds * 2)); + var sw = Stopwatch.StartNew(); + while (!ct.IsCancellationRequested && sw.Elapsed < grace) + { + if (cluster.SelfMember.Status == MemberStatus.Up) return; // joined — all good + try { await Task.Delay(TimeSpan.FromSeconds(1), ct).ConfigureAwait(false); } + catch (OperationCanceledException) { return; } + } + + if (!ct.IsCancellationRequested && cluster.SelfMember.Status != MemberStatus.Up) + _log.LogWarning( + "Bootstrap guard: committed peer-first but this node is still {Status} after {Grace}s — its founder likely died in the probe→join window. This node will NOT self-form on its own (peer-first never does). RESTART it to recover: the guard will re-run, find the founder down, and form alone.", + cluster.SelfMember.Status, (int)grace.TotalSeconds); + } + + /// + /// Polls the partner's Akka endpoint with a plain TCP connect until it is reachable or the probe + /// window elapses. Reachable-at-TCP is a sound proxy for "the partner is coming up": a node binds + /// its Akka port near the start of startup, well before it forms or joins a cluster. + /// + private async Task ProbePartnerAsync(string host, int port, CancellationToken ct) + { + var window = TimeSpan.FromSeconds(Math.Max(0, _clusterOptions.BootstrapGuard.PartnerProbeSeconds)); + var interval = TimeSpan.FromMilliseconds(Math.Max(50, _clusterOptions.BootstrapGuard.PartnerProbeIntervalMs)); + var sw = Stopwatch.StartNew(); + + while (!ct.IsCancellationRequested) + { + if (await TryConnectAsync(host, port, ct).ConfigureAwait(false)) return true; + if (sw.Elapsed >= window) return false; + try { await Task.Delay(interval, ct).ConfigureAwait(false); } + catch (OperationCanceledException) { return false; } + } + return false; + } + + private async Task TryConnectAsync(string host, int port, CancellationToken ct) + { + try + { + using var client = new TcpClient(); + using var cts = CancellationTokenSource.CreateLinkedTokenSource(ct); + cts.CancelAfter(Math.Max(100, _clusterOptions.BootstrapGuard.ProbeConnectTimeoutMs)); + await client.ConnectAsync(host, port, cts.Token).ConfigureAwait(false); + return client.Connected; + } + catch (OperationCanceledException) when (ct.IsCancellationRequested) + { + throw; // host shutdown — propagate + } + catch + { + return false; // connect refused / timed out / DNS not resolvable yet — partner not up + } + } +} diff --git a/src/ZB.MOM.WW.ScadaBridge.Host/Program.cs b/src/ZB.MOM.WW.ScadaBridge.Host/Program.cs index 433d2990..02b30671 100644 --- a/src/ZB.MOM.WW.ScadaBridge.Host/Program.cs +++ b/src/ZB.MOM.WW.ScadaBridge.Host/Program.cs @@ -394,6 +394,20 @@ try builder.Services.AddSingleton(sp => sp.GetRequiredService().GetOrCreateActorSystem()); + // Simultaneous-cold-start split-brain guard (Gitea #33, dark switch + // ScadaBridge:Cluster:BootstrapGuard:Enabled, default off). Registered AFTER the + // AkkaHostedService/ActorSystem bridge so its StartAsync resolves the system the main path + // built; when the guard is off it no-ops (Akka auto-joined from config seeds), so registering + // it unconditionally is safe. When on, the node started unjoined (empty HOCON seed list, see + // AkkaHostedService.BuildHocon) and this coordinator issues the single reachability-gated + // JoinSeedNodes. It resolves the ActorSystem lazily via GetOrCreateActorSystem so it never + // races startup or caches a null. + builder.Services.AddHostedService(sp => new ClusterBootstrapCoordinator( + () => sp.GetRequiredService().GetOrCreateActorSystem(), + sp.GetRequiredService>(), + sp.GetRequiredService>(), + sp.GetRequiredService>())); + // Register the production IActiveNodeGate implementation so // standby-node gating is actually enforced (the InboundApiEndpointFilter // consults IActiveNodeGate and defaults to "allow" when none is registered, diff --git a/src/ZB.MOM.WW.ScadaBridge.Host/SiteServiceRegistration.cs b/src/ZB.MOM.WW.ScadaBridge.Host/SiteServiceRegistration.cs index f14aac2e..9734962a 100644 --- a/src/ZB.MOM.WW.ScadaBridge.Host/SiteServiceRegistration.cs +++ b/src/ZB.MOM.WW.ScadaBridge.Host/SiteServiceRegistration.cs @@ -168,6 +168,19 @@ public static class SiteServiceRegistration services.AddSingleton(sp => sp.GetRequiredService().GetOrCreateActorSystem()); + // Simultaneous-cold-start split-brain guard (Gitea #33, dark switch + // ScadaBridge:Cluster:BootstrapGuard:Enabled, default off). This is the case the guard was + // built for — a shared power event that powers up both site VMs together. Registered AFTER + // the AkkaHostedService/ActorSystem bridge (mirrors the Central composition root in + // Program.cs); it no-ops when the guard is off, so registering it unconditionally is safe. + // When on, the node started unjoined (empty HOCON seed list) and this coordinator issues the + // single reachability-gated JoinSeedNodes, resolving the ActorSystem lazily. + services.AddHostedService(sp => new ClusterBootstrapCoordinator( + () => sp.GetRequiredService().GetOrCreateActorSystem(), + sp.GetRequiredService>(), + sp.GetRequiredService>(), + sp.GetRequiredService>())); + // Cluster node status provider for health reports services.AddSingleton(sp => { diff --git a/tests/ZB.MOM.WW.ScadaBridge.ClusterInfrastructure.Tests/ClusterBootstrapGuardTests.cs b/tests/ZB.MOM.WW.ScadaBridge.ClusterInfrastructure.Tests/ClusterBootstrapGuardTests.cs new file mode 100644 index 00000000..e71dcbca --- /dev/null +++ b/tests/ZB.MOM.WW.ScadaBridge.ClusterInfrastructure.Tests/ClusterBootstrapGuardTests.cs @@ -0,0 +1,147 @@ +namespace ZB.MOM.WW.ScadaBridge.ClusterInfrastructure.Tests; + +/// +/// Unit tests for — the pure decision core of the +/// simultaneous-cold-start split-brain guard (Gitea #33). The runtime probe + JoinSeedNodes wiring +/// lives in ClusterBootstrapCoordinator and is covered by the real-ActorSystem +/// ClusterBootstrapCoordinatorTests in the integration suite. +/// +public class ClusterBootstrapGuardTests +{ + private const string A = "akka.tcp://scadabridge@scadabridge-site-a-a:8082"; + private const string B = "akka.tcp://scadabridge@scadabridge-site-a-b:8082"; + + [Fact] + public void Lower_address_node_is_the_founder_and_needs_no_probe() + { + // scadabridge-site-a-a < scadabridge-site-a-b, so on node-a the guard makes it the founder. + var role = ClusterBootstrapGuard.Analyze(new[] { A, B }, "scadabridge-site-a-a", 8082); + + Assert.True(role.Applies); + Assert.True(role.IsFounder); + Assert.Equal(new[] { A, B }, role.SelfFirstOrder); + } + + [Fact] + public void Higher_address_node_is_not_the_founder_and_targets_the_partner_for_probing() + { + // On node-b the partner is node-a (the lower/founder). + var role = ClusterBootstrapGuard.Analyze(new[] { B, A }, "scadabridge-site-a-b", 8082); + + Assert.True(role.Applies); + Assert.False(role.IsFounder); + Assert.Equal("scadabridge-site-a-a", role.PartnerHost); + Assert.Equal(8082, role.PartnerPort); + } + + [Fact] + public void Tie_break_is_symmetric_regardless_of_seed_order() + { + // Both nodes must agree on who founds no matter how each lists its seeds. + var onLower = ClusterBootstrapGuard.Analyze(new[] { A, B }, "scadabridge-site-a-a", 8082); + var onHigher = ClusterBootstrapGuard.Analyze(new[] { B, A }, "scadabridge-site-a-b", 8082); + + Assert.True(onLower.IsFounder); + Assert.False(onHigher.IsFounder); + // Exactly one founder. + Assert.True(onLower.IsFounder ^ onHigher.IsFounder); + } + + [Fact] + public void Tie_break_is_case_insensitive_so_a_casing_mismatch_cannot_make_both_founders() + { + // The two sides' seed config differ only in host casing (review note 1): if the tie-break + // were case-sensitive, both could conclude they are the founder and re-open the split. + var lowerSelf = "akka.tcp://scadabridge@SITE-A-A:8082"; + var higherSelf = "akka.tcp://scadabridge@site-a-b:8082"; + + var onLower = ClusterBootstrapGuard.Analyze(new[] { lowerSelf, higherSelf }, "SITE-A-A", 8082); + var onHigher = ClusterBootstrapGuard.Analyze(new[] { higherSelf, lowerSelf }, "site-a-b", 8082); + + Assert.True(onLower.IsFounder ^ onHigher.IsFounder); + Assert.True(onLower.IsFounder); // "site-a-a" < "site-a-b" ordinal-ignore-case + } + + [Fact] + public void Higher_node_joins_the_founder_when_partner_is_reachable() + { + var role = ClusterBootstrapGuard.Analyze(new[] { B, A }, "scadabridge-site-a-b", 8082); + + // Reachable ⇒ peer-first ⇒ JoinSeedNodeProcess ⇒ joins the founder, never self-forms. + Assert.Equal(new[] { A, B }, ClusterBootstrapGuard.HigherNodeOrder(role, partnerReachable: true)); // partner (node-a) first + } + + [Fact] + public void Higher_node_forms_alone_when_partner_is_unreachable() + { + var role = ClusterBootstrapGuard.Analyze(new[] { B, A }, "scadabridge-site-a-b", 8082); + + // Unreachable after the window ⇒ partner is dead ⇒ self-first ⇒ cold-start-alone preserved. + Assert.Equal(new[] { B, A }, ClusterBootstrapGuard.HigherNodeOrder(role, partnerReachable: false)); // self (node-b) first + } + + [Fact] + public void Guard_does_not_apply_to_a_single_seed_node() + { + // A single-node install lists exactly one seed (itself) — not a pair, guard is inert. + var role = ClusterBootstrapGuard.Analyze(new[] { "akka.tcp://scadabridge@lone:8081" }, "lone", 8081); + + Assert.False(role.Applies); + } + + [Fact] + public void Guard_does_not_apply_when_self_is_absent_from_the_two_seeds() + { + var role = ClusterBootstrapGuard.Analyze(new[] { A, B }, "scadabridge-central-a", 8081); + + Assert.False(role.Applies); + } + + [Fact] + public void Guard_does_not_apply_to_three_or_more_seeds() + { + var role = ClusterBootstrapGuard.Analyze( + new[] { A, B, "akka.tcp://scadabridge@scadabridge-site-a-c:8082" }, "scadabridge-site-a-a", 8082); + + Assert.False(role.Applies); + } + + [Fact] + public void Guard_does_not_apply_when_a_seed_is_malformed() + { + var role = ClusterBootstrapGuard.Analyze(new[] { A, "not-a-seed" }, "scadabridge-site-a-a", 8082); + + Assert.False(role.Applies); + } + + [Theory] + [InlineData("akka.tcp://scadabridge@scadabridge-site-a-a:8082", "scadabridge-site-a-a", 8082)] + [InlineData("akka.tcp://scadabridge@10.0.0.5:8082/", "10.0.0.5", 8082)] + [InlineData(" akka.tcp://scadabridge@host.lan:1234 ", "host.lan", 1234)] + public void TryParse_extracts_host_and_port(string seed, string expectedHost, int expectedPort) + { + Assert.True(ClusterBootstrapGuard.TryParse(seed, out var host, out var port)); + Assert.Equal(expectedHost, host); + Assert.Equal(expectedPort, port); + } + + [Theory] + [InlineData("")] + [InlineData("garbage")] + [InlineData("akka.tcp://scadabridge@host")] // no port + public void TryParse_rejects_malformed_seeds(string seed) + { + Assert.False(ClusterBootstrapGuard.TryParse(seed, out _, out _)); + } + + [Fact] + public void Port_participates_in_the_tie_break_when_hosts_are_equal() + { + // Shared-host loopback pair (test rig): same host, different ports — port breaks the tie. + const string low = "akka.tcp://scadabridge@127.0.0.1:8081"; + const string high = "akka.tcp://scadabridge@127.0.0.1:8082"; + + Assert.True(ClusterBootstrapGuard.Analyze(new[] { low, high }, "127.0.0.1", 8081).IsFounder); + Assert.False(ClusterBootstrapGuard.Analyze(new[] { high, low }, "127.0.0.1", 8082).IsFounder); + } +} diff --git a/tests/ZB.MOM.WW.ScadaBridge.ClusterInfrastructure.Tests/ClusterOptionsValidatorTests.cs b/tests/ZB.MOM.WW.ScadaBridge.ClusterInfrastructure.Tests/ClusterOptionsValidatorTests.cs index ffec614c..7430a854 100644 --- a/tests/ZB.MOM.WW.ScadaBridge.ClusterInfrastructure.Tests/ClusterOptionsValidatorTests.cs +++ b/tests/ZB.MOM.WW.ScadaBridge.ClusterInfrastructure.Tests/ClusterOptionsValidatorTests.cs @@ -215,4 +215,68 @@ public class ClusterOptionsValidatorTests Assert.Contains("StableAfter", result.FailureMessage); Assert.Contains("HeartbeatInterval", result.FailureMessage); } + + // ---- BootstrapGuard timing validation (Gitea #33) ---- + + [Fact] + public void BootstrapGuard_disabled_with_zero_timings_passes_validation() + { + // The knobs are inert when the guard is off, so a disabled guard must never block a boot — + // even if the timing values are nonsense (0 here). + var options = ValidOptions(); + options.BootstrapGuard = new ClusterBootstrapGuardOptions + { + Enabled = false, + PartnerProbeSeconds = 0, + PartnerProbeIntervalMs = 0, + ProbeConnectTimeoutMs = 0, + }; + + var result = new ClusterOptionsValidator().Validate(null, options); + + Assert.True(result.Succeeded, result.FailureMessage); + } + + [Fact] + public void BootstrapGuard_enabled_with_default_timings_passes_validation() + { + var options = ValidOptions(); + options.BootstrapGuard = new ClusterBootstrapGuardOptions { Enabled = true }; + + var result = new ClusterOptionsValidator().Validate(null, options); + + Assert.True(result.Succeeded, result.FailureMessage); + } + + [Fact] + public void BootstrapGuard_enabled_with_nonpositive_probe_seconds_fails_validation() + { + // A zero PartnerProbeSeconds silently degrades the guard to "never wait, always conclude the + // partner is dead" — the higher node forms alone immediately and re-opens the split. Reject it. + var options = ValidOptions(); + options.BootstrapGuard = new ClusterBootstrapGuardOptions { Enabled = true, PartnerProbeSeconds = 0 }; + + var result = new ClusterOptionsValidator().Validate(null, options); + + Assert.True(result.Failed); + Assert.Contains("PartnerProbeSeconds", result.FailureMessage); + } + + [Fact] + public void BootstrapGuard_enabled_with_nonpositive_interval_or_timeout_fails_validation() + { + var options = ValidOptions(); + options.BootstrapGuard = new ClusterBootstrapGuardOptions + { + Enabled = true, + PartnerProbeIntervalMs = 0, + ProbeConnectTimeoutMs = -1, + }; + + var result = new ClusterOptionsValidator().Validate(null, options); + + Assert.True(result.Failed); + Assert.Contains("PartnerProbeIntervalMs", result.FailureMessage); + Assert.Contains("ProbeConnectTimeoutMs", result.FailureMessage); + } } diff --git a/tests/ZB.MOM.WW.ScadaBridge.IntegrationTests/Cluster/ClusterBootstrapCoordinatorTests.cs b/tests/ZB.MOM.WW.ScadaBridge.IntegrationTests/Cluster/ClusterBootstrapCoordinatorTests.cs new file mode 100644 index 00000000..442c7286 --- /dev/null +++ b/tests/ZB.MOM.WW.ScadaBridge.IntegrationTests/Cluster/ClusterBootstrapCoordinatorTests.cs @@ -0,0 +1,204 @@ +using Akka.Actor; +using Akka.Cluster; +using Akka.Configuration; +using Microsoft.Extensions.Logging.Abstractions; +using Microsoft.Extensions.Options; +using ZB.MOM.WW.ScadaBridge.ClusterInfrastructure; +using ZB.MOM.WW.ScadaBridge.Host; +using ZB.MOM.WW.ScadaBridge.Host.Actors; + +namespace ZB.MOM.WW.ScadaBridge.IntegrationTests.Cluster; + +/// +/// Real-ActorSystem tests for the simultaneous-cold-start split-brain guard (Gitea #33): +/// + , driven exactly as +/// production drives them — the node's ActorSystem is built from the production BuildHocon output +/// with the guard enabled (which emits an EMPTY seed list, so Akka does not auto-join), then the real +/// coordinator issues the reachability-gated JoinSeedNodes. This subsystem has a history of bugs +/// invisible to pure unit tests, so the load-bearing correctness properties are asserted against running +/// nodes, mirroring . +/// +/// +/// Ephemeral loopback ports share host 127.0.0.1, so the port breaks the guard's ordinal +/// host:port tie — Min(port) is the founder, Max(port) is the higher (probing) node. +/// +public sealed class ClusterBootstrapCoordinatorTests : IAsyncLifetime +{ + private readonly List _systems = new(); + private readonly List _coordinators = new(); + + /// Starts a node with the bootstrap guard ENABLED, wired exactly as the composition roots + /// do: the ActorSystem is built with an empty HOCON seed list (guard on) and the real coordinator + /// issues the reachability-gated join. Seeds are listed self-first (as every shipped config is); the + /// guard, not the order, decides founder vs. joiner. + private async Task StartGuardedNodeAsync(int selfPort, int peerPort, int probeSeconds = 4) + { + var self = $"akka.tcp://scadabridge@127.0.0.1:{selfPort}"; + var peer = $"akka.tcp://scadabridge@127.0.0.1:{peerPort}"; + var nodeOptions = new NodeOptions { Role = "Central", NodeHostname = "127.0.0.1", RemotingPort = selfPort }; + var clusterOptions = new ClusterOptions + { + SeedNodes = new List { self, peer }, + BootstrapGuard = new ClusterBootstrapGuardOptions + { + Enabled = true, + PartnerProbeSeconds = probeSeconds, + PartnerProbeIntervalMs = 250, + ProbeConnectTimeoutMs = 500, + }, + StableAfter = TimeSpan.FromSeconds(3), + HeartbeatInterval = TimeSpan.FromMilliseconds(500), + FailureDetectionThreshold = TimeSpan.FromSeconds(2), + MinNrOfMembers = 1, + }; + + var hocon = AkkaHostedService.BuildHocon( + nodeOptions, clusterOptions, new[] { "Central" }, + TimeSpan.FromSeconds(1), TimeSpan.FromSeconds(3)); + var system = ActorSystem.Create("scadabridge", ConfigurationFactory.ParseString(hocon)); + _systems.Add(system); + + var coordinator = new ClusterBootstrapCoordinator( + () => system, Options.Create(clusterOptions), Options.Create(nodeOptions), + NullLogger.Instance); + _coordinators.Add(coordinator); + await coordinator.StartAsync(CancellationToken.None); + return system; + } + + private static async Task WaitForUpAsync(ActorSystem system, int expected, TimeSpan timeout) + { + var cluster = Akka.Cluster.Cluster.Get(system); + var deadline = DateTime.UtcNow + timeout; + while (DateTime.UtcNow < deadline) + { + if (cluster.State.Members.Count(m => m.Status == MemberStatus.Up) >= expected) return true; + await Task.Delay(200); + } + return cluster.State.Members.Count(m => m.Status == MemberStatus.Up) >= expected; + } + + /// + /// The FOUNDER (lower address) with the guard on forms a cluster alone when its partner is dead — + /// exactly as un-guarded self-first does. The guard must not regress the preferred founder. + /// + [Fact] + public async Task Founder_forms_alone_when_partner_is_dead() + { + var low = TwoNodeClusterFixture.GetFreeTcpPort(); + var high = TwoNodeClusterFixture.GetFreeTcpPort(); + var founderPort = Math.Min(low, high); + var deadPeerPort = Math.Max(low, high); // nothing listening + + var system = await StartGuardedNodeAsync(founderPort, deadPeerPort); + + Assert.True(await WaitForUpAsync(system, 1, TimeSpan.FromSeconds(30)), + "the founder (lower address) must self-form immediately when its partner is down"); + } + + /// + /// THE LOAD-BEARING CASE. The HIGHER node with the guard on must still cold-start ALONE when its + /// partner is genuinely dead: it probes the (dead) founder, times out, and falls back to self-first. + /// If the guard's peer-first logic were wrong, this node would hang in JoinSeedNodeProcess forever — + /// the very regression a naive "always let the lower node found" rule would introduce. + /// + [Fact] + public async Task Higher_node_forms_alone_when_partner_is_dead_after_probing() + { + var low = TwoNodeClusterFixture.GetFreeTcpPort(); + var high = TwoNodeClusterFixture.GetFreeTcpPort(); + var higherPort = Math.Max(low, high); + var deadFounderPort = Math.Min(low, high); // the founder is dead + + var system = await StartGuardedNodeAsync(higherPort, deadFounderPort, probeSeconds: 3); + + // Must come Up AFTER the probe window (~3s) expires and it falls back to self-first. + Assert.True(await WaitForUpAsync(system, 1, TimeSpan.FromSeconds(30)), + "the higher node must fall back to self-first and form alone once the dead partner never answers the probe"); + } + + /// + /// FALSIFIABILITY CONTROL for the test above: the higher node does NOT form alone during the probe + /// window — it is genuinely waiting/probing, not self-forming immediately (which would mean the + /// guard is inert). If this ever starts seeing a member before the window, the probe is not gating. + /// + [Fact] + public async Task Higher_node_does_not_form_before_the_probe_window_expires() + { + var low = TwoNodeClusterFixture.GetFreeTcpPort(); + var high = TwoNodeClusterFixture.GetFreeTcpPort(); + var higherPort = Math.Max(low, high); + var deadFounderPort = Math.Min(low, high); + + var system = await StartGuardedNodeAsync(higherPort, deadFounderPort, probeSeconds: 10); + + // Well within the 10s probe window: still no cluster, because it is probing the dead founder. + Assert.False(await WaitForUpAsync(system, 1, TimeSpan.FromSeconds(3)), + "the higher node must probe for the founder, not self-form immediately"); + } + + /// + /// The higher node JOINS a live founder rather than forming a second cluster: start the founder, + /// then the higher node — its probe finds the founder reachable, it commits peer-first, and the + /// pair converges to ONE cluster of two. + /// + [Fact] + public async Task Higher_node_joins_a_live_founder() + { + var low = TwoNodeClusterFixture.GetFreeTcpPort(); + var high = TwoNodeClusterFixture.GetFreeTcpPort(); + var founderPort = Math.Min(low, high); + var higherPort = Math.Max(low, high); + + var founder = await StartGuardedNodeAsync(founderPort, higherPort); + Assert.True(await WaitForUpAsync(founder, 1, TimeSpan.FromSeconds(30)), + "the founder must be Up before the higher node probes it"); + + var higher = await StartGuardedNodeAsync(higherPort, founderPort); + + Assert.True(await WaitForUpAsync(higher, 2, TimeSpan.FromSeconds(60)), + "the higher node must join the founder, forming one cluster of two"); + Assert.True(await WaitForUpAsync(founder, 2, TimeSpan.FromSeconds(60)), + "the founder must see the higher node as a member of ITS cluster"); + } + + /// + /// THE HEADLINE #33 CASE. Both nodes cold-start truly simultaneously with the guard on. Without the + /// guard each would run FirstSeedNodeProcess, race, and form its own 1-node cluster (split brain). + /// With it, the founder forms and the higher node probes-then-joins, so the pair converges to + /// EXACTLY ONE cluster of two — no two oldest, no dual singletons. + /// + [Fact] + public async Task Both_nodes_cold_starting_together_form_exactly_one_cluster() + { + var low = TwoNodeClusterFixture.GetFreeTcpPort(); + var high = TwoNodeClusterFixture.GetFreeTcpPort(); + var founderPort = Math.Min(low, high); + var higherPort = Math.Max(low, high); + + // Start both without serialization — as a shared power event does. + var founderTask = StartGuardedNodeAsync(founderPort, higherPort); + var higherTask = StartGuardedNodeAsync(higherPort, founderPort); + var founder = await founderTask; + var higher = await higherTask; + + Assert.True(await WaitForUpAsync(founder, 2, TimeSpan.FromSeconds(60)), + "the pair must converge to one 2-member cluster, not two 1-member clusters"); + Assert.True(await WaitForUpAsync(higher, 2, TimeSpan.FromSeconds(60)), + "the higher node must be in the SAME cluster as the founder"); + } + + public Task InitializeAsync() => Task.CompletedTask; + + public async Task DisposeAsync() + { + foreach (var c in _coordinators) + { + try { await c.StopAsync(CancellationToken.None); } catch { /* teardown */ } + } + foreach (var s in _systems) + { + try { await s.Terminate().WaitAsync(TimeSpan.FromSeconds(10)); } catch { /* teardown */ } + } + } +}