cf3bd52f93
Two-node keep-oldest could NEVER survive a crash of the oldest/active node:
Akka.NET 1.5.62 KeepOldest.OldestDecision only lets down-if-alone rescue a
side with >= 2 members, so the 1-vs-1 survivor takes DownReachable and downs
ITSELF — proven live on the rig ('SBR took decision ... including myself')
before this change. static-quorum(1) is worse (IsTooManyMembers -> DownAll);
keep-majority just re-keys the fatal crash to the lowest address.
SplitBrainResolverStrategy gains 'auto-down' (new default): BuildHocon emits
Akka's AutoDowning provider with auto-down-unreachable-after = StableAfter.
The leader among the REACHABLE members downs the unreachable peer, so the
survivor takes over singletons and /health/active in ~25s regardless of which
node died. Accepted trade (explicit owner decision): a real network partition
runs dual-active until an operator restarts one side. keep-oldest remains
supported; DownIfAlone validation is now scoped to it.
Live drill on the rebuilt rig: active-crash TAKEOVER in 28s (victim still
down; all 7 singletons Younger->Oldest), standby-crash removal 27s with 0
routing blips; victims rejoin as standby in 2s. New real-cluster tests pin
both directions (SbrFailoverTests.AutoDown_*); TwoNodeClusterFixture gains a
strategy knob. All 16 appsettings flipped (src, docker, docker-env2, and the
gitignored wonder-app-vd03 overlay on disk — owner must sync to the host).
Docs: decision record docs/plans/2026-07-21-auto-down-availability-decision.md,
Component-ClusterInfrastructure downing section rewritten, drill + README
reworked (active mode now asserts takeover), deferred-work SBR row resolved.
82 lines
4.7 KiB
C#
82 lines
4.7 KiB
C#
using ZB.MOM.WW.Configuration;
|
|
|
|
namespace ZB.MOM.WW.ScadaBridge.ClusterInfrastructure;
|
|
|
|
/// <summary>
|
|
/// Validates <see cref="ClusterOptions"/> at startup. The values it
|
|
/// guards carry cluster-wide consequences — the design doc
|
|
/// (<c>Component-ClusterInfrastructure.md</c>) is emphatic that misconfiguring
|
|
/// them produces a total cluster shutdown or an indefinitely blocked singleton.
|
|
/// Registered with <c>ValidateOnStart()</c> so a bad <c>appsettings.json</c>
|
|
/// fails fast at boot rather than failing far from the cause.
|
|
/// </summary>
|
|
public sealed class ClusterOptionsValidator : OptionsValidatorBase<ClusterOptions>
|
|
{
|
|
/// <summary>
|
|
/// Downing strategies supported for ScadaBridge's two-node clusters.
|
|
/// <c>auto-down</c> (default) survives a crash of either node at the accepted cost
|
|
/// of dual-active during a real partition; <c>keep-oldest</c> is partition-safe but
|
|
/// cannot survive a crash of the oldest node. Quorum strategies are rejected:
|
|
/// <c>static-quorum</c> quorum-size 1 trips Akka's IsTooManyMembers guard (DownAll
|
|
/// on any unreachability in a 2-node cluster) and <c>keep-majority</c> keys the
|
|
/// fatal crash to the lowest-address node instead of the oldest.
|
|
/// </summary>
|
|
private static readonly HashSet<string> AllowedStrategies = new(StringComparer.OrdinalIgnoreCase)
|
|
{
|
|
"auto-down",
|
|
"keep-oldest"
|
|
};
|
|
|
|
/// <inheritdoc />
|
|
protected override void Validate(ValidationBuilder builder, ClusterOptions options)
|
|
{
|
|
// The design doc states "both nodes are seed nodes — each node lists
|
|
// both itself and its partner" so a properly-configured deployment lists
|
|
// two. Accepting a single-seed configuration silently defeats the
|
|
// "no startup ordering dependency" guarantee called out by
|
|
// Component-ClusterInfrastructure.md (Node Configuration).
|
|
var minSeeds = options.AllowSingleNodeCluster ? 1 : 2;
|
|
builder.RequireThat(options.SeedNodes is not null && options.SeedNodes.Count >= minSeeds,
|
|
options.AllowSingleNodeCluster
|
|
? "ClusterOptions.SeedNodes must contain at least 1 seed node."
|
|
: "ClusterOptions.SeedNodes must contain at least 2 seed nodes "
|
|
+ "(Component-ClusterInfrastructure.md → Node Configuration: both nodes are seed nodes); "
|
|
+ "for a deliberate single-node install set ClusterOptions.AllowSingleNodeCluster = true instead of listing a phantom seed.");
|
|
|
|
builder.RequireThat(
|
|
!string.IsNullOrWhiteSpace(options.SplitBrainResolverStrategy)
|
|
&& AllowedStrategies.Contains(options.SplitBrainResolverStrategy),
|
|
$"ClusterOptions.SplitBrainResolverStrategy must be 'auto-down' or 'keep-oldest' for a " +
|
|
$"two-node cluster; '{options.SplitBrainResolverStrategy}' would risk a total cluster " +
|
|
"shutdown on a partition or an unreachability event.");
|
|
|
|
builder.RequireThat(options.MinNrOfMembers == 1,
|
|
$"ClusterOptions.MinNrOfMembers must be 1 (was {options.MinNrOfMembers}); " +
|
|
"any other value blocks the cluster singleton after failover and halts all data collection.");
|
|
|
|
builder.RequireThat(options.StableAfter > TimeSpan.Zero,
|
|
"ClusterOptions.StableAfter must be a positive duration.");
|
|
|
|
builder.RequireThat(options.HeartbeatInterval > TimeSpan.Zero,
|
|
"ClusterOptions.HeartbeatInterval must be a positive duration.");
|
|
|
|
builder.RequireThat(options.FailureDetectionThreshold > TimeSpan.Zero,
|
|
"ClusterOptions.FailureDetectionThreshold must be a positive duration.");
|
|
|
|
builder.RequireThat(options.HeartbeatInterval < options.FailureDetectionThreshold,
|
|
$"ClusterOptions.HeartbeatInterval ({options.HeartbeatInterval}) must be well below " +
|
|
$"FailureDetectionThreshold ({options.FailureDetectionThreshold}); otherwise nodes are " +
|
|
"declared unreachable before a heartbeat can arrive.");
|
|
|
|
// DownIfAlone is a keep-oldest knob; under auto-down each side downs the
|
|
// unreachable peer regardless, so the flag is inert and any value is fine.
|
|
var isKeepOldest = string.Equals(
|
|
options.SplitBrainResolverStrategy, "keep-oldest", StringComparison.OrdinalIgnoreCase);
|
|
builder.RequireThat(!isKeepOldest || options.DownIfAlone,
|
|
"ClusterOptions.DownIfAlone must be true for the keep-oldest resolver "
|
|
+ "(Component-ClusterInfrastructure.md → Split-Brain Resolution); with it false the "
|
|
+ "oldest node can run as an isolated single-node cluster during a partition while the "
|
|
+ "younger node forms its own, producing two live clusters.");
|
|
}
|
|
}
|