33b15f10a4
Introduce ICentralTransport as the site->central choke point inside SiteCommunicationActor. The seven site->central sends (notification submit/ status, audit + cached-telemetry ingest, reconcile, health, heartbeat) now delegate to an injected transport instead of owning ClusterClient.Send inline. - AkkaCentralTransport: verbatim extraction of today's ClusterClient.Send path, including the exact sender-forwarding that routes central's reply straight back to the waiting Ask. Default when no transport is injected -> behaviour unchanged, existing SiteCommunicationActorTests pass as-is. - GrpcCentralTransport + CentralChannelProvider: dial CentralControlService with sticky failover + background failback (1s-doubling-cap-60s), PSK + x-scadabridge-site via ControlPlaneCredentials, per-call deadlines mirroring today's Ask timeouts. Cross-node retry ONLY on provably-unsent connect failures; never on DeadlineExceeded. Heartbeat stays fire-and-forget. - StaticSitePskProvider: site's single own-key provider (fail-closed). - CommunicationOptions: CentralTransport flag (default Akka) + CentralGrpcEndpoints (validator: required when transport=Grpc). Host selects the impl; the ClusterClient is created only on the Akka path. Tests: actor-with-fake-transport (7 delegations + fault routing + heartbeat no-fault), AkkaCentralTransport sender-forwarding, GrpcCentralTransport over in-process TestServer (failover flip, sticky, failback, PSK+header, deadline, no-retry-on-deadline), validator. Communication.Tests 371 green, Host.Tests 391 green; the three above-seam suites pass unmodified.
519 lines
24 KiB
C#
519 lines
24 KiB
C#
using Akka.Actor;
|
|
using Akka.Cluster;
|
|
using Akka.Event;
|
|
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Artifacts;
|
|
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Audit;
|
|
using ZB.MOM.WW.ScadaBridge.Commons.Messages.DebugView;
|
|
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Deployment;
|
|
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Health;
|
|
using ZB.MOM.WW.ScadaBridge.Commons.Messages.InboundApi;
|
|
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Integration;
|
|
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Lifecycle;
|
|
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Management;
|
|
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Notification;
|
|
using ZB.MOM.WW.ScadaBridge.Commons.Messages.RemoteQuery;
|
|
|
|
namespace ZB.MOM.WW.ScadaBridge.Communication.Actors;
|
|
|
|
/// <summary>
|
|
/// Site-side actor that receives messages from central via ClusterClient and routes
|
|
/// them to the appropriate local actors. Also sends heartbeats and health reports
|
|
/// to central via the registered ClusterClient.
|
|
///
|
|
/// Routes all 8 message patterns to local handlers.
|
|
/// </summary>
|
|
public class SiteCommunicationActor : ReceiveActor, IWithTimers
|
|
{
|
|
private readonly ILoggingAdapter _log = Context.GetLogger();
|
|
private readonly string _siteId;
|
|
private readonly CommunicationOptions _options;
|
|
|
|
/// <summary>
|
|
/// Predicate that returns <c>true</c> when this node is
|
|
/// the active member of the local site cluster (used to stamp
|
|
/// <see cref="HeartbeatMessage.IsActive"/>). Production builds default to
|
|
/// the Akka <see cref="Cluster"/> leader check; tests inject a stub so they
|
|
/// do not need a real cluster.
|
|
/// </summary>
|
|
private readonly Func<bool> _isActiveCheck;
|
|
private readonly Func<string, string?> _failOverRole;
|
|
|
|
/// <summary>
|
|
/// Reference to the local Deployment Manager singleton proxy.
|
|
/// </summary>
|
|
private readonly IActorRef _deploymentManagerProxy;
|
|
|
|
/// <summary>
|
|
/// The site→central transport. Finalized in <see cref="PreStart"/> to the injected instance,
|
|
/// or a default <see cref="AkkaCentralTransport"/> (ClusterClient) when none is supplied — so
|
|
/// the seven site→central sends delegate here rather than owning the wire plumbing inline.
|
|
/// </summary>
|
|
private ICentralTransport _transport;
|
|
|
|
/// <summary>The transport supplied by the Host (null selects the default Akka transport).</summary>
|
|
private readonly ICentralTransport? _injectedTransport;
|
|
|
|
/// <summary>
|
|
/// Local actor references for routing specific message patterns.
|
|
/// Populated via registration messages.
|
|
/// </summary>
|
|
private IActorRef? _eventLogHandler;
|
|
private IActorRef? _parkedMessageHandler;
|
|
private IActorRef? _integrationHandler;
|
|
private IActorRef? _artifactHandler;
|
|
|
|
/// <summary>Akka timer scheduler injected by the framework via <see cref="IWithTimers"/>.</summary>
|
|
public ITimerScheduler Timers { get; set; } = null!;
|
|
|
|
/// <summary>Initializes the actor, wires all message pattern handlers, and schedules the periodic heartbeat.</summary>
|
|
/// <param name="siteId">The site identifier included in outbound messages.</param>
|
|
/// <param name="options">Communication options including heartbeat interval and transport settings.</param>
|
|
/// <param name="deploymentManagerProxy">Local reference to the Deployment Manager singleton proxy.</param>
|
|
/// <param name="isActiveCheck">
|
|
/// Optional override returning <c>true</c> when this node
|
|
/// is the active member of the site cluster. <c>null</c> uses the shared oldest-Up
|
|
/// evaluator (production wiring passes the Host's singleton-host delegate); tests
|
|
/// pass a stub so they do not need to load Akka.Cluster into the <c>TestKit</c>
|
|
/// ActorSystem.
|
|
/// </param>
|
|
/// <param name="transport">
|
|
/// The site→central transport. <c>null</c> (the default, used by every existing test and by
|
|
/// the Host's Akka path) selects an <see cref="AkkaCentralTransport"/> over ClusterClient; the
|
|
/// Host injects a <see cref="Grpc.GrpcCentralTransport"/> when
|
|
/// <c>ScadaBridge:Communication:CentralTransport</c> is <c>Grpc</c>.
|
|
/// </param>
|
|
public SiteCommunicationActor(
|
|
string siteId,
|
|
CommunicationOptions options,
|
|
IActorRef deploymentManagerProxy,
|
|
Func<bool>? isActiveCheck = null,
|
|
Func<string, string?>? failOverRole = null,
|
|
ICentralTransport? transport = null)
|
|
{
|
|
_siteId = siteId;
|
|
_options = options;
|
|
_deploymentManagerProxy = deploymentManagerProxy;
|
|
_isActiveCheck = isActiveCheck ?? DefaultIsActiveCheck;
|
|
_failOverRole = failOverRole ?? DefaultFailOverRole;
|
|
_injectedTransport = transport;
|
|
// Finalized in PreStart (where _log is usable for the default transport); assigned here
|
|
// too so the field is definitely-assigned for the constructor's Receive closures.
|
|
_transport = transport!;
|
|
|
|
// Registration. Feeding the ClusterClient into the transport is a no-op unless the
|
|
// default/Akka transport is in use — the gRPC transport dials configured endpoints and
|
|
// never receives this message (the Host does not create a ClusterClient for it).
|
|
Receive<RegisterCentralClient>(msg =>
|
|
{
|
|
(_transport as AkkaCentralTransport)?.SetCentralClient(msg.Client);
|
|
});
|
|
Receive<RegisterLocalHandler>(HandleRegisterLocalHandler);
|
|
|
|
// Pattern 1: Instance Deployment — forward to Deployment Manager
|
|
Receive<RefreshDeploymentCommand>(msg =>
|
|
{
|
|
_log.Debug("Routing RefreshDeploymentCommand for {0} to DeploymentManager", msg.InstanceUniqueName);
|
|
_deploymentManagerProxy.Forward(msg);
|
|
});
|
|
|
|
// Pattern 2: Lifecycle — forward to Deployment Manager
|
|
Receive<DisableInstanceCommand>(msg => _deploymentManagerProxy.Forward(msg));
|
|
Receive<EnableInstanceCommand>(msg => _deploymentManagerProxy.Forward(msg));
|
|
Receive<DeleteInstanceCommand>(msg => _deploymentManagerProxy.Forward(msg));
|
|
|
|
// Query-the-site-before-redeploy — forward to
|
|
// the Deployment Manager, which owns the deployed-config store and
|
|
// answers with the instance's currently-applied deployment identity.
|
|
Receive<DeploymentStateQueryRequest>(msg => _deploymentManagerProxy.Forward(msg));
|
|
|
|
// Pattern 3: Artifact Deployment — forward to artifact handler if registered
|
|
Receive<DeployArtifactsCommand>(msg =>
|
|
{
|
|
if (_artifactHandler != null)
|
|
_artifactHandler.Forward(msg);
|
|
else
|
|
{
|
|
_log.Warning("No artifact handler registered, replying with failure");
|
|
Sender.Tell(new ArtifactDeploymentResponse(
|
|
msg.DeploymentId, _siteId, false, "Artifact handler not available", DateTimeOffset.UtcNow));
|
|
}
|
|
});
|
|
|
|
// Pattern 4: Integration Routing — forward to integration handler
|
|
Receive<IntegrationCallRequest>(msg =>
|
|
{
|
|
if (_integrationHandler != null)
|
|
_integrationHandler.Forward(msg);
|
|
else
|
|
{
|
|
Sender.Tell(new IntegrationCallResponse(
|
|
msg.CorrelationId, _siteId, false, null, "Integration handler not available", DateTimeOffset.UtcNow));
|
|
}
|
|
});
|
|
|
|
// Pattern 5: Debug View — forward to Deployment Manager (which routes to Instance Actor)
|
|
Receive<SubscribeDebugViewRequest>(msg => _deploymentManagerProxy.Forward(msg));
|
|
Receive<UnsubscribeDebugViewRequest>(msg => _deploymentManagerProxy.Forward(msg));
|
|
|
|
// Pattern 6a: Debug Snapshot (one-shot) — forward to Deployment Manager
|
|
Receive<DebugSnapshotRequest>(msg => _deploymentManagerProxy.Forward(msg));
|
|
|
|
// Inbound API Route.To() — forward to Deployment Manager for instance routing
|
|
Receive<RouteToCallRequest>(msg => _deploymentManagerProxy.Forward(msg));
|
|
Receive<RouteToGetAttributesRequest>(msg => _deploymentManagerProxy.Forward(msg));
|
|
Receive<RouteToSetAttributesRequest>(msg => _deploymentManagerProxy.Forward(msg));
|
|
Receive<RouteToWaitForAttributeRequest>(msg => _deploymentManagerProxy.Forward(msg));
|
|
|
|
// OPC UA Tag Browser (interactive design-time query) — forward to the
|
|
// Deployment Manager singleton, which always lands on the active site
|
|
// node. Routing to the site-local /user/dcl-manager directly is wrong
|
|
// because the standby node has a dcl-manager too, but its
|
|
// DataConnectionActor children (which own the live OPC UA sessions)
|
|
// only exist on the singleton's node. The singleton then re-forwards
|
|
// to its own /user/dcl-manager, which DOES have the connection.
|
|
Receive<BrowseNodeCommand>(msg => _deploymentManagerProxy.Forward(msg));
|
|
|
|
// Test Bindings (interactive design-time read) — same routing rationale
|
|
// as BrowseNodeCommand above: the singleton always lands on the
|
|
// active site node, which is the node that owns the DataConnectionActor
|
|
// children holding the live OPC UA sessions.
|
|
Receive<ReadTagValuesCommand>(msg => _deploymentManagerProxy.Forward(msg));
|
|
|
|
// OPC UA tag-picker address-space search and secured-write execute
|
|
// — same singleton routing rationale as BrowseNodeCommand above: the
|
|
// DataConnectionActor children that own the live OPC UA sessions exist only
|
|
// on the singleton's (active) node, so these must hop through the Deployment
|
|
// Manager proxy too. Without these forwards the commands dead-letter and the
|
|
// central Ask times out. Forward preserves the central Ask sender so the
|
|
// result routes straight back to the waiting Ask.
|
|
Receive<SearchAddressSpaceCommand>(msg => _deploymentManagerProxy.Forward(msg));
|
|
Receive<Commons.Messages.DataConnection.WriteTagRequest>(msg => _deploymentManagerProxy.Forward(msg));
|
|
|
|
// OPC UA endpoint Verify — probes a (possibly unsaved) endpoint config
|
|
// WITHOUT persisting it. The Deployment Manager singleton's dcl-manager runs
|
|
// the probe directly (no existing connection required), so — like the
|
|
// commands above — Verify routes through the singleton's active node.
|
|
Receive<VerifyEndpointCommand>(msg => _deploymentManagerProxy.Forward(msg));
|
|
|
|
// OPC UA server-certificate trust management — forward to the
|
|
// Deployment Manager singleton, which owns the cross-node trust broadcast.
|
|
// The trusted-peer PKI store is node-wide per site node, so a trust/remove
|
|
// decision must reach BOTH nodes' CertStoreActor; the singleton broadcasts
|
|
// to every site node (list answers from the singleton's own node). The
|
|
// singleton always lands on the active node, the same routing rationale as
|
|
// BrowseNodeCommand above. Forward preserves the central Ask sender so the
|
|
// CertTrustResult routes straight back to the waiting Ask.
|
|
Receive<TrustServerCertCommand>(msg => _deploymentManagerProxy.Forward(msg));
|
|
Receive<ListServerCertsCommand>(msg => _deploymentManagerProxy.Forward(msg));
|
|
Receive<RemoveServerCertCommand>(msg => _deploymentManagerProxy.Forward(msg));
|
|
|
|
// Pattern 7: Remote Queries
|
|
Receive<EventLogQueryRequest>(msg =>
|
|
{
|
|
if (_eventLogHandler != null)
|
|
_eventLogHandler.Forward(msg);
|
|
else
|
|
{
|
|
Sender.Tell(new EventLogQueryResponse(
|
|
msg.CorrelationId, _siteId, [], null, false, false,
|
|
"Event log handler not available", DateTimeOffset.UtcNow));
|
|
}
|
|
});
|
|
|
|
Receive<ParkedMessageQueryRequest>(msg =>
|
|
{
|
|
if (_parkedMessageHandler != null)
|
|
_parkedMessageHandler.Forward(msg);
|
|
else
|
|
{
|
|
Sender.Tell(new ParkedMessageQueryResponse(
|
|
msg.CorrelationId, _siteId, [], 0, msg.PageNumber, msg.PageSize, false,
|
|
"Parked message handler not available", DateTimeOffset.UtcNow));
|
|
}
|
|
});
|
|
|
|
Receive<ParkedMessageRetryRequest>(msg =>
|
|
{
|
|
if (_parkedMessageHandler != null)
|
|
_parkedMessageHandler.Forward(msg);
|
|
else
|
|
{
|
|
Sender.Tell(new ParkedMessageRetryResponse(
|
|
msg.CorrelationId, false, "Parked message handler not available"));
|
|
}
|
|
});
|
|
|
|
Receive<ParkedMessageDiscardRequest>(msg =>
|
|
{
|
|
if (_parkedMessageHandler != null)
|
|
_parkedMessageHandler.Forward(msg);
|
|
else
|
|
{
|
|
Sender.Tell(new ParkedMessageDiscardResponse(
|
|
msg.CorrelationId, false, "Parked message handler not available"));
|
|
}
|
|
});
|
|
|
|
// Central→site Retry/Discard relay for parked cached
|
|
// operations. SiteCallAuditActor relays these over the command/control
|
|
// channel; the parked-message handler executes them against the local
|
|
// S&F buffer and replies a ParkedOperationActionAck that routes back to
|
|
// the relaying SiteCallAuditActor's Ask.
|
|
Receive<RetryParkedOperation>(msg =>
|
|
{
|
|
if (_parkedMessageHandler != null)
|
|
_parkedMessageHandler.Forward(msg);
|
|
else
|
|
{
|
|
Sender.Tell(new ParkedOperationActionAck(
|
|
msg.CorrelationId, Applied: false, "Parked message handler not available"));
|
|
}
|
|
});
|
|
|
|
Receive<DiscardParkedOperation>(msg =>
|
|
{
|
|
if (_parkedMessageHandler != null)
|
|
_parkedMessageHandler.Forward(msg);
|
|
else
|
|
{
|
|
Sender.Tell(new ParkedOperationActionAck(
|
|
msg.CorrelationId, Applied: false, "Parked message handler not available"));
|
|
}
|
|
});
|
|
|
|
// Central→site manual failover relay. Central and the site are separate clusters,
|
|
// so central can only ask — this node performs the graceful Leave locally, scoped to
|
|
// the SITE-SPECIFIC role, because that is what site singletons (the Deployment
|
|
// Manager) are placed on. Either node may receive this (the actor is per-node, not a
|
|
// singleton, and contact rotation picks whichever answers); the target is resolved
|
|
// from cluster state, not from who received the message.
|
|
Receive<TriggerSiteFailover>(HandleTriggerSiteFailover);
|
|
|
|
// The seven site→central sends now delegate to the injected transport (ClusterClient by
|
|
// default, gRPC when configured). Each handler captures the current Sender as the reply
|
|
// target so central's reply routes straight back to the waiting Ask, not through this
|
|
// actor — the exact sender-forwarding the ClusterClient path relied on. The per-message
|
|
// "no transport / not-accepted" fallbacks live inside the transport now.
|
|
|
|
// Notification Outbox: forward a buffered notification (S&F forwarder's Ask → ack back).
|
|
Receive<NotificationSubmit>(msg => _transport.SubmitNotification(msg, Sender));
|
|
|
|
// Notification Outbox: forward a Notify.Status query (Notify helper's Ask → response back).
|
|
Receive<NotificationStatusQuery>(msg => _transport.QueryNotificationStatus(msg, Sender));
|
|
|
|
// Audit Log: forward a batch of site-local audit events (telemetry drain's Ask → reply back).
|
|
Receive<IngestAuditEventsCommand>(msg => _transport.IngestAuditEvents(msg, Sender));
|
|
|
|
// Audit Log: forward a batch of combined cached-call telemetry (telemetry drain's Ask → reply back).
|
|
Receive<IngestCachedTelemetryCommand>(msg => _transport.IngestCachedTelemetry(msg, Sender));
|
|
|
|
// Site startup reconciliation: forward the node's local-inventory request (reconcile Ask → response back).
|
|
Receive<ReconcileSiteRequest>(msg => _transport.ReconcileSite(msg, Sender));
|
|
|
|
// Internal: send heartbeat tick.
|
|
Receive<SendHeartbeat>(_ => SendHeartbeatToCentral());
|
|
|
|
// Internal: forward the periodic health report (health transport's Ask → ack back), so a
|
|
// lost report is observable end-to-end and the sender can restore its per-interval counters.
|
|
Receive<SiteHealthReport>(msg => _transport.ReportSiteHealth(msg, Sender));
|
|
}
|
|
|
|
/// <inheritdoc />
|
|
protected override SupervisorStrategy SupervisorStrategy()
|
|
{
|
|
return new OneForOneStrategy(
|
|
maxNrOfRetries: -1,
|
|
withinTimeRange: Timeout.InfiniteTimeSpan,
|
|
decider: Decider.From(ex =>
|
|
{
|
|
_log.Warning(ex, "Child actor of SiteCommunicationActor faulted, resuming (state preserved)");
|
|
return Directive.Resume;
|
|
}));
|
|
}
|
|
|
|
/// <inheritdoc />
|
|
protected override void PreStart()
|
|
{
|
|
// Finalize the transport now that the actor context (and _log) exist. The default Akka
|
|
// transport is given this actor's logging adapter so the "no ClusterClient registered"
|
|
// warnings are preserved exactly. PreStart always runs before any message, so the Receive
|
|
// closures above see a non-null _transport.
|
|
_transport = _injectedTransport ?? new AkkaCentralTransport(_log);
|
|
|
|
_log.Info("SiteCommunicationActor started for site {0}", _siteId);
|
|
|
|
// Schedule periodic heartbeat to central. Uses the application heartbeat
|
|
// cadence — distinct from the Akka.Remote transport failure-detector
|
|
// interval — so the two can be tuned independently.
|
|
Timers.StartPeriodicTimer(
|
|
"heartbeat",
|
|
new SendHeartbeat(),
|
|
TimeSpan.FromSeconds(1), // initial delay
|
|
_options.ApplicationHeartbeatInterval);
|
|
}
|
|
|
|
private void HandleRegisterLocalHandler(RegisterLocalHandler msg)
|
|
{
|
|
switch (msg.HandlerType)
|
|
{
|
|
case LocalHandlerType.EventLog:
|
|
_eventLogHandler = msg.Handler;
|
|
break;
|
|
case LocalHandlerType.ParkedMessages:
|
|
_parkedMessageHandler = msg.Handler;
|
|
break;
|
|
case LocalHandlerType.Integration:
|
|
_integrationHandler = msg.Handler;
|
|
break;
|
|
case LocalHandlerType.Artifacts:
|
|
_artifactHandler = msg.Handler;
|
|
break;
|
|
}
|
|
|
|
_log.Info("Registered local handler for {0}", msg.HandlerType);
|
|
}
|
|
|
|
private void SendHeartbeatToCentral()
|
|
{
|
|
var hostname = Environment.MachineName;
|
|
|
|
// Stamp HeartbeatMessage.IsActive with this node's
|
|
// true active/standby role rather than hard-coding `true`. The field is
|
|
// part of the wire contract (additive-only-evolution) so a future
|
|
// central health dashboard can distinguish "active node down, standby
|
|
// up" from "site fully offline" without a new message type.
|
|
bool isActive;
|
|
try
|
|
{
|
|
isActive = _isActiveCheck();
|
|
}
|
|
catch (Exception ex)
|
|
{
|
|
// Defensive: never let a cluster-state read failure abort the
|
|
// heartbeat itself (heartbeats are health signal — their absence is
|
|
// already meaningful). Fall back to the safest non-claiming value:
|
|
// standby. Logged at Debug because this path normally only fires
|
|
// during ActorSystem warm-up.
|
|
_log.Debug(ex,
|
|
"Active-node check threw while sending heartbeat for site {0}; reporting IsActive=false",
|
|
_siteId);
|
|
isActive = false;
|
|
}
|
|
|
|
var heartbeat = new HeartbeatMessage(
|
|
_siteId,
|
|
hostname,
|
|
IsActive: isActive,
|
|
DateTimeOffset.UtcNow);
|
|
|
|
// Fire-and-forget on both transports: a failure here must never fault the heartbeat timer
|
|
// path. Both real transports swallow their own errors; this catch is a belt-and-braces
|
|
// guarantee that no transport (including a future one) can turn a heartbeat into a fault.
|
|
try
|
|
{
|
|
_transport.SendHeartbeat(heartbeat, Self);
|
|
}
|
|
catch (Exception ex)
|
|
{
|
|
_log.Debug(ex, "Heartbeat send for site {0} failed; swallowed (heartbeats are fire-and-forget)", _siteId);
|
|
}
|
|
}
|
|
|
|
/// <summary>
|
|
/// Default active-node check used when no override is supplied: oldest-Up member
|
|
/// semantics via the shared <see cref="ClusterState.ActiveNodeEvaluator"/> — the
|
|
/// same predicate as the S&F delivery gate and the replication resync authority
|
|
/// (review 02 round 2, N1). Unscoped by role: a site cluster's members all carry
|
|
/// the site role, so role scoping is a no-op here; production wiring passes the
|
|
/// Host's role-scoped IClusterNodeProvider delegate anyway. Any other state
|
|
/// (still joining, leaving) reports standby — safe-by-default.
|
|
/// </summary>
|
|
private bool DefaultIsActiveCheck() =>
|
|
ClusterState.ActiveNodeEvaluator.SelfIsOldestUp(Cluster.Get(Context.System));
|
|
|
|
/// <summary>
|
|
/// Handles a central-initiated site failover. Refuses a command addressed to a different
|
|
/// site (a misroute must never silently fail over a site the operator did not select) and
|
|
/// refuses when the site pair has no peer to take over. The ack is sent BEFORE the Leave
|
|
/// takes effect on the wire, so it still reaches central even when this node is the one
|
|
/// leaving.
|
|
/// </summary>
|
|
private void HandleTriggerSiteFailover(TriggerSiteFailover msg)
|
|
{
|
|
if (!string.Equals(msg.SiteId, _siteId, StringComparison.Ordinal))
|
|
{
|
|
_log.Warning(
|
|
"Refusing TriggerSiteFailover addressed to site {Requested}; this node serves {Actual}",
|
|
msg.SiteId, _siteId);
|
|
Sender.Tell(new SiteFailoverAck(
|
|
msg.CorrelationId, Accepted: false, TargetAddress: null,
|
|
ErrorMessage: $"Command addressed to site '{msg.SiteId}' but this node serves '{_siteId}'."));
|
|
return;
|
|
}
|
|
|
|
var role = $"site-{_siteId}";
|
|
try
|
|
{
|
|
var target = _failOverRole(role);
|
|
if (target is null)
|
|
{
|
|
_log.Warning(
|
|
"Refusing TriggerSiteFailover for {SiteId}: fewer than 2 Up members in role {Role}, "
|
|
+ "so there is no standby to take over", _siteId, role);
|
|
Sender.Tell(new SiteFailoverAck(
|
|
msg.CorrelationId, Accepted: false, TargetAddress: null,
|
|
ErrorMessage: "No standby available — failing over a lone node would be an outage."));
|
|
return;
|
|
}
|
|
|
|
_log.Warning(
|
|
"Manual failover requested by central for site {SiteId}: {Target} is leaving the "
|
|
+ "site cluster gracefully; its singletons hand over to the standby.", _siteId, target);
|
|
Sender.Tell(new SiteFailoverAck(msg.CorrelationId, Accepted: true, target, ErrorMessage: null));
|
|
}
|
|
catch (Exception ex)
|
|
{
|
|
// A fault here must be reported to the operator, not thrown into supervision —
|
|
// restarting the communication actor would drop central's Ask into a timeout and
|
|
// lose the reason.
|
|
_log.Error(ex, "TriggerSiteFailover for {SiteId} faulted", _siteId);
|
|
Sender.Tell(new SiteFailoverAck(
|
|
msg.CorrelationId, Accepted: false, TargetAddress: null, ErrorMessage: ex.Message));
|
|
}
|
|
}
|
|
|
|
/// <summary>
|
|
/// Production failover action: gracefully Leave the oldest Up member carrying
|
|
/// <paramref name="role"/>, via the shared <see cref="ClusterState.ClusterFailoverCoordinator"/>
|
|
/// so the central and site paths cannot drift. Injected in tests for the same reason
|
|
/// <see cref="DefaultIsActiveCheck"/> is — a real Leave needs Akka.Cluster in the
|
|
/// ActorSystem, which the TestKit system does not load.
|
|
/// </summary>
|
|
/// <param name="role">Site-specific role scope.</param>
|
|
/// <returns>Address of the leaving node, or null when there is no peer.</returns>
|
|
private string? DefaultFailOverRole(string role) =>
|
|
ClusterState.ClusterFailoverCoordinator.FailOverOldest(Context.System, role)?.ToString();
|
|
|
|
// ── Internal messages ──
|
|
|
|
internal record SendHeartbeat;
|
|
}
|
|
|
|
/// <summary>
|
|
/// Command to register a ClusterClient for communicating with the central cluster.
|
|
/// </summary>
|
|
public record RegisterCentralClient(IActorRef Client);
|
|
|
|
/// <summary>
|
|
/// Command to register a local actor as a handler for a specific message pattern.
|
|
/// </summary>
|
|
public record RegisterLocalHandler(LocalHandlerType HandlerType, IActorRef Handler);
|
|
|
|
public enum LocalHandlerType
|
|
{
|
|
EventLog,
|
|
ParkedMessages,
|
|
Integration,
|
|
Artifacts
|
|
}
|