696 lines
30 KiB
C#
696 lines
30 KiB
C#
using Akka.Actor;
|
|
using Akka.Event;
|
|
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Streaming;
|
|
using ZB.MOM.WW.ScadaBridge.Commons.Observability;
|
|
using ZB.MOM.WW.ScadaBridge.Communication.Grpc;
|
|
|
|
namespace ZB.MOM.WW.ScadaBridge.Communication.Actors;
|
|
|
|
/// <summary>
|
|
/// One-per-site aggregator on the active central node backing the operator Alarm
|
|
/// Summary live cache (plan #10, Task 4). Holds a transient in-memory
|
|
/// <c>Dictionary<AlarmKey, AlarmStateChanged></c> of the whole site's current
|
|
/// alarm state — NO persistence (locked <c>[PERM]</c>: no central alarm store). Created
|
|
/// and torn down by <see cref="SiteAlarmLiveCacheService"/> under viewer reference-count.
|
|
/// <para>
|
|
/// <b>Seed-then-stream ordering</b> (copied from <see cref="DebugStreamBridgeActor"/>):
|
|
/// the site-wide, alarm-only gRPC stream (<c>SubscribeSite</c>) is opened FIRST in
|
|
/// <see cref="PreStart"/> so live deltas start flowing during the seed's snapshot-build
|
|
/// + transit window; deltas arriving while a seed/reconcile fan-out is in flight are
|
|
/// <em>buffered in arrival order</em>. When the fan-out completes the cache is rebuilt
|
|
/// authoritatively from the fresh snapshot, then the buffer is flushed with per-key
|
|
/// dedup against the seed, then the actor passes live deltas straight into the cache.
|
|
/// </para>
|
|
/// <para>
|
|
/// <b>Placeholder reconciliation:</b> the snapshot fan-out carries
|
|
/// <see cref="AlarmStateChanged.IsConfiguredPlaceholder"/> rows (a configured native
|
|
/// source binding with no active conditions); the live stream drops them server-side.
|
|
/// Placeholders are seeded and never expected on the live stream — a live delta for a
|
|
/// real alarm has a distinct <c>AlarmKey</c> so it can never wipe a placeholder row and
|
|
/// vice-versa. Placeholder-vs-real coherence is refreshed by the periodic reconcile.
|
|
/// </para>
|
|
/// <para>
|
|
/// <b>Failover + drift:</b> a gRPC error flips NodeA↔NodeB with the same retry budget +
|
|
/// stability window as <see cref="DebugStreamBridgeActor"/>, and each reconnect triggers
|
|
/// a RE-SEED (never silently serve stale). A periodic reconcile snapshot
|
|
/// (<see cref="_reconcileInterval"/>, default 60s) corrects instance-set drift and any
|
|
/// missed delta.
|
|
/// </para>
|
|
/// All state is mutated only on the actor thread: gRPC callbacks and fan-out results are
|
|
/// marshalled back via <c>Self.Tell</c>, so the cache needs no internal lock. The
|
|
/// published snapshot handed to the service is a fresh immutable list (reference swap),
|
|
/// so Blazor render threads never observe a partially-mutated cache.
|
|
/// </summary>
|
|
public sealed class SiteAlarmAggregatorActor : ReceiveActor, IWithTimers
|
|
{
|
|
private readonly ILoggingAdapter _log = Context.GetLogger();
|
|
private readonly string _siteIdentifier;
|
|
private readonly string _correlationId;
|
|
private readonly Func<CancellationToken, Task<IReadOnlyList<AlarmStateChanged>>> _seedFn;
|
|
private readonly Action<IReadOnlyList<AlarmStateChanged>> _publish;
|
|
private readonly SiteStreamGrpcClientFactory _grpcFactory;
|
|
private readonly string _grpcNodeAAddress;
|
|
private readonly string _grpcNodeBAddress;
|
|
private readonly TimeSpan _reconcileInterval;
|
|
private readonly TimeSpan _publishCoalesce;
|
|
|
|
private const int MaxRetries = 3;
|
|
private const string ReconnectTimerKey = "alarm-grpc-reconnect";
|
|
private const string StabilityTimerKey = "alarm-grpc-stability";
|
|
private const string ReconcileTimerKey = "alarm-reconcile";
|
|
private const string PublishTimerKey = "alarm-publish-coalesce";
|
|
|
|
/// <summary>True while a coalesced publish is armed (dirty deltas awaiting one tick). Actor-thread only.</summary>
|
|
private bool _publishPending;
|
|
|
|
/// <summary>Delay between gRPC reconnection attempts (ctor-injected; production default 5s).</summary>
|
|
private readonly TimeSpan _reconnectDelay;
|
|
|
|
/// <summary>
|
|
/// How long a freshly-opened gRPC stream must stay up before its retry budget is
|
|
/// considered recovered (mirrors <see cref="DebugStreamBridgeActor.StabilityWindow"/>);
|
|
/// ctor-injected (production default 60s).
|
|
/// </summary>
|
|
private readonly TimeSpan _stabilityWindow;
|
|
|
|
private int _retryCount;
|
|
private bool _useNodeA = true;
|
|
private bool _stopped;
|
|
|
|
/// <summary>
|
|
/// True while there is no live gRPC stream — either it was given up (retry budget
|
|
/// exhausted) or it ended gracefully (server status OK at the site's max stream
|
|
/// lifetime). Reconcile snapshots keep serving in the meantime; the next reconcile tick
|
|
/// self-heals the stream by resetting the retry budget and reopening it, so neither a
|
|
/// sustained site outage nor a routine 4h stream expiry permanently drops the live feed.
|
|
/// Actor-thread only.
|
|
/// </summary>
|
|
private bool _streamDown;
|
|
|
|
/// <summary>
|
|
/// Why the stream is down: <c>true</c> = the retry budget was exhausted, so the
|
|
/// self-healing reopen must also reset it; <c>false</c> = it ended gracefully, and the
|
|
/// budget — which a completion neither spends nor refunds — is carried across the reopen
|
|
/// untouched (a stream flapping between faults and clean closes must still trip
|
|
/// MaxRetries). Actor-thread only.
|
|
/// </summary>
|
|
private bool _retryBudgetExhausted;
|
|
private CancellationTokenSource? _grpcCts;
|
|
private CancellationTokenSource? _lifetimeCts;
|
|
|
|
/// <summary>Current whole-site alarm state, keyed by <see cref="AlarmKey"/>. Actor-thread only.</summary>
|
|
private readonly Dictionary<string, AlarmStateChanged> _cache = new();
|
|
|
|
/// <summary>True once the first seed has completed and been published. Actor-thread only.</summary>
|
|
private bool _seeded;
|
|
|
|
/// <summary>True while a seed/reconcile snapshot fan-out is in flight (deltas buffer). Actor-thread only.</summary>
|
|
private bool _fanoutInFlight;
|
|
|
|
/// <summary>Ordered buffer of live deltas that arrived while a fan-out was in flight. Actor-thread only.</summary>
|
|
private readonly List<AlarmStateChanged> _buffer = new();
|
|
|
|
/// <summary>
|
|
/// A failover re-seed was requested while a fan-out was already in flight; it must run
|
|
/// right after the in-flight one completes rather than being silently dropped (N7.1) —
|
|
/// the in-flight snapshot's read-time predates the stream death. Actor-thread only.
|
|
/// </summary>
|
|
private bool _reseedQueued;
|
|
|
|
/// <summary>
|
|
/// Monotonic stream generation stamped on each opened gRPC stream and echoed back on its
|
|
/// error callback: a late error raced out of a previous (cancelled) stream carries a stale
|
|
/// generation and is ignored so it never burns retry budget or double-flips (N7.2).
|
|
/// Actor-thread only.
|
|
/// </summary>
|
|
private int _streamGeneration;
|
|
|
|
private const int BufferWarnThreshold = 10_000;
|
|
private bool _bufferWarned;
|
|
|
|
/// <inheritdoc />
|
|
public ITimerScheduler Timers { get; set; } = null!;
|
|
|
|
/// <summary>
|
|
/// Creates a per-site alarm aggregator.
|
|
/// </summary>
|
|
/// <param name="siteIdentifier">Site identifier (for logging / gRPC client keying).</param>
|
|
/// <param name="correlationId">Correlation id for the site-wide gRPC subscription.</param>
|
|
/// <param name="seedFn">
|
|
/// Snapshot fan-out that returns the whole site's current alarm rows (including
|
|
/// placeholders), best-effort and tolerant of per-instance failure. Re-run on every
|
|
/// seed and reconcile, so it re-enumerates the site's Enabled instances each call.
|
|
/// </param>
|
|
/// <param name="publish">
|
|
/// Publishes a fresh immutable snapshot of the cache to the owning service, which
|
|
/// stores it and raises viewer <c>onChanged</c> callbacks. Invoked on the actor thread.
|
|
/// </param>
|
|
/// <param name="grpcFactory">Factory caching one gRPC client per (site, endpoint).</param>
|
|
/// <param name="grpcNodeAAddress">gRPC address of the site's node A.</param>
|
|
/// <param name="grpcNodeBAddress">gRPC address of the site's node B.</param>
|
|
/// <param name="reconcileInterval">Periodic reconcile snapshot cadence.</param>
|
|
/// <param name="publishCoalesce">
|
|
/// Publish-coalescing window for live deltas: a positive value batches a delta storm
|
|
/// into one publish per window (review 02 round 2, N6); <see cref="TimeSpan.Zero"/>
|
|
/// restores per-delta publishing (legacy). Seed/reconcile publishes stay immediate.
|
|
/// </param>
|
|
/// <param name="reconnectDelay">Delay between gRPC reconnection attempts (production 5s).</param>
|
|
/// <param name="stabilityWindow">
|
|
/// How long a fresh gRPC stream must stay up before its retry budget recovers (production 60s).
|
|
/// </param>
|
|
public SiteAlarmAggregatorActor(
|
|
string siteIdentifier,
|
|
string correlationId,
|
|
Func<CancellationToken, Task<IReadOnlyList<AlarmStateChanged>>> seedFn,
|
|
Action<IReadOnlyList<AlarmStateChanged>> publish,
|
|
SiteStreamGrpcClientFactory grpcFactory,
|
|
string grpcNodeAAddress,
|
|
string grpcNodeBAddress,
|
|
TimeSpan reconcileInterval,
|
|
TimeSpan publishCoalesce,
|
|
TimeSpan reconnectDelay,
|
|
TimeSpan stabilityWindow)
|
|
{
|
|
_siteIdentifier = siteIdentifier;
|
|
_correlationId = correlationId;
|
|
_seedFn = seedFn;
|
|
_publish = publish;
|
|
_grpcFactory = grpcFactory;
|
|
_grpcNodeAAddress = grpcNodeAAddress;
|
|
_grpcNodeBAddress = grpcNodeBAddress;
|
|
_reconcileInterval = reconcileInterval;
|
|
_publishCoalesce = publishCoalesce;
|
|
_reconnectDelay = reconnectDelay;
|
|
_stabilityWindow = stabilityWindow;
|
|
|
|
// Live delta from the site-wide alarm stream (marshalled in via Self.Tell).
|
|
// A received delta must NOT reset the retry budget (a flapping stream that
|
|
// delivers one delta between failures would otherwise never trip MaxRetries).
|
|
Receive<AlarmStateChanged>(HandleLiveDelta);
|
|
|
|
// A seed/reconcile fan-out completed.
|
|
Receive<SeedCompleted>(OnSeedCompleted);
|
|
|
|
// A seed/reconcile fan-out threw as a whole (individual per-instance faults are
|
|
// swallowed inside seedFn and degrade to fewer rows, not a whole-fan-out failure).
|
|
Receive<SeedFailed>(OnSeedFailed);
|
|
|
|
// Periodic reconcile tick (and the re-seed kicked after a reconnect).
|
|
Receive<RunReconcile>(_ => OnReconcileTick());
|
|
|
|
// Coalesced-publish tick: one publish for a batch of dirtying deltas (N6).
|
|
Receive<PublishCoalesced>(_ =>
|
|
{
|
|
_publishPending = false;
|
|
if (!_stopped) Publish();
|
|
});
|
|
|
|
// Stream stayed up for StabilityWindow — recover the retry budget.
|
|
Receive<GrpcAlarmStreamStable>(_ =>
|
|
{
|
|
if (_stopped) return;
|
|
_retryCount = 0;
|
|
_log.Debug("Site-alarm gRPC stream for {0} stable; retry count reset", _siteIdentifier);
|
|
});
|
|
|
|
// gRPC stream error — flip node + reconnect + re-seed.
|
|
Receive<GrpcAlarmStreamError>(msg =>
|
|
{
|
|
// Ignore a late error raced out of a previous (cancelled) stream — the
|
|
// RpcException(Cancelled) filter at SiteStreamGrpcClient.cs covers the normal
|
|
// path, but a genuine socket fault can beat the cancel (N7.2).
|
|
if (msg.Generation != _streamGeneration)
|
|
{
|
|
_log.Debug("Ignoring stale gRPC error from stream generation {0} (current {1})",
|
|
msg.Generation, _streamGeneration);
|
|
return;
|
|
}
|
|
_log.Warning("Site-alarm gRPC stream error for {0}: {1}", _siteIdentifier, msg.Exception.Message);
|
|
HandleGrpcError();
|
|
});
|
|
|
|
// gRPC stream ended GRACEFULLY (server status OK) — the site's 4h max stream
|
|
// lifetime elapsing or a graceful site shutdown. Not a fault: the stream is marked
|
|
// down so the reconcile tick reopens it (which also re-seeds), but the retry budget
|
|
// is untouched and the node is not flipped. Same generation fence as the error path.
|
|
Receive<GrpcAlarmStreamCompleted>(msg =>
|
|
{
|
|
if (_stopped) return;
|
|
if (msg.Generation != _streamGeneration)
|
|
{
|
|
_log.Debug("Ignoring stale gRPC completion from stream generation {0} (current {1})",
|
|
msg.Generation, _streamGeneration);
|
|
return;
|
|
}
|
|
HandleGrpcCompleted();
|
|
});
|
|
|
|
Receive<ReconnectAlarmStream>(_ => OpenGrpcStream());
|
|
|
|
// Owning service asks us to stop (last viewer left + linger elapsed).
|
|
Receive<StopSiteAlarmAggregator>(_ =>
|
|
{
|
|
_log.Info("Stopping site-alarm aggregator for {0}", _siteIdentifier);
|
|
CleanupGrpc();
|
|
_stopped = true;
|
|
Context.Stop(Self);
|
|
});
|
|
}
|
|
|
|
/// <inheritdoc />
|
|
protected override void PreStart()
|
|
{
|
|
_log.Info("Starting site-alarm aggregator for site {0}", _siteIdentifier);
|
|
// Telemetry: this aggregator is now a running per-site live cache (gauge +1). Balanced
|
|
// in PostStop, which Akka always runs on termination for any reason.
|
|
ScadaBridgeTelemetry.LiveAlarmAggregatorStarted();
|
|
_lifetimeCts = new CancellationTokenSource();
|
|
|
|
// Stream-first: open the site-wide alarm stream BEFORE the first seed so deltas
|
|
// in the seed window are captured (buffered) rather than lost.
|
|
OpenGrpcStream();
|
|
|
|
// Kick the initial seed fan-out.
|
|
StartFanout(isInitial: true);
|
|
|
|
// Periodic reconcile backstop.
|
|
Timers.StartPeriodicTimer(ReconcileTimerKey, new RunReconcile(), _reconcileInterval, _reconcileInterval);
|
|
}
|
|
|
|
/// <inheritdoc />
|
|
protected override void PostStop()
|
|
{
|
|
_grpcCts?.Cancel();
|
|
_grpcCts?.Dispose();
|
|
_grpcCts = null;
|
|
_lifetimeCts?.Cancel();
|
|
_lifetimeCts?.Dispose();
|
|
_lifetimeCts = null;
|
|
// Telemetry: this aggregator is no longer running (gauge -1). Balances PreStart.
|
|
ScadaBridgeTelemetry.LiveAlarmAggregatorStopped();
|
|
base.PostStop();
|
|
}
|
|
|
|
// ── Reconcile tick ──────────────────────────────────────────────────────────
|
|
|
|
/// <summary>
|
|
/// Periodic reconcile: always re-run the snapshot fan-out (corrects drift + missed
|
|
/// deltas), and if the live stream was previously given up, self-heal it by resetting
|
|
/// the retry budget and reopening — so a sustained outage never permanently kills the
|
|
/// live feed.
|
|
/// </summary>
|
|
private void OnReconcileTick()
|
|
{
|
|
if (_stopped) return;
|
|
StartFanout(isInitial: false);
|
|
if (_streamDown)
|
|
{
|
|
_log.Info("Site-alarm gRPC stream for {0} was down; reopening on reconcile tick", _siteIdentifier);
|
|
if (_retryBudgetExhausted)
|
|
{
|
|
_retryBudgetExhausted = false;
|
|
_retryCount = 0;
|
|
}
|
|
// Telemetry: a reconcile-driven reopen after the stream was given up is a reconnect.
|
|
ScadaBridgeTelemetry.RecordLiveAlarmStreamReconnect();
|
|
OpenGrpcStream();
|
|
}
|
|
}
|
|
|
|
// ── Seed / reconcile fan-out ────────────────────────────────────────────────
|
|
|
|
/// <summary>
|
|
/// Kicks a snapshot fan-out as a background task, marshalling the result back via
|
|
/// <c>Self.Tell</c>. While in flight, live deltas buffer. A reconcile that arrives
|
|
/// while a fan-out is already running is skipped (no stacking).
|
|
/// </summary>
|
|
private void StartFanout(bool isInitial)
|
|
{
|
|
if (_stopped) return;
|
|
if (_fanoutInFlight)
|
|
{
|
|
// A failover re-seed requested mid-fan-out must run right after the in-flight
|
|
// one — its snapshot read-time predates the stream death, so skipping it would
|
|
// serve stale up to the next 60s reconcile (N7.1). An initial-seed collision
|
|
// never queues (there is only ever one).
|
|
if (!isInitial) _reseedQueued = true;
|
|
return;
|
|
}
|
|
|
|
_fanoutInFlight = true;
|
|
var self = Self;
|
|
var ct = _lifetimeCts?.Token ?? CancellationToken.None;
|
|
|
|
Task.Run(async () =>
|
|
{
|
|
try
|
|
{
|
|
var alarms = await _seedFn(ct);
|
|
self.Tell(new SeedCompleted(alarms, isInitial));
|
|
}
|
|
catch (OperationCanceledException) when (ct.IsCancellationRequested)
|
|
{
|
|
// Actor stopping — drop silently.
|
|
}
|
|
catch (Exception ex)
|
|
{
|
|
self.Tell(new SeedFailed(ex, isInitial));
|
|
}
|
|
}, ct);
|
|
}
|
|
|
|
private void OnSeedCompleted(SeedCompleted msg)
|
|
{
|
|
if (_stopped) return;
|
|
|
|
// Rebuild the cache authoritatively from the fresh snapshot (this is what makes
|
|
// reconcile able to DROP rows for instances/alarms that disappeared — a merge
|
|
// could never remove a stale row since the live stream sends no "removed" event).
|
|
_cache.Clear();
|
|
foreach (var alarm in msg.Alarms)
|
|
{
|
|
var key = AlarmKey(alarm);
|
|
if (!_cache.TryGetValue(key, out var existing) || alarm.Timestamp >= existing.Timestamp)
|
|
_cache[key] = alarm;
|
|
}
|
|
|
|
// Flush the deltas buffered during the fan-out, deduped against the seed: a
|
|
// buffered delta whose key is in the fresh snapshot with an equal-or-newer
|
|
// timestamp is already reflected → drop; a strictly-newer (or new-key) delta is
|
|
// applied. Inclusive-on-snapshot boundary matches DebugStreamBridgeActor.
|
|
FlushBuffer();
|
|
|
|
_fanoutInFlight = false;
|
|
_seeded = true;
|
|
|
|
_log.Debug("Site-alarm {0} {1} complete: {2} alarm row(s)",
|
|
_siteIdentifier, msg.IsInitial ? "seed" : "reconcile", _cache.Count);
|
|
|
|
// The fresh snapshot already carries the buffered deltas; drop any armed coalesce
|
|
// tick so we publish once, immediately.
|
|
Timers.Cancel(PublishTimerKey);
|
|
_publishPending = false;
|
|
Publish();
|
|
|
|
// A failover re-seed requested while this fan-out was in flight runs now (N7.1).
|
|
if (_reseedQueued)
|
|
{
|
|
_reseedQueued = false;
|
|
StartFanout(isInitial: false);
|
|
}
|
|
}
|
|
|
|
private void OnSeedFailed(SeedFailed msg)
|
|
{
|
|
if (_stopped) return;
|
|
|
|
_log.Warning(msg.Exception,
|
|
"Site-alarm {0} {1} fan-out failed; keeping current cache and relying on the next reconcile",
|
|
_siteIdentifier, msg.IsInitial ? "seed" : "reconcile");
|
|
|
|
// Don't lose deltas captured during the failed window — apply them pass-through
|
|
// into the (possibly stale/empty) cache. The next reconcile re-seeds authoritatively.
|
|
_fanoutInFlight = false;
|
|
FlushBuffer(dedupAgainstSeed: false);
|
|
|
|
// Only publish if we already had a seed (so IsLive doesn't flip true on a
|
|
// failed initial seed — the page keeps its poll fallback until we truly seed).
|
|
if (_seeded)
|
|
{
|
|
Timers.Cancel(PublishTimerKey);
|
|
_publishPending = false;
|
|
Publish();
|
|
}
|
|
|
|
// A failover re-seed requested while this fan-out was in flight runs now (N7.1).
|
|
if (_reseedQueued)
|
|
{
|
|
_reseedQueued = false;
|
|
StartFanout(isInitial: false);
|
|
}
|
|
}
|
|
|
|
/// <summary>
|
|
/// Flushes the pre-fan-out buffer in arrival order. When <paramref name="dedupAgainstSeed"/>
|
|
/// is true (normal completion) a buffered delta already reflected in the just-installed
|
|
/// snapshot (same key, timestamp <= cache entry) is dropped; otherwise every buffered
|
|
/// delta is applied pass-through.
|
|
/// </summary>
|
|
private void FlushBuffer(bool dedupAgainstSeed = true)
|
|
{
|
|
if (_buffer.Count == 0) return;
|
|
|
|
foreach (var delta in _buffer)
|
|
{
|
|
if (dedupAgainstSeed)
|
|
ApplyDelta(delta, requireStrictlyNewer: true);
|
|
else
|
|
ApplyDelta(delta, requireStrictlyNewer: false);
|
|
}
|
|
_buffer.Clear();
|
|
}
|
|
|
|
// ── Live delta handling ─────────────────────────────────────────────────────
|
|
|
|
private void HandleLiveDelta(AlarmStateChanged delta)
|
|
{
|
|
if (_stopped) return;
|
|
|
|
if (_fanoutInFlight)
|
|
{
|
|
_buffer.Add(delta);
|
|
if (!_bufferWarned && _buffer.Count > BufferWarnThreshold)
|
|
{
|
|
_bufferWarned = true;
|
|
_log.Warning(
|
|
"Site-alarm pre-seed buffer for {0} exceeded {1} deltas while a fan-out was in flight " +
|
|
"(deltas retained, not dropped).",
|
|
_siteIdentifier, BufferWarnThreshold);
|
|
}
|
|
return;
|
|
}
|
|
|
|
// Pass-through: apply and (coalesced) publish only if the cache actually changed.
|
|
if (ApplyDelta(delta, requireStrictlyNewer: false))
|
|
SchedulePublish();
|
|
}
|
|
|
|
/// <summary>
|
|
/// Coalesced publish: with a positive window, the first dirtying delta arms a
|
|
/// single-shot timer and further deltas ride the same tick — one snapshot copy and
|
|
/// one viewer fan-out per window instead of per transition (N6). Zero = legacy
|
|
/// immediate publish. Last write wins, so batching never changes final state.
|
|
/// </summary>
|
|
private void SchedulePublish()
|
|
{
|
|
if (_publishCoalesce <= TimeSpan.Zero) { Publish(); return; }
|
|
if (_publishPending) return;
|
|
_publishPending = true;
|
|
Timers.StartSingleTimer(PublishTimerKey, new PublishCoalesced(), _publishCoalesce);
|
|
}
|
|
|
|
/// <summary>
|
|
/// Applies one alarm delta to the cache keyed by <see cref="AlarmKey"/>. Returns
|
|
/// <c>true</c> if the cache changed. A stale (older) delta for an existing key is
|
|
/// ignored. Never carries a placeholder (the live stream drops those), so a live
|
|
/// delta for a real alarm can only add/replace its own key — it never touches a
|
|
/// placeholder row under a different key.
|
|
/// </summary>
|
|
private bool ApplyDelta(AlarmStateChanged delta, bool requireStrictlyNewer)
|
|
{
|
|
var key = AlarmKey(delta);
|
|
if (_cache.TryGetValue(key, out var existing))
|
|
{
|
|
var newer = requireStrictlyNewer
|
|
? delta.Timestamp > existing.Timestamp
|
|
: delta.Timestamp >= existing.Timestamp;
|
|
if (!newer) return false;
|
|
}
|
|
_cache[key] = delta;
|
|
return true;
|
|
}
|
|
|
|
private void Publish()
|
|
{
|
|
// Fresh immutable list — reference swap; readers never see a partial mutation.
|
|
var snapshot = _cache.Values.ToList();
|
|
try
|
|
{
|
|
_publish(snapshot);
|
|
}
|
|
catch (Exception ex)
|
|
{
|
|
_log.Warning(ex, "Site-alarm publish callback threw for {0}; ignoring", _siteIdentifier);
|
|
}
|
|
}
|
|
|
|
// ── gRPC stream lifecycle (mirrors DebugStreamBridgeActor) ──────────────────
|
|
|
|
private void OpenGrpcStream()
|
|
{
|
|
if (_stopped) return;
|
|
|
|
_streamDown = false;
|
|
var endpoint = _useNodeA ? _grpcNodeAAddress : _grpcNodeBAddress;
|
|
_log.Info("Opening site-alarm gRPC stream for {0} to {1}", _siteIdentifier, endpoint);
|
|
|
|
_grpcCts?.Cancel();
|
|
_grpcCts?.Dispose();
|
|
_grpcCts = new CancellationTokenSource();
|
|
|
|
Timers.StartSingleTimer(StabilityTimerKey, new GrpcAlarmStreamStable(), _stabilityWindow);
|
|
|
|
var generation = ++_streamGeneration;
|
|
var client = _grpcFactory.GetOrCreate(_siteIdentifier, endpoint);
|
|
var self = Self;
|
|
var ct = _grpcCts.Token;
|
|
|
|
// The subscription task itself is observed (below): a fault escaping
|
|
// SubscribeSiteAsync — or a Task.Run that never started because ct was already
|
|
// cancelled — would otherwise leave the actor waiting on a stream that does not
|
|
// exist, with the exception silently unobserved.
|
|
Task.Run(async () =>
|
|
{
|
|
await client.SubscribeSiteAsync(
|
|
_correlationId,
|
|
alarm => self.Tell(alarm),
|
|
ex => self.Tell(new GrpcAlarmStreamError(ex, generation)),
|
|
() => self.Tell(new GrpcAlarmStreamCompleted(generation)),
|
|
ct);
|
|
}, ct).ContinueWith(t =>
|
|
{
|
|
if (t.IsFaulted)
|
|
self.Tell(new GrpcAlarmStreamError(t.Exception!.GetBaseException(), generation));
|
|
else if (t.IsCanceled && !ct.IsCancellationRequested)
|
|
self.Tell(new GrpcAlarmStreamCompleted(generation));
|
|
// RanToCompletion: SubscribeSiteAsync already reported its own outcome.
|
|
}, TaskContinuationOptions.ExecuteSynchronously);
|
|
}
|
|
|
|
private void HandleGrpcError()
|
|
{
|
|
if (_stopped) return;
|
|
|
|
// Stream failed before the stability window — retry budget NOT recovered.
|
|
Timers.Cancel(StabilityTimerKey);
|
|
|
|
_retryCount++;
|
|
|
|
if (_retryCount > MaxRetries)
|
|
{
|
|
// Give up the stream, but do NOT stop the aggregator: the periodic reconcile
|
|
// still refreshes the cache from site snapshots, so the page keeps a
|
|
// (slower) live-ish view rather than going dark. A later reconcile-triggered
|
|
// reconnect is not attempted here; the stream is simply left down.
|
|
_log.Error("Site-alarm gRPC stream for {0} exceeded max retries ({1}); leaving stream down, " +
|
|
"reconcile snapshots continue and the next reconcile tick will retry the stream",
|
|
_siteIdentifier, MaxRetries);
|
|
_streamDown = true;
|
|
_retryBudgetExhausted = true;
|
|
CleanupGrpc();
|
|
return;
|
|
}
|
|
|
|
// Unsubscribe the failed stream on the previous endpoint (TryGet, never
|
|
// GetOrCreate) so the site releases its relay actor instead of leaving a zombie.
|
|
var previousEndpoint = _useNodeA ? _grpcNodeAAddress : _grpcNodeBAddress;
|
|
_grpcFactory.TryGet(_siteIdentifier, previousEndpoint)?.Unsubscribe(_correlationId);
|
|
|
|
// Flip to the other node.
|
|
_useNodeA = !_useNodeA;
|
|
|
|
// Telemetry: a NodeA↔NodeB failover flip is a reconnect + re-seed.
|
|
ScadaBridgeTelemetry.RecordLiveAlarmStreamReconnect();
|
|
|
|
// A failover flip must RE-SEED (never silently serve stale) — kick a reconcile
|
|
// fan-out alongside the reconnect. Buffering during the fan-out keeps the new
|
|
// stream's deltas coherent with the fresh snapshot.
|
|
StartFanout(isInitial: false);
|
|
|
|
if (_retryCount == 1)
|
|
Self.Tell(new ReconnectAlarmStream());
|
|
else
|
|
Timers.StartSingleTimer(ReconnectTimerKey, new ReconnectAlarmStream(), _reconnectDelay);
|
|
}
|
|
|
|
/// <summary>
|
|
/// Handles a graceful end of stream. The stream is torn down and left down for the
|
|
/// reconcile tick to reopen — deliberately NOT reopened inline, so a site that keeps
|
|
/// completing streams immediately can never spin this actor into a hot reconnect loop.
|
|
/// The retry budget is neither spent nor reset here (completion is not a fault), and the
|
|
/// endpoint is not flipped: the node that just closed a stream cleanly is healthy.
|
|
/// </summary>
|
|
private void HandleGrpcCompleted()
|
|
{
|
|
// The stream is gone, so its armed stability timer must not later "recover" a
|
|
// budget that its successor has since spent.
|
|
Timers.Cancel(StabilityTimerKey);
|
|
|
|
_log.Info("Site-alarm gRPC stream for {0} completed gracefully (server end of stream); " +
|
|
"reopening on the next reconcile tick", _siteIdentifier);
|
|
|
|
_streamDown = true;
|
|
CleanupGrpc();
|
|
}
|
|
|
|
private void CleanupGrpc()
|
|
{
|
|
_grpcCts?.Cancel();
|
|
_grpcCts?.Dispose();
|
|
_grpcCts = null;
|
|
|
|
var endpoint = _useNodeA ? _grpcNodeAAddress : _grpcNodeBAddress;
|
|
_grpcFactory.TryGet(_siteIdentifier, endpoint)?.Unsubscribe(_correlationId);
|
|
}
|
|
|
|
// ── Dedup key (copied from DebugStreamBridgeActor.AlarmKey with attribution) ──
|
|
|
|
/// <summary>
|
|
/// NUL delimiter so distinct identities never collide on a shared boundary. Cannot
|
|
/// appear in an instance/alarm name. Mirrors <see cref="DebugStreamBridgeActor"/>.
|
|
/// </summary>
|
|
private const char KeyDelimiter = '\u0000';
|
|
|
|
/// <summary>
|
|
/// Per-alarm dedup identity = (InstanceUniqueName, AlarmName, SourceReference) —
|
|
/// identical to <c>DebugStreamBridgeActor.AlarmKey</c> so native per-condition alarms
|
|
/// sharing an AlarmName but differing by source reference are not conflated. Each
|
|
/// nullable component is guarded to prevent silent null/empty key collisions.
|
|
/// </summary>
|
|
private static string AlarmKey(AlarmStateChanged a) =>
|
|
string.Concat(
|
|
a.InstanceUniqueName ?? string.Empty, KeyDelimiter,
|
|
a.AlarmName ?? string.Empty, KeyDelimiter,
|
|
a.SourceReference ?? string.Empty);
|
|
}
|
|
|
|
/// <summary>Message asking a <see cref="SiteAlarmAggregatorActor"/> to stop (last viewer left).</summary>
|
|
public sealed record StopSiteAlarmAggregator;
|
|
|
|
/// <summary>Internal: a seed/reconcile snapshot fan-out completed with the whole-site alarm rows.</summary>
|
|
internal sealed record SeedCompleted(IReadOnlyList<AlarmStateChanged> Alarms, bool IsInitial);
|
|
|
|
/// <summary>Internal: a seed/reconcile snapshot fan-out threw as a whole.</summary>
|
|
internal sealed record SeedFailed(Exception Exception, bool IsInitial);
|
|
|
|
/// <summary>Internal: periodic reconcile tick (and the re-seed kicked after a reconnect).</summary>
|
|
internal sealed record RunReconcile;
|
|
|
|
/// <summary>Internal: coalesced-publish tick — flush the dirty cache to viewers once (N6).</summary>
|
|
internal sealed record PublishCoalesced;
|
|
|
|
/// <summary>Internal: site-alarm gRPC stream error occurred, stamped with the stream
|
|
/// generation it came from so a late error from a cancelled stream can be ignored (N7.2).</summary>
|
|
internal sealed record GrpcAlarmStreamError(Exception Exception, int Generation);
|
|
|
|
/// <summary>Internal: the site-alarm gRPC stream ended gracefully (server status OK — max
|
|
/// stream lifetime or site shutdown), stamped with its stream generation so a late completion
|
|
/// from a cancelled stream can be ignored.</summary>
|
|
internal sealed record GrpcAlarmStreamCompleted(int Generation);
|
|
|
|
/// <summary>Internal: reconnect the site-alarm gRPC stream (flip node).</summary>
|
|
internal sealed record ReconnectAlarmStream;
|
|
|
|
/// <summary>Internal: the current site-alarm gRPC stream has stayed up long enough to recover the retry budget.</summary>
|
|
internal sealed record GrpcAlarmStreamStable;
|