Merge Phase 1A: site→central control plane over gRPC (PR #26)
This commit is contained in:
@@ -39,6 +39,7 @@ services:
|
||||
ports:
|
||||
- "9001:5000" # Web UI + Inbound API
|
||||
- "9011:8081" # Akka remoting (host access for CLI/debugging)
|
||||
- "9013:8083" # gRPC control plane (CentralControlService, T1A.2)
|
||||
volumes:
|
||||
- ./central-node-a/appsettings.Central.json:/app/appsettings.Central.json:ro
|
||||
- ./central-node-a/logs:/app/logs
|
||||
@@ -86,6 +87,7 @@ services:
|
||||
ports:
|
||||
- "9002:5000" # Web UI + Inbound API
|
||||
- "9012:8081" # Akka remoting
|
||||
- "9014:8083" # gRPC control plane (CentralControlService, T1A.2)
|
||||
volumes:
|
||||
- ./central-node-b/appsettings.Central.json:/app/appsettings.Central.json:ro
|
||||
- ./central-node-b/logs:/app/logs
|
||||
|
||||
+75
-49
@@ -1,92 +1,118 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# Regenerates the gRPC C# files from sitestream.proto.
|
||||
# Regenerates the gRPC C# files from the Communication project's .proto files.
|
||||
#
|
||||
# Background: protoc (linux/arm64) segfaults inside our Docker build container
|
||||
# (Grpc.Tools 2.71.0). As a workaround the generated Sitestream.cs +
|
||||
# SitestreamGrpc.cs are checked into src/ZB.MOM.WW.ScadaBridge.Communication/SiteStreamGrpc/
|
||||
# and the Protobuf ItemGroup in the .csproj is commented out — Docker just
|
||||
# compiles the checked-in C# files.
|
||||
# (Grpc.Tools 2.71.0). As a workaround the generated C# is checked into
|
||||
# src/ZB.MOM.WW.ScadaBridge.Communication/ — SiteStreamGrpc/ for sitestream.proto
|
||||
# and CentralControlGrpc/ for central_control.proto — and the Protobuf ItemGroup
|
||||
# in the .csproj is commented out, so Docker just compiles the checked-in files.
|
||||
#
|
||||
# Run this script ON YOUR DEV MACHINE whenever Protos/sitestream.proto changes:
|
||||
# Run this script ON YOUR DEV MACHINE whenever a .proto changes:
|
||||
#
|
||||
# 1. Temporarily uncomments the Protobuf ItemGroup so Grpc.Tools runs.
|
||||
# 2. dotnet build (regen writes fresh files to obj/).
|
||||
# 3. Copies the regenerated files back into SiteStreamGrpc/.
|
||||
# 4. Re-comments the Protobuf ItemGroup so Docker builds stay safe.
|
||||
# docker/regen-proto.sh [sitestream|centralcontrol|all] (default: all)
|
||||
#
|
||||
# 1. Injects a Protobuf ItemGroup for the selected proto(s) so Grpc.Tools runs.
|
||||
# 2. Deletes the stale checked-in C# so a failed regen is obvious.
|
||||
# 3. dotnet build (regen writes fresh files to obj/).
|
||||
# 4. Copies the regenerated files back into the source tree.
|
||||
# 5. Restores the original csproj so no active Protobuf item is left behind.
|
||||
#
|
||||
# Only the SELECTED protos get a Protobuf item. Enabling one whose generated C#
|
||||
# is still checked in would define every generated type twice, which is why the
|
||||
# per-proto selection exists. central_control.proto imports sitestream.proto,
|
||||
# but protoc resolves that from the project-relative path — the import needs no
|
||||
# Protobuf item of its own.
|
||||
#
|
||||
# Once we move to a Dockerfile base image that ships a working linux/arm64
|
||||
# protoc, this script can be retired and Docker can regen the proto on every
|
||||
# protoc, this script can be retired and Docker can regen the protos on every
|
||||
# build like every other normal .NET project.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
TARGET="${1:-all}"
|
||||
case "$TARGET" in
|
||||
sitestream|centralcontrol|all) ;;
|
||||
*) echo "usage: $0 [sitestream|centralcontrol|all]" >&2; exit 2 ;;
|
||||
esac
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
|
||||
COMM_DIR="$REPO_ROOT/src/ZB.MOM.WW.ScadaBridge.Communication"
|
||||
CSPROJ="$COMM_DIR/ZB.MOM.WW.ScadaBridge.Communication.csproj"
|
||||
GEN_DIR="$COMM_DIR/SiteStreamGrpc"
|
||||
GEN="$COMM_DIR/obj/Debug/net10.0/Protos"
|
||||
|
||||
echo "=== Regenerating gRPC files from sitestream.proto ==="
|
||||
echo "=== Regenerating gRPC files ($TARGET) ==="
|
||||
|
||||
if [[ ! -f "$CSPROJ" ]]; then
|
||||
echo "ERROR: csproj not found at $CSPROJ" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Backup so we can always restore the comment state on failure.
|
||||
# Backup so we can always restore the comment state on failure. Leaving the
|
||||
# csproj with an active Protobuf item is the one outcome that breaks Docker, so
|
||||
# every exit path restores this copy.
|
||||
BACKUP="$(mktemp)"
|
||||
cp "$CSPROJ" "$BACKUP"
|
||||
trap 'cp "$BACKUP" "$CSPROJ"; rm -f "$BACKUP"; echo "Restored csproj from backup."' ERR
|
||||
|
||||
# 1. Uncomment the Protobuf ItemGroup (strip the surrounding <!-- ... --> wrapper).
|
||||
python3 - <<PY
|
||||
import re, pathlib
|
||||
p = pathlib.Path("$CSPROJ")
|
||||
src = p.read_text()
|
||||
# Find the commented Protobuf block and unwrap it.
|
||||
new = re.sub(
|
||||
r"<!--\s*\n(\s*<ItemGroup>\s*\n\s*<Protobuf [^>]*/>\s*\n\s*</ItemGroup>)\s*\n\s*-->",
|
||||
r"\1",
|
||||
src,
|
||||
count=1,
|
||||
)
|
||||
if new == src:
|
||||
raise SystemExit("Couldn't find commented Protobuf ItemGroup to enable.")
|
||||
p.write_text(new)
|
||||
# 1. Inject an ItemGroup holding just the selected protos, immediately before
|
||||
# the closing </Project>. The documented commented-out block is left alone.
|
||||
python3 - "$CSPROJ" "$TARGET" <<'PY'
|
||||
import pathlib, sys
|
||||
|
||||
csproj, target = pathlib.Path(sys.argv[1]), sys.argv[2]
|
||||
|
||||
protos = []
|
||||
if target in ("sitestream", "all"):
|
||||
protos.append("sitestream.proto")
|
||||
if target in ("centralcontrol", "all"):
|
||||
protos.append("central_control.proto")
|
||||
|
||||
items = "\n".join(
|
||||
f' <Protobuf Include="Protos\\{p}" GrpcServices="Both" />' for p in protos)
|
||||
block = f" <ItemGroup>\n{items}\n </ItemGroup>\n\n</Project>"
|
||||
|
||||
src = csproj.read_text()
|
||||
if "</Project>" not in src:
|
||||
raise SystemExit("Couldn't find </Project> to inject the Protobuf ItemGroup before.")
|
||||
csproj.write_text(src.replace("</Project>", block, 1))
|
||||
PY
|
||||
|
||||
# 2. Delete the stale files so any failure to regen is obvious.
|
||||
rm -f "$GEN_DIR/Sitestream.cs" "$GEN_DIR/SitestreamGrpc.cs"
|
||||
if [[ "$TARGET" == "sitestream" || "$TARGET" == "all" ]]; then
|
||||
rm -f "$COMM_DIR/SiteStreamGrpc/Sitestream.cs" "$COMM_DIR/SiteStreamGrpc/SitestreamGrpc.cs"
|
||||
fi
|
||||
if [[ "$TARGET" == "centralcontrol" || "$TARGET" == "all" ]]; then
|
||||
rm -f "$COMM_DIR/CentralControlGrpc/CentralControl.cs" \
|
||||
"$COMM_DIR/CentralControlGrpc/CentralControlGrpc.cs"
|
||||
fi
|
||||
|
||||
# 3. Regenerate by building.
|
||||
echo "Building Communication project (regen)..."
|
||||
dotnet build "$CSPROJ" --nologo -v minimal | tail -5
|
||||
|
||||
# 4. Copy generated files back into the source tree.
|
||||
mkdir -p "$GEN_DIR"
|
||||
cp "$COMM_DIR/obj/Debug/net10.0/Protos/Sitestream.cs" "$GEN_DIR/Sitestream.cs"
|
||||
cp "$COMM_DIR/obj/Debug/net10.0/Protos/SitestreamGrpc.cs" "$GEN_DIR/SitestreamGrpc.cs"
|
||||
echo "Copied regenerated files to $GEN_DIR/"
|
||||
|
||||
# 5. Re-comment the Protobuf ItemGroup so Docker builds keep working.
|
||||
python3 - <<PY
|
||||
import re, pathlib
|
||||
p = pathlib.Path("$CSPROJ")
|
||||
src = p.read_text()
|
||||
new = re.sub(
|
||||
r"(\s*<ItemGroup>\s*\n\s*<Protobuf [^>]*/>\s*\n\s*</ItemGroup>)",
|
||||
r"\n <!--\1\n -->",
|
||||
src,
|
||||
count=1,
|
||||
)
|
||||
p.write_text(new)
|
||||
PY
|
||||
if [[ "$TARGET" == "sitestream" || "$TARGET" == "all" ]]; then
|
||||
mkdir -p "$COMM_DIR/SiteStreamGrpc"
|
||||
cp "$GEN/Sitestream.cs" "$GEN/SitestreamGrpc.cs" "$COMM_DIR/SiteStreamGrpc/"
|
||||
echo "Copied regenerated files to SiteStreamGrpc/"
|
||||
fi
|
||||
if [[ "$TARGET" == "centralcontrol" || "$TARGET" == "all" ]]; then
|
||||
mkdir -p "$COMM_DIR/CentralControlGrpc"
|
||||
cp "$GEN/CentralControl.cs" "$GEN/CentralControlGrpc.cs" "$COMM_DIR/CentralControlGrpc/"
|
||||
echo "Copied regenerated files to CentralControlGrpc/"
|
||||
fi
|
||||
|
||||
# 5. Restore the backed-up csproj — i.e. drop the injected ItemGroup — so Docker
|
||||
# builds keep working.
|
||||
cp "$BACKUP" "$CSPROJ"
|
||||
rm -f "$BACKUP"
|
||||
trap - ERR
|
||||
|
||||
echo ""
|
||||
echo "Done. Review and commit:"
|
||||
echo " git diff src/ZB.MOM.WW.ScadaBridge.Communication/Protos/sitestream.proto"
|
||||
echo " git diff src/ZB.MOM.WW.ScadaBridge.Communication/Protos/"
|
||||
echo " git diff src/ZB.MOM.WW.ScadaBridge.Communication/SiteStreamGrpc/"
|
||||
echo " git diff src/ZB.MOM.WW.ScadaBridge.Communication/CentralControlGrpc/"
|
||||
echo " git diff -- src/ZB.MOM.WW.ScadaBridge.Communication/*.csproj # must be EMPTY"
|
||||
|
||||
@@ -141,3 +141,58 @@ reaches the page HTML** — has not executed since. Fix is a valid 32-hex SID in
|
||||
the same code path — but a live `SubscribeInstance` under load is untested here.
|
||||
- Key **rotation** on a live pair.
|
||||
- `docker-env2` was updated with its own key but not redeployed or gated.
|
||||
|
||||
---
|
||||
|
||||
## Phase 1A — central control plane (site→central over gRPC) — **PASS** (2026-07-22)
|
||||
|
||||
Branch `feat/grpc-central-control` @ `0e162cb2`. Rig rebuilt; **site-a flipped to
|
||||
`CentralTransport=Grpc`** with `CentralGrpcEndpoints=[central-a:8083, central-b:8083]`,
|
||||
**site-b/c left on Akka** (default) to prove coexistence. The site-a flag flip was a
|
||||
DoD-test-only rig edit — reverted from the branch, never committed (the plan keeps the default
|
||||
`Akka` until Phase 4).
|
||||
|
||||
### The defect this gate caught (T1A.2 shipped it; fixed in `0e162cb2`)
|
||||
|
||||
**First rebuild: central's entire HTTP surface was gone.** central-a logged only
|
||||
`Now listening on: http://[::]:8083` — no `:5000`. Central UI, the Management + Inbound API,
|
||||
and every `/health/*` endpoint (Traefik routing + `IActiveNodeGate` both depend on them) were
|
||||
dead. The node booted, joined the cluster and served gRPC fine; **no startup error.**
|
||||
|
||||
Cause: `builder.WebHost.ConfigureKestrel(o => o.ListenAnyIP(8083, Http2))` puts Kestrel into
|
||||
explicit-endpoints mode, which **suppresses the URLs from `ASPNETCORE_URLS`/`--urls`** — it is
|
||||
not additive, contrary to the comment T1A.2 shipped. Central's whole HTTP/1 surface lives on
|
||||
that URL (`http://+:5000` on the rig; a different port in production). The site branch has the
|
||||
same `ConfigureKestrel` shape but nothing on `ASPNETCORE_URLS` to lose — it binds every port it
|
||||
needs explicitly — which is why the pattern looked safe.
|
||||
|
||||
Every unit + E2E test uses `TestServer`, which never binds real Kestrel, so the whole suite
|
||||
(6,872) stayed green. Only a live node exposes a missing listener. Fix: parse the port(s) from
|
||||
the configured URLs and re-declare them (`Http1AndHttp2`) alongside the gRPC port (`Http2`) in
|
||||
the one `ConfigureKestrel` call — `Program.ParseHttpBindPorts` + `CentralHttpBindPortsTests`
|
||||
(14 cases). After the fix: `Now listening on: http://[::]:5000` **and** `:8083`; 9001 ready
|
||||
`200`/active Healthy, 9002 standby, Traefik LB `200`, CLI over the LB works.
|
||||
|
||||
### Checks (post-fix rebuild)
|
||||
|
||||
| # | Check | Result |
|
||||
|---|---|---|
|
||||
| 1 | site-a rides authenticated gRPC to `CentralControlService` | **PASS** — Heartbeat, `ReportSiteHealth`, `ReconcileSite` all HTTP/2 → 200; **0** auth failures on either central |
|
||||
| 2 | Heartbeat drives the active flag | **PASS** — 144+ heartbeats over gRPC; site-a `online=True` at central |
|
||||
| 3 | Health page live | **PASS** — `ReportSiteHealth` lands; central shows site-a `online=True`, sequence advancing, alongside Akka site-b/c |
|
||||
| 4 | Reconcile works after site restart | **PASS** — restarted `site-a-a`; it logged `Site→central transport: gRPC to 2 central endpoint(s)`, then `Reconcile pass … complete: 0 fetched, 0 failed, 0 orphan(s)` |
|
||||
| 5 | Coexistence | **PASS** — site-b/c log `Created ClusterClient to central`; both `online=True` — Akka and gRPC sites side by side |
|
||||
| 6 | Central HTTP surface intact under the new gRPC listener | **PASS** — after the fix (see above) |
|
||||
|
||||
### Not independently exercised on this gate
|
||||
|
||||
- **Notification e2e (`SubmitNotification`/`QueryNotificationStatus`) and audit ingest
|
||||
(`IngestAuditEvents`/`IngestCachedTelemetry`)** were NOT driven live: the rig has templates
|
||||
but **no deployed instance**, so nothing emits site→central notifications or audit rows on its
|
||||
own. These four RPCs traverse the identical `CentralControlGrpcService` →
|
||||
`CentralCommunicationActor` Ask path that checks 1–3 proved live under real auth, and their
|
||||
payload mappers carry 32 round-trip goldens — but the payloads themselves were not put over
|
||||
the wire here. **Phase 2's central-kill S&F soak is where they get their live workout;** flag
|
||||
for a fuller 1A proof if a deployed-instance rig is set up before then.
|
||||
- Cross-node failover/failback of the site→central channel under a central-node kill (unit-proven
|
||||
via TestServer; not exercised on the rig at 1A).
|
||||
|
||||
@@ -0,0 +1,185 @@
|
||||
using Akka.Actor;
|
||||
using Akka.Cluster.Tools.Client;
|
||||
using Akka.Event;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Audit;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Deployment;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Health;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Notification;
|
||||
|
||||
namespace ZB.MOM.WW.ScadaBridge.Communication.Actors;
|
||||
|
||||
/// <summary>
|
||||
/// The <see cref="ICentralTransport"/> that carries the seven site→central sends over Akka
|
||||
/// <c>ClusterClient</c> — the transport in production today, and the default. Every method is a
|
||||
/// verbatim lift of the corresponding <c>SiteCommunicationActor</c> send block: it forwards a
|
||||
/// <see cref="ClusterClient.Send"/> to <c>/user/central-communication</c> with the
|
||||
/// <paramref name="replyTo"/> as the send's sender, so central's reply routes straight back to the
|
||||
/// waiting Ask rather than through the site communication actor.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The <c>ClusterClient</c> reference arrives after construction via
|
||||
/// <see cref="SetCentralClient"/> (the actor forwards the <c>RegisterCentralClient</c> message it
|
||||
/// receives once the Host builds the client). Until then — and if central contact points are not
|
||||
/// configured at all — the client is null and each method answers the same transient-failure reply
|
||||
/// the old inline handlers did.
|
||||
/// </remarks>
|
||||
public sealed class AkkaCentralTransport : ICentralTransport
|
||||
{
|
||||
/// <summary>The receptionist-registered path of the central communication actor.</summary>
|
||||
private const string CentralPath = "/user/central-communication";
|
||||
|
||||
private readonly ILoggingAdapter? _log;
|
||||
private IActorRef? _centralClient;
|
||||
|
||||
/// <summary>Creates the transport with no logging adapter (behaviourally identical; warnings are dropped).</summary>
|
||||
public AkkaCentralTransport()
|
||||
{
|
||||
}
|
||||
|
||||
/// <summary>Creates the transport bound to the site communication actor's logging adapter.</summary>
|
||||
/// <param name="log">Logging adapter used for the "no ClusterClient registered" warnings.</param>
|
||||
public AkkaCentralTransport(ILoggingAdapter log)
|
||||
{
|
||||
_log = log;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Registers the central <c>ClusterClient</c> once the Host has built it. Called from the site
|
||||
/// communication actor's <c>RegisterCentralClient</c> handler.
|
||||
/// </summary>
|
||||
/// <param name="centralClient">The ClusterClient reaching the central cluster.</param>
|
||||
public void SetCentralClient(IActorRef centralClient)
|
||||
{
|
||||
_centralClient = centralClient;
|
||||
_log?.Info("Registered central ClusterClient");
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public void SubmitNotification(NotificationSubmit message, IActorRef replyTo)
|
||||
{
|
||||
if (_centralClient == null)
|
||||
{
|
||||
// No ClusterClient registered yet (e.g. central contact points not
|
||||
// configured, or registration not yet completed). A non-accepted ack
|
||||
// makes the S&F forwarder treat this as transient and retry later.
|
||||
_log?.Warning(
|
||||
"Cannot forward NotificationSubmit {0} — no central ClusterClient registered",
|
||||
message.NotificationId);
|
||||
replyTo.Tell(new NotificationSubmitAck(
|
||||
message.NotificationId, Accepted: false, Error: "Central ClusterClient not registered"));
|
||||
return;
|
||||
}
|
||||
|
||||
_log?.Debug("Forwarding NotificationSubmit {0} to central", message.NotificationId);
|
||||
_centralClient.Tell(new ClusterClient.Send(CentralPath, message), replyTo);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public void QueryNotificationStatus(NotificationStatusQuery message, IActorRef replyTo)
|
||||
{
|
||||
if (_centralClient == null)
|
||||
{
|
||||
// No ClusterClient registered yet. Reply Found: false so Notify.Status
|
||||
// falls back to the site S&F buffer to decide Forwarding vs Unknown.
|
||||
_log?.Warning(
|
||||
"Cannot forward NotificationStatusQuery {0} — no central ClusterClient registered",
|
||||
message.NotificationId);
|
||||
replyTo.Tell(new NotificationStatusResponse(
|
||||
message.CorrelationId, Found: false, Status: "Unknown",
|
||||
RetryCount: 0, LastError: null, DeliveredAt: null));
|
||||
return;
|
||||
}
|
||||
|
||||
_log?.Debug("Forwarding NotificationStatusQuery {0} to central", message.NotificationId);
|
||||
_centralClient.Tell(new ClusterClient.Send(CentralPath, message), replyTo);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public void IngestAuditEvents(IngestAuditEventsCommand message, IActorRef replyTo)
|
||||
{
|
||||
if (_centralClient == null)
|
||||
{
|
||||
// No ClusterClient registered yet. Faulting the Ask makes the
|
||||
// SiteAuditTelemetryActor drain loop treat this as transient and keep
|
||||
// the rows Pending for the next tick.
|
||||
_log?.Warning(
|
||||
"Cannot forward IngestAuditEventsCommand ({0} events) — no central ClusterClient registered",
|
||||
message.Events.Count);
|
||||
replyTo.Tell(new Status.Failure(
|
||||
new InvalidOperationException("Central ClusterClient not registered")));
|
||||
return;
|
||||
}
|
||||
|
||||
_log?.Debug("Forwarding IngestAuditEventsCommand ({0} events) to central", message.Events.Count);
|
||||
_centralClient.Tell(new ClusterClient.Send(CentralPath, message), replyTo);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public void IngestCachedTelemetry(IngestCachedTelemetryCommand message, IActorRef replyTo)
|
||||
{
|
||||
if (_centralClient == null)
|
||||
{
|
||||
_log?.Warning(
|
||||
"Cannot forward IngestCachedTelemetryCommand ({0} entries) — no central ClusterClient registered",
|
||||
message.Entries.Count);
|
||||
replyTo.Tell(new Status.Failure(
|
||||
new InvalidOperationException("Central ClusterClient not registered")));
|
||||
return;
|
||||
}
|
||||
|
||||
_log?.Debug("Forwarding IngestCachedTelemetryCommand ({0} entries) to central", message.Entries.Count);
|
||||
_centralClient.Tell(new ClusterClient.Send(CentralPath, message), replyTo);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public void ReconcileSite(ReconcileSiteRequest message, IActorRef replyTo)
|
||||
{
|
||||
if (_centralClient == null)
|
||||
{
|
||||
// No ClusterClient registered yet. Faulting the Ask makes the
|
||||
// SiteReconciliationActor treat the pass as best-effort-failed; it
|
||||
// logs a warning and retries reconcile on the next node startup.
|
||||
_log?.Warning(
|
||||
"Cannot forward ReconcileSiteRequest for site {0} node {1} — no central ClusterClient registered",
|
||||
message.SiteIdentifier, message.NodeId);
|
||||
replyTo.Tell(new Status.Failure(
|
||||
new InvalidOperationException("Central ClusterClient not registered")));
|
||||
return;
|
||||
}
|
||||
|
||||
_log?.Debug(
|
||||
"Forwarding ReconcileSiteRequest for site {0} node {1} ({2} local instance(s)) to central",
|
||||
message.SiteIdentifier, message.NodeId, message.LocalNameToRevisionHash.Count);
|
||||
_centralClient.Tell(new ClusterClient.Send(CentralPath, message), replyTo);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public void ReportSiteHealth(SiteHealthReport message, IActorRef replyTo)
|
||||
{
|
||||
if (_centralClient == null)
|
||||
{
|
||||
// No ClusterClient registered yet. A non-accepted ack makes the
|
||||
// sender's counter-restore path treat this tick as a loss.
|
||||
_log?.Warning(
|
||||
"Cannot forward SiteHealthReport #{0} — no central ClusterClient registered",
|
||||
message.SequenceNumber);
|
||||
replyTo.Tell(new SiteHealthReportAck(
|
||||
message.SiteId, message.SequenceNumber, Accepted: false,
|
||||
Error: "Central ClusterClient not registered"));
|
||||
return;
|
||||
}
|
||||
|
||||
_centralClient.Tell(new ClusterClient.Send(CentralPath, message), replyTo);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public void SendHeartbeat(HeartbeatMessage message, IActorRef self)
|
||||
{
|
||||
if (_centralClient == null)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
_centralClient.Tell(new ClusterClient.Send(CentralPath, message), self);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,78 @@
|
||||
using Akka.Actor;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Audit;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Deployment;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Health;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Notification;
|
||||
|
||||
namespace ZB.MOM.WW.ScadaBridge.Communication.Actors;
|
||||
|
||||
/// <summary>
|
||||
/// The site→central transport seam: one method per the seven messages
|
||||
/// <see cref="SiteCommunicationActor"/> sends to <c>/user/central-communication</c> today.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// <para>
|
||||
/// The actor's receive handlers no longer own the wire plumbing — they capture the current
|
||||
/// <c>Sender</c> and hand it to the transport as <paramref name="replyTo"/>. Two implementations
|
||||
/// exist behind the <c>ScadaBridge:Communication:CentralTransport</c> flag: the default
|
||||
/// <see cref="AkkaCentralTransport"/> (verbatim of the old <c>ClusterClient.Send</c> path,
|
||||
/// including the exact sender-forwarding that routes central's reply straight back to the waiting
|
||||
/// Ask) and <see cref="Grpc.GrpcCentralTransport"/> (a gRPC dial of <c>CentralControlService</c>).
|
||||
/// </para>
|
||||
/// <para>
|
||||
/// <b>Reply/fault contract, identical on both transports.</b> Each Ask-returning method (all but
|
||||
/// the heartbeat) guarantees exactly one reply eventually lands at <paramref name="replyTo"/>:
|
||||
/// either the real reply type the caller Asks for, or a transient-failure signal. The Akka path
|
||||
/// sends a not-accepted ack / <see cref="Status.Failure"/> when no ClusterClient is registered;
|
||||
/// the gRPC path sends <see cref="Status.Failure"/> on any non-OK status (a timeout or an
|
||||
/// <c>Unavailable</c> that could not be failed over). Both are what the S&F / audit / health
|
||||
/// layers above the seam already treat as transient — rows stay buffered, counters restore, the
|
||||
/// pass re-runs.
|
||||
/// </para>
|
||||
/// <para>
|
||||
/// <b>The heartbeat stays fire-and-forget end-to-end.</b> <see cref="SendHeartbeat"/> takes no
|
||||
/// <c>replyTo</c> and never surfaces a fault: a transport failure is swallowed and logged, exactly
|
||||
/// as the old <c>Tell</c> dropped it. A failing heartbeat must never fault the site's heartbeat
|
||||
/// timer path.
|
||||
/// </para>
|
||||
/// </remarks>
|
||||
public interface ICentralTransport
|
||||
{
|
||||
/// <summary>Forwards a buffered notification; central replies <see cref="NotificationSubmitAck"/> to <paramref name="replyTo"/>.</summary>
|
||||
/// <param name="message">The notification submission.</param>
|
||||
/// <param name="replyTo">The actor (the S&F forwarder's Ask) the ack routes back to.</param>
|
||||
void SubmitNotification(NotificationSubmit message, IActorRef replyTo);
|
||||
|
||||
/// <summary>Forwards a Notify.Status query; central replies <see cref="NotificationStatusResponse"/> to <paramref name="replyTo"/>.</summary>
|
||||
/// <param name="message">The status query.</param>
|
||||
/// <param name="replyTo">The actor (the Notify helper's Ask) the response routes back to.</param>
|
||||
void QueryNotificationStatus(NotificationStatusQuery message, IActorRef replyTo);
|
||||
|
||||
/// <summary>Pushes a batch of audit events; central replies <see cref="IngestAuditEventsReply"/> to <paramref name="replyTo"/>.</summary>
|
||||
/// <param name="message">The audit-event ingest command.</param>
|
||||
/// <param name="replyTo">The actor (the telemetry drain's Ask) the reply routes back to.</param>
|
||||
void IngestAuditEvents(IngestAuditEventsCommand message, IActorRef replyTo);
|
||||
|
||||
/// <summary>Pushes a batch of combined cached-call telemetry; central replies <see cref="IngestCachedTelemetryReply"/> to <paramref name="replyTo"/>.</summary>
|
||||
/// <param name="message">The cached-telemetry ingest command.</param>
|
||||
/// <param name="replyTo">The actor (the telemetry drain's Ask) the reply routes back to.</param>
|
||||
void IngestCachedTelemetry(IngestCachedTelemetryCommand message, IActorRef replyTo);
|
||||
|
||||
/// <summary>Reports a node's startup inventory; central replies <see cref="ReconcileSiteResponse"/> to <paramref name="replyTo"/>.</summary>
|
||||
/// <param name="message">The reconcile request.</param>
|
||||
/// <param name="replyTo">The actor (the reconciliation Ask) the response routes back to.</param>
|
||||
void ReconcileSite(ReconcileSiteRequest message, IActorRef replyTo);
|
||||
|
||||
/// <summary>Reports periodic site health; central replies <see cref="SiteHealthReportAck"/> to <paramref name="replyTo"/>.</summary>
|
||||
/// <param name="message">The health report.</param>
|
||||
/// <param name="replyTo">The actor (the health transport's Ask) the ack routes back to.</param>
|
||||
void ReportSiteHealth(SiteHealthReport message, IActorRef replyTo);
|
||||
|
||||
/// <summary>
|
||||
/// Sends an application heartbeat, fire-and-forget. Never replies and never faults; a failure
|
||||
/// is swallowed and logged.
|
||||
/// </summary>
|
||||
/// <param name="message">The heartbeat.</param>
|
||||
/// <param name="self">The site communication actor, used as the sender on the Akka path (ignored on gRPC).</param>
|
||||
void SendHeartbeat(HeartbeatMessage message, IActorRef self);
|
||||
}
|
||||
@@ -1,6 +1,5 @@
|
||||
using Akka.Actor;
|
||||
using Akka.Cluster;
|
||||
using Akka.Cluster.Tools.Client;
|
||||
using Akka.Event;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Artifacts;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Audit;
|
||||
@@ -45,10 +44,14 @@ public class SiteCommunicationActor : ReceiveActor, IWithTimers
|
||||
private readonly IActorRef _deploymentManagerProxy;
|
||||
|
||||
/// <summary>
|
||||
/// ClusterClient reference for sending messages to the central cluster.
|
||||
/// Set via RegisterCentralClient message.
|
||||
/// The site→central transport. Finalized in <see cref="PreStart"/> to the injected instance,
|
||||
/// or a default <see cref="AkkaCentralTransport"/> (ClusterClient) when none is supplied — so
|
||||
/// the seven site→central sends delegate here rather than owning the wire plumbing inline.
|
||||
/// </summary>
|
||||
private IActorRef? _centralClient;
|
||||
private ICentralTransport _transport;
|
||||
|
||||
/// <summary>The transport supplied by the Host (null selects the default Akka transport).</summary>
|
||||
private readonly ICentralTransport? _injectedTransport;
|
||||
|
||||
/// <summary>
|
||||
/// Local actor references for routing specific message patterns.
|
||||
@@ -73,24 +76,36 @@ public class SiteCommunicationActor : ReceiveActor, IWithTimers
|
||||
/// pass a stub so they do not need to load Akka.Cluster into the <c>TestKit</c>
|
||||
/// ActorSystem.
|
||||
/// </param>
|
||||
/// <param name="transport">
|
||||
/// The site→central transport. <c>null</c> (the default, used by every existing test and by
|
||||
/// the Host's Akka path) selects an <see cref="AkkaCentralTransport"/> over ClusterClient; the
|
||||
/// Host injects a <see cref="Grpc.GrpcCentralTransport"/> when
|
||||
/// <c>ScadaBridge:Communication:CentralTransport</c> is <c>Grpc</c>.
|
||||
/// </param>
|
||||
public SiteCommunicationActor(
|
||||
string siteId,
|
||||
CommunicationOptions options,
|
||||
IActorRef deploymentManagerProxy,
|
||||
Func<bool>? isActiveCheck = null,
|
||||
Func<string, string?>? failOverRole = null)
|
||||
Func<string, string?>? failOverRole = null,
|
||||
ICentralTransport? transport = null)
|
||||
{
|
||||
_siteId = siteId;
|
||||
_options = options;
|
||||
_deploymentManagerProxy = deploymentManagerProxy;
|
||||
_isActiveCheck = isActiveCheck ?? DefaultIsActiveCheck;
|
||||
_failOverRole = failOverRole ?? DefaultFailOverRole;
|
||||
_injectedTransport = transport;
|
||||
// Finalized in PreStart (where _log is usable for the default transport); assigned here
|
||||
// too so the field is definitely-assigned for the constructor's Receive closures.
|
||||
_transport = transport!;
|
||||
|
||||
// Registration
|
||||
// Registration. Feeding the ClusterClient into the transport is a no-op unless the
|
||||
// default/Akka transport is in use — the gRPC transport dials configured endpoints and
|
||||
// never receives this message (the Host does not create a ClusterClient for it).
|
||||
Receive<RegisterCentralClient>(msg =>
|
||||
{
|
||||
_centralClient = msg.Client;
|
||||
_log.Info("Registered central ClusterClient");
|
||||
(_transport as AkkaCentralTransport)?.SetCentralClient(msg.Client);
|
||||
});
|
||||
Receive<RegisterLocalHandler>(HandleRegisterLocalHandler);
|
||||
|
||||
@@ -274,157 +289,33 @@ public class SiteCommunicationActor : ReceiveActor, IWithTimers
|
||||
// from cluster state, not from who received the message.
|
||||
Receive<TriggerSiteFailover>(HandleTriggerSiteFailover);
|
||||
|
||||
// Notification Outbox: forward a buffered notification submitted by the site
|
||||
// Store-and-Forward Engine to the central cluster. The original Sender (the
|
||||
// S&F forwarder's Ask) is forwarded as the ClusterClient.Send sender so the
|
||||
// NotificationSubmitAck routes straight back to the waiting Ask, not here.
|
||||
Receive<NotificationSubmit>(msg =>
|
||||
{
|
||||
if (_centralClient == null)
|
||||
{
|
||||
// No ClusterClient registered yet (e.g. central contact points not
|
||||
// configured, or registration not yet completed). A non-accepted ack
|
||||
// makes the S&F forwarder treat this as transient and retry later.
|
||||
_log.Warning(
|
||||
"Cannot forward NotificationSubmit {0} — no central ClusterClient registered",
|
||||
msg.NotificationId);
|
||||
Sender.Tell(new NotificationSubmitAck(
|
||||
msg.NotificationId, Accepted: false, Error: "Central ClusterClient not registered"));
|
||||
return;
|
||||
}
|
||||
// The seven site→central sends now delegate to the injected transport (ClusterClient by
|
||||
// default, gRPC when configured). Each handler captures the current Sender as the reply
|
||||
// target so central's reply routes straight back to the waiting Ask, not through this
|
||||
// actor — the exact sender-forwarding the ClusterClient path relied on. The per-message
|
||||
// "no transport / not-accepted" fallbacks live inside the transport now.
|
||||
|
||||
_log.Debug("Forwarding NotificationSubmit {0} to central", msg.NotificationId);
|
||||
_centralClient.Tell(
|
||||
new ClusterClient.Send("/user/central-communication", msg), Sender);
|
||||
});
|
||||
// Notification Outbox: forward a buffered notification (S&F forwarder's Ask → ack back).
|
||||
Receive<NotificationSubmit>(msg => _transport.SubmitNotification(msg, Sender));
|
||||
|
||||
// Notification Outbox: forward a Notify.Status query to the central cluster.
|
||||
// The original Sender (the Notify helper's Ask) is forwarded as the
|
||||
// ClusterClient.Send sender so the NotificationStatusResponse routes straight
|
||||
// back to the waiting Ask, not here.
|
||||
Receive<NotificationStatusQuery>(msg =>
|
||||
{
|
||||
if (_centralClient == null)
|
||||
{
|
||||
// No ClusterClient registered yet. Reply Found: false so Notify.Status
|
||||
// falls back to the site S&F buffer to decide Forwarding vs Unknown.
|
||||
_log.Warning(
|
||||
"Cannot forward NotificationStatusQuery {0} — no central ClusterClient registered",
|
||||
msg.NotificationId);
|
||||
Sender.Tell(new NotificationStatusResponse(
|
||||
msg.CorrelationId, Found: false, Status: "Unknown",
|
||||
RetryCount: 0, LastError: null, DeliveredAt: null));
|
||||
return;
|
||||
}
|
||||
// Notification Outbox: forward a Notify.Status query (Notify helper's Ask → response back).
|
||||
Receive<NotificationStatusQuery>(msg => _transport.QueryNotificationStatus(msg, Sender));
|
||||
|
||||
_log.Debug("Forwarding NotificationStatusQuery {0} to central", msg.NotificationId);
|
||||
_centralClient.Tell(
|
||||
new ClusterClient.Send("/user/central-communication", msg), Sender);
|
||||
});
|
||||
// Audit Log: forward a batch of site-local audit events (telemetry drain's Ask → reply back).
|
||||
Receive<IngestAuditEventsCommand>(msg => _transport.IngestAuditEvents(msg, Sender));
|
||||
|
||||
// Audit Log: forward a batch of site-local audit events to the
|
||||
// central cluster. The site SiteAuditTelemetryActor drains its SQLite
|
||||
// Pending queue through the ClusterClientSiteAuditClient, which Asks
|
||||
// this actor; the original Sender (that Ask) is passed as the
|
||||
// ClusterClient.Send sender so the IngestAuditEventsReply routes
|
||||
// straight back to the waiting Ask, not here. Mirrors NotificationSubmit.
|
||||
Receive<IngestAuditEventsCommand>(msg =>
|
||||
{
|
||||
if (_centralClient == null)
|
||||
{
|
||||
// No ClusterClient registered yet (e.g. central contact points
|
||||
// not configured, or registration not yet completed). Faulting
|
||||
// the Ask makes the SiteAuditTelemetryActor drain loop treat
|
||||
// this as transient and keep the rows Pending for the next tick.
|
||||
_log.Warning(
|
||||
"Cannot forward IngestAuditEventsCommand ({0} events) — no central ClusterClient registered",
|
||||
msg.Events.Count);
|
||||
Sender.Tell(new Status.Failure(
|
||||
new InvalidOperationException("Central ClusterClient not registered")));
|
||||
return;
|
||||
}
|
||||
// Audit Log: forward a batch of combined cached-call telemetry (telemetry drain's Ask → reply back).
|
||||
Receive<IngestCachedTelemetryCommand>(msg => _transport.IngestCachedTelemetry(msg, Sender));
|
||||
|
||||
_log.Debug("Forwarding IngestAuditEventsCommand ({0} events) to central", msg.Events.Count);
|
||||
_centralClient.Tell(
|
||||
new ClusterClient.Send("/user/central-communication", msg), Sender);
|
||||
});
|
||||
// Site startup reconciliation: forward the node's local-inventory request (reconcile Ask → response back).
|
||||
Receive<ReconcileSiteRequest>(msg => _transport.ReconcileSite(msg, Sender));
|
||||
|
||||
// Audit Log: forward a batch of combined cached-call telemetry
|
||||
// packets to the central cluster. Same forward + reply-routing pattern
|
||||
// as IngestAuditEventsCommand; central replies with an
|
||||
// IngestCachedTelemetryReply.
|
||||
Receive<IngestCachedTelemetryCommand>(msg =>
|
||||
{
|
||||
if (_centralClient == null)
|
||||
{
|
||||
_log.Warning(
|
||||
"Cannot forward IngestCachedTelemetryCommand ({0} entries) — no central ClusterClient registered",
|
||||
msg.Entries.Count);
|
||||
Sender.Tell(new Status.Failure(
|
||||
new InvalidOperationException("Central ClusterClient not registered")));
|
||||
return;
|
||||
}
|
||||
|
||||
_log.Debug("Forwarding IngestCachedTelemetryCommand ({0} entries) to central", msg.Entries.Count);
|
||||
_centralClient.Tell(
|
||||
new ClusterClient.Send("/user/central-communication", msg), Sender);
|
||||
});
|
||||
|
||||
// Site startup reconciliation: forward the node's local-inventory
|
||||
// ReconcileSiteRequest to the central cluster. The original Sender (the
|
||||
// SiteReconciliationActor's Ask) is passed as the ClusterClient.Send sender so
|
||||
// the ReconcileSiteResponse routes straight back to the waiting Ask, not here.
|
||||
// Mirrors IngestAuditEventsCommand.
|
||||
Receive<ReconcileSiteRequest>(msg =>
|
||||
{
|
||||
if (_centralClient == null)
|
||||
{
|
||||
// No ClusterClient registered yet (e.g. central contact points not
|
||||
// configured, or registration not yet completed). Faulting the Ask makes
|
||||
// the SiteReconciliationActor treat the pass as best-effort-failed; it
|
||||
// logs a warning and retries reconcile on the next node startup.
|
||||
_log.Warning(
|
||||
"Cannot forward ReconcileSiteRequest for site {0} node {1} — no central ClusterClient registered",
|
||||
msg.SiteIdentifier, msg.NodeId);
|
||||
Sender.Tell(new Status.Failure(
|
||||
new InvalidOperationException("Central ClusterClient not registered")));
|
||||
return;
|
||||
}
|
||||
|
||||
_log.Debug(
|
||||
"Forwarding ReconcileSiteRequest for site {0} node {1} ({2} local instance(s)) to central",
|
||||
msg.SiteIdentifier, msg.NodeId, msg.LocalNameToRevisionHash.Count);
|
||||
_centralClient.Tell(
|
||||
new ClusterClient.Send("/user/central-communication", msg), Sender);
|
||||
});
|
||||
|
||||
// Internal: send heartbeat tick
|
||||
// Internal: send heartbeat tick.
|
||||
Receive<SendHeartbeat>(_ => SendHeartbeatToCentral());
|
||||
|
||||
// Internal: forward health report to central. The original Sender (the
|
||||
// AkkaHealthReportTransport's Ask) is forwarded as the ClusterClient.Send
|
||||
// sender so the central SiteHealthReportAck routes straight back to the
|
||||
// waiting Ask — making report delivery observable end-to-end (review 01
|
||||
// [Medium]). Mirrors the NotificationSubmit ack pattern above.
|
||||
Receive<SiteHealthReport>(msg =>
|
||||
{
|
||||
if (_centralClient == null)
|
||||
{
|
||||
// No ClusterClient registered yet. A non-accepted ack makes the
|
||||
// sender's counter-restore path treat this tick as a loss.
|
||||
_log.Warning(
|
||||
"Cannot forward SiteHealthReport #{0} — no central ClusterClient registered",
|
||||
msg.SequenceNumber);
|
||||
Sender.Tell(new SiteHealthReportAck(
|
||||
msg.SiteId, msg.SequenceNumber, Accepted: false,
|
||||
Error: "Central ClusterClient not registered"));
|
||||
return;
|
||||
}
|
||||
|
||||
_centralClient.Tell(
|
||||
new ClusterClient.Send("/user/central-communication", msg), Sender);
|
||||
});
|
||||
|
||||
// Internal: forward the periodic health report (health transport's Ask → ack back), so a
|
||||
// lost report is observable end-to-end and the sender can restore its per-interval counters.
|
||||
Receive<SiteHealthReport>(msg => _transport.ReportSiteHealth(msg, Sender));
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
@@ -443,6 +334,12 @@ public class SiteCommunicationActor : ReceiveActor, IWithTimers
|
||||
/// <inheritdoc />
|
||||
protected override void PreStart()
|
||||
{
|
||||
// Finalize the transport now that the actor context (and _log) exist. The default Akka
|
||||
// transport is given this actor's logging adapter so the "no ClusterClient registered"
|
||||
// warnings are preserved exactly. PreStart always runs before any message, so the Receive
|
||||
// closures above see a non-null _transport.
|
||||
_transport = _injectedTransport ?? new AkkaCentralTransport(_log);
|
||||
|
||||
_log.Info("SiteCommunicationActor started for site {0}", _siteId);
|
||||
|
||||
// Schedule periodic heartbeat to central. Uses the application heartbeat
|
||||
@@ -478,9 +375,6 @@ public class SiteCommunicationActor : ReceiveActor, IWithTimers
|
||||
|
||||
private void SendHeartbeatToCentral()
|
||||
{
|
||||
if (_centralClient == null)
|
||||
return;
|
||||
|
||||
var hostname = Environment.MachineName;
|
||||
|
||||
// Stamp HeartbeatMessage.IsActive with this node's
|
||||
@@ -512,8 +406,17 @@ public class SiteCommunicationActor : ReceiveActor, IWithTimers
|
||||
IsActive: isActive,
|
||||
DateTimeOffset.UtcNow);
|
||||
|
||||
_centralClient.Tell(
|
||||
new ClusterClient.Send("/user/central-communication", heartbeat), Self);
|
||||
// Fire-and-forget on both transports: a failure here must never fault the heartbeat timer
|
||||
// path. Both real transports swallow their own errors; this catch is a belt-and-braces
|
||||
// guarantee that no transport (including a future one) can turn a heartbeat into a fault.
|
||||
try
|
||||
{
|
||||
_transport.SendHeartbeat(heartbeat, Self);
|
||||
}
|
||||
catch (Exception ex)
|
||||
{
|
||||
_log.Debug(ex, "Heartbeat send for site {0} failed; swallowed (heartbeats are fire-and-forget)", _siteId);
|
||||
}
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,704 @@
|
||||
// <auto-generated>
|
||||
// Generated by the protocol buffer compiler. DO NOT EDIT!
|
||||
// source: Protos/central_control.proto
|
||||
// </auto-generated>
|
||||
#pragma warning disable 0414, 1591, 8981, 0612
|
||||
#region Designer generated code
|
||||
|
||||
using grpc = global::Grpc.Core;
|
||||
|
||||
namespace ZB.MOM.WW.ScadaBridge.Communication.Grpc {
|
||||
/// <summary>
|
||||
/// Central-hosted control plane (Phase 1A of the ClusterClient→gRPC migration).
|
||||
///
|
||||
/// Direction: SITE is the client, CENTRAL is the server — the inverse of
|
||||
/// SiteStreamService, where central dials the site. That asymmetry is deliberate
|
||||
/// and mirrors the direction the Akka ClusterClient traffic flows today: these
|
||||
/// seven calls are exactly the seven messages SiteCommunicationActor sends to
|
||||
/// /user/central-communication.
|
||||
///
|
||||
/// Every call is gated by ControlPlaneAuthInterceptor: `authorization: Bearer <psk>`
|
||||
/// plus the `x-scadabridge-site` metadata header naming which site's preshared key
|
||||
/// central must verify against.
|
||||
/// </summary>
|
||||
public static partial class CentralControlService
|
||||
{
|
||||
static readonly string __ServiceName = "scadabridge.centralcontrol.v1.CentralControlService";
|
||||
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
static void __Helper_SerializeMessage(global::Google.Protobuf.IMessage message, grpc::SerializationContext context)
|
||||
{
|
||||
#if !GRPC_DISABLE_PROTOBUF_BUFFER_SERIALIZATION
|
||||
if (message is global::Google.Protobuf.IBufferMessage)
|
||||
{
|
||||
context.SetPayloadLength(message.CalculateSize());
|
||||
global::Google.Protobuf.MessageExtensions.WriteTo(message, context.GetBufferWriter());
|
||||
context.Complete();
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
context.Complete(global::Google.Protobuf.MessageExtensions.ToByteArray(message));
|
||||
}
|
||||
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
static class __Helper_MessageCache<T>
|
||||
{
|
||||
public static readonly bool IsBufferMessage = global::System.Reflection.IntrospectionExtensions.GetTypeInfo(typeof(global::Google.Protobuf.IBufferMessage)).IsAssignableFrom(typeof(T));
|
||||
}
|
||||
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
static T __Helper_DeserializeMessage<T>(grpc::DeserializationContext context, global::Google.Protobuf.MessageParser<T> parser) where T : global::Google.Protobuf.IMessage<T>
|
||||
{
|
||||
#if !GRPC_DISABLE_PROTOBUF_BUFFER_SERIALIZATION
|
||||
if (__Helper_MessageCache<T>.IsBufferMessage)
|
||||
{
|
||||
return parser.ParseFrom(context.PayloadAsReadOnlySequence());
|
||||
}
|
||||
#endif
|
||||
return parser.ParseFrom(context.PayloadAsNewBuffer());
|
||||
}
|
||||
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
static readonly grpc::Marshaller<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationSubmitDto> __Marshaller_scadabridge_centralcontrol_v1_NotificationSubmitDto = grpc::Marshallers.Create(__Helper_SerializeMessage, context => __Helper_DeserializeMessage(context, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationSubmitDto.Parser));
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
static readonly grpc::Marshaller<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationSubmitAckDto> __Marshaller_scadabridge_centralcontrol_v1_NotificationSubmitAckDto = grpc::Marshallers.Create(__Helper_SerializeMessage, context => __Helper_DeserializeMessage(context, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationSubmitAckDto.Parser));
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
static readonly grpc::Marshaller<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationStatusQueryDto> __Marshaller_scadabridge_centralcontrol_v1_NotificationStatusQueryDto = grpc::Marshallers.Create(__Helper_SerializeMessage, context => __Helper_DeserializeMessage(context, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationStatusQueryDto.Parser));
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
static readonly grpc::Marshaller<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationStatusResponseDto> __Marshaller_scadabridge_centralcontrol_v1_NotificationStatusResponseDto = grpc::Marshallers.Create(__Helper_SerializeMessage, context => __Helper_DeserializeMessage(context, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationStatusResponseDto.Parser));
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
static readonly grpc::Marshaller<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.AuditEventBatch> __Marshaller_sitestream_AuditEventBatch = grpc::Marshallers.Create(__Helper_SerializeMessage, context => __Helper_DeserializeMessage(context, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.AuditEventBatch.Parser));
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
static readonly grpc::Marshaller<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.IngestAck> __Marshaller_sitestream_IngestAck = grpc::Marshallers.Create(__Helper_SerializeMessage, context => __Helper_DeserializeMessage(context, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.IngestAck.Parser));
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
static readonly grpc::Marshaller<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.CachedTelemetryBatch> __Marshaller_sitestream_CachedTelemetryBatch = grpc::Marshallers.Create(__Helper_SerializeMessage, context => __Helper_DeserializeMessage(context, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.CachedTelemetryBatch.Parser));
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
static readonly grpc::Marshaller<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.ReconcileSiteRequestDto> __Marshaller_scadabridge_centralcontrol_v1_ReconcileSiteRequestDto = grpc::Marshallers.Create(__Helper_SerializeMessage, context => __Helper_DeserializeMessage(context, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.ReconcileSiteRequestDto.Parser));
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
static readonly grpc::Marshaller<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.ReconcileSiteResponseDto> __Marshaller_scadabridge_centralcontrol_v1_ReconcileSiteResponseDto = grpc::Marshallers.Create(__Helper_SerializeMessage, context => __Helper_DeserializeMessage(context, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.ReconcileSiteResponseDto.Parser));
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
static readonly grpc::Marshaller<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.SiteHealthReportDto> __Marshaller_scadabridge_centralcontrol_v1_SiteHealthReportDto = grpc::Marshallers.Create(__Helper_SerializeMessage, context => __Helper_DeserializeMessage(context, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.SiteHealthReportDto.Parser));
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
static readonly grpc::Marshaller<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.SiteHealthReportAckDto> __Marshaller_scadabridge_centralcontrol_v1_SiteHealthReportAckDto = grpc::Marshallers.Create(__Helper_SerializeMessage, context => __Helper_DeserializeMessage(context, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.SiteHealthReportAckDto.Parser));
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
static readonly grpc::Marshaller<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.HeartbeatDto> __Marshaller_scadabridge_centralcontrol_v1_HeartbeatDto = grpc::Marshallers.Create(__Helper_SerializeMessage, context => __Helper_DeserializeMessage(context, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.HeartbeatDto.Parser));
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
static readonly grpc::Marshaller<global::Google.Protobuf.WellKnownTypes.Empty> __Marshaller_google_protobuf_Empty = grpc::Marshallers.Create(__Helper_SerializeMessage, context => __Helper_DeserializeMessage(context, global::Google.Protobuf.WellKnownTypes.Empty.Parser));
|
||||
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
static readonly grpc::Method<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationSubmitDto, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationSubmitAckDto> __Method_SubmitNotification = new grpc::Method<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationSubmitDto, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationSubmitAckDto>(
|
||||
grpc::MethodType.Unary,
|
||||
__ServiceName,
|
||||
"SubmitNotification",
|
||||
__Marshaller_scadabridge_centralcontrol_v1_NotificationSubmitDto,
|
||||
__Marshaller_scadabridge_centralcontrol_v1_NotificationSubmitAckDto);
|
||||
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
static readonly grpc::Method<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationStatusQueryDto, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationStatusResponseDto> __Method_QueryNotificationStatus = new grpc::Method<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationStatusQueryDto, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationStatusResponseDto>(
|
||||
grpc::MethodType.Unary,
|
||||
__ServiceName,
|
||||
"QueryNotificationStatus",
|
||||
__Marshaller_scadabridge_centralcontrol_v1_NotificationStatusQueryDto,
|
||||
__Marshaller_scadabridge_centralcontrol_v1_NotificationStatusResponseDto);
|
||||
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
static readonly grpc::Method<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.AuditEventBatch, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.IngestAck> __Method_IngestAuditEvents = new grpc::Method<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.AuditEventBatch, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.IngestAck>(
|
||||
grpc::MethodType.Unary,
|
||||
__ServiceName,
|
||||
"IngestAuditEvents",
|
||||
__Marshaller_sitestream_AuditEventBatch,
|
||||
__Marshaller_sitestream_IngestAck);
|
||||
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
static readonly grpc::Method<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.CachedTelemetryBatch, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.IngestAck> __Method_IngestCachedTelemetry = new grpc::Method<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.CachedTelemetryBatch, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.IngestAck>(
|
||||
grpc::MethodType.Unary,
|
||||
__ServiceName,
|
||||
"IngestCachedTelemetry",
|
||||
__Marshaller_sitestream_CachedTelemetryBatch,
|
||||
__Marshaller_sitestream_IngestAck);
|
||||
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
static readonly grpc::Method<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.ReconcileSiteRequestDto, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.ReconcileSiteResponseDto> __Method_ReconcileSite = new grpc::Method<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.ReconcileSiteRequestDto, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.ReconcileSiteResponseDto>(
|
||||
grpc::MethodType.Unary,
|
||||
__ServiceName,
|
||||
"ReconcileSite",
|
||||
__Marshaller_scadabridge_centralcontrol_v1_ReconcileSiteRequestDto,
|
||||
__Marshaller_scadabridge_centralcontrol_v1_ReconcileSiteResponseDto);
|
||||
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
static readonly grpc::Method<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.SiteHealthReportDto, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.SiteHealthReportAckDto> __Method_ReportSiteHealth = new grpc::Method<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.SiteHealthReportDto, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.SiteHealthReportAckDto>(
|
||||
grpc::MethodType.Unary,
|
||||
__ServiceName,
|
||||
"ReportSiteHealth",
|
||||
__Marshaller_scadabridge_centralcontrol_v1_SiteHealthReportDto,
|
||||
__Marshaller_scadabridge_centralcontrol_v1_SiteHealthReportAckDto);
|
||||
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
static readonly grpc::Method<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.HeartbeatDto, global::Google.Protobuf.WellKnownTypes.Empty> __Method_Heartbeat = new grpc::Method<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.HeartbeatDto, global::Google.Protobuf.WellKnownTypes.Empty>(
|
||||
grpc::MethodType.Unary,
|
||||
__ServiceName,
|
||||
"Heartbeat",
|
||||
__Marshaller_scadabridge_centralcontrol_v1_HeartbeatDto,
|
||||
__Marshaller_google_protobuf_Empty);
|
||||
|
||||
/// <summary>Service descriptor</summary>
|
||||
public static global::Google.Protobuf.Reflection.ServiceDescriptor Descriptor
|
||||
{
|
||||
get { return global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.CentralControlReflection.Descriptor.Services[0]; }
|
||||
}
|
||||
|
||||
/// <summary>Base class for server-side implementations of CentralControlService</summary>
|
||||
[grpc::BindServiceMethod(typeof(CentralControlService), "BindService")]
|
||||
public abstract partial class CentralControlServiceBase
|
||||
{
|
||||
/// <summary>
|
||||
/// Store-and-forward handoff of one notification for central delivery. The
|
||||
/// ack is idempotent on notification_id — a duplicate submit after a lost ack
|
||||
/// must not produce a second delivery.
|
||||
/// </summary>
|
||||
/// <param name="request">The request received from the client.</param>
|
||||
/// <param name="context">The context of the server-side call handler being invoked.</param>
|
||||
/// <returns>The response to send back to the client (wrapped by a task).</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual global::System.Threading.Tasks.Task<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationSubmitAckDto> SubmitNotification(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationSubmitDto request, grpc::ServerCallContext context)
|
||||
{
|
||||
throw new grpc::RpcException(new grpc::Status(grpc::StatusCode.Unimplemented, ""));
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Notify.Status(id) round-trip for a notification that has already left the
|
||||
/// site buffer. `found = false` sends the caller back to the site-local buffer
|
||||
/// to decide Forwarding vs Unknown.
|
||||
/// </summary>
|
||||
/// <param name="request">The request received from the client.</param>
|
||||
/// <param name="context">The context of the server-side call handler being invoked.</param>
|
||||
/// <returns>The response to send back to the client (wrapped by a task).</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual global::System.Threading.Tasks.Task<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationStatusResponseDto> QueryNotificationStatus(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationStatusQueryDto request, grpc::ServerCallContext context)
|
||||
{
|
||||
throw new grpc::RpcException(new grpc::Status(grpc::StatusCode.Unimplemented, ""));
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Audit Log (#23) push telemetry. Reuses the SiteStreamService messages: the
|
||||
/// batch a site drains from its SQLite hot path is byte-identical whichever
|
||||
/// transport carries it.
|
||||
/// </summary>
|
||||
/// <param name="request">The request received from the client.</param>
|
||||
/// <param name="context">The context of the server-side call handler being invoked.</param>
|
||||
/// <returns>The response to send back to the client (wrapped by a task).</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual global::System.Threading.Tasks.Task<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.IngestAck> IngestAuditEvents(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.AuditEventBatch request, grpc::ServerCallContext context)
|
||||
{
|
||||
throw new grpc::RpcException(new grpc::Status(grpc::StatusCode.Unimplemented, ""));
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Audit Log (#23) M3 combined cached-call telemetry (audit row + SiteCalls
|
||||
/// operational upsert, written in one central transaction).
|
||||
/// </summary>
|
||||
/// <param name="request">The request received from the client.</param>
|
||||
/// <param name="context">The context of the server-side call handler being invoked.</param>
|
||||
/// <returns>The response to send back to the client (wrapped by a task).</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual global::System.Threading.Tasks.Task<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.IngestAck> IngestCachedTelemetry(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.CachedTelemetryBatch request, grpc::ServerCallContext context)
|
||||
{
|
||||
throw new grpc::RpcException(new grpc::Status(grpc::StatusCode.Unimplemented, ""));
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Node-startup self-heal: the node's local deployed inventory in, fetch
|
||||
/// tokens for whatever it is missing or stale out.
|
||||
/// </summary>
|
||||
/// <param name="request">The request received from the client.</param>
|
||||
/// <param name="context">The context of the server-side call handler being invoked.</param>
|
||||
/// <returns>The response to send back to the client (wrapped by a task).</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual global::System.Threading.Tasks.Task<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.ReconcileSiteResponseDto> ReconcileSite(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.ReconcileSiteRequestDto request, grpc::ServerCallContext context)
|
||||
{
|
||||
throw new grpc::RpcException(new grpc::Status(grpc::StatusCode.Unimplemented, ""));
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Periodic site health report (30 s cadence). The ack makes delivery
|
||||
/// observable end-to-end so the sender can restore its per-interval counters
|
||||
/// when a report is lost.
|
||||
/// </summary>
|
||||
/// <param name="request">The request received from the client.</param>
|
||||
/// <param name="context">The context of the server-side call handler being invoked.</param>
|
||||
/// <returns>The response to send back to the client (wrapped by a task).</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual global::System.Threading.Tasks.Task<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.SiteHealthReportAckDto> ReportSiteHealth(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.SiteHealthReportDto request, grpc::ServerCallContext context)
|
||||
{
|
||||
throw new grpc::RpcException(new grpc::Status(grpc::StatusCode.Unimplemented, ""));
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Application heartbeat. Returns Empty because the message is
|
||||
/// fire-and-forget: nothing on the site consumes a reply, and a failure here
|
||||
/// must never surface as a fault on the heartbeat timer path.
|
||||
/// </summary>
|
||||
/// <param name="request">The request received from the client.</param>
|
||||
/// <param name="context">The context of the server-side call handler being invoked.</param>
|
||||
/// <returns>The response to send back to the client (wrapped by a task).</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual global::System.Threading.Tasks.Task<global::Google.Protobuf.WellKnownTypes.Empty> Heartbeat(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.HeartbeatDto request, grpc::ServerCallContext context)
|
||||
{
|
||||
throw new grpc::RpcException(new grpc::Status(grpc::StatusCode.Unimplemented, ""));
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
/// <summary>Client for CentralControlService</summary>
|
||||
public partial class CentralControlServiceClient : grpc::ClientBase<CentralControlServiceClient>
|
||||
{
|
||||
/// <summary>Creates a new client for CentralControlService</summary>
|
||||
/// <param name="channel">The channel to use to make remote calls.</param>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public CentralControlServiceClient(grpc::ChannelBase channel) : base(channel)
|
||||
{
|
||||
}
|
||||
/// <summary>Creates a new client for CentralControlService that uses a custom <c>CallInvoker</c>.</summary>
|
||||
/// <param name="callInvoker">The callInvoker to use to make remote calls.</param>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public CentralControlServiceClient(grpc::CallInvoker callInvoker) : base(callInvoker)
|
||||
{
|
||||
}
|
||||
/// <summary>Protected parameterless constructor to allow creation of test doubles.</summary>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
protected CentralControlServiceClient() : base()
|
||||
{
|
||||
}
|
||||
/// <summary>Protected constructor to allow creation of configured clients.</summary>
|
||||
/// <param name="configuration">The client configuration.</param>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
protected CentralControlServiceClient(ClientBaseConfiguration configuration) : base(configuration)
|
||||
{
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Store-and-forward handoff of one notification for central delivery. The
|
||||
/// ack is idempotent on notification_id — a duplicate submit after a lost ack
|
||||
/// must not produce a second delivery.
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="headers">The initial metadata to send with the call. This parameter is optional.</param>
|
||||
/// <param name="deadline">An optional deadline for the call. The call will be cancelled if deadline is hit.</param>
|
||||
/// <param name="cancellationToken">An optional token for canceling the call.</param>
|
||||
/// <returns>The response received from the server.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationSubmitAckDto SubmitNotification(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationSubmitDto request, grpc::Metadata headers = null, global::System.DateTime? deadline = null, global::System.Threading.CancellationToken cancellationToken = default(global::System.Threading.CancellationToken))
|
||||
{
|
||||
return SubmitNotification(request, new grpc::CallOptions(headers, deadline, cancellationToken));
|
||||
}
|
||||
/// <summary>
|
||||
/// Store-and-forward handoff of one notification for central delivery. The
|
||||
/// ack is idempotent on notification_id — a duplicate submit after a lost ack
|
||||
/// must not produce a second delivery.
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="options">The options for the call.</param>
|
||||
/// <returns>The response received from the server.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationSubmitAckDto SubmitNotification(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationSubmitDto request, grpc::CallOptions options)
|
||||
{
|
||||
return CallInvoker.BlockingUnaryCall(__Method_SubmitNotification, null, options, request);
|
||||
}
|
||||
/// <summary>
|
||||
/// Store-and-forward handoff of one notification for central delivery. The
|
||||
/// ack is idempotent on notification_id — a duplicate submit after a lost ack
|
||||
/// must not produce a second delivery.
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="headers">The initial metadata to send with the call. This parameter is optional.</param>
|
||||
/// <param name="deadline">An optional deadline for the call. The call will be cancelled if deadline is hit.</param>
|
||||
/// <param name="cancellationToken">An optional token for canceling the call.</param>
|
||||
/// <returns>The call object.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual grpc::AsyncUnaryCall<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationSubmitAckDto> SubmitNotificationAsync(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationSubmitDto request, grpc::Metadata headers = null, global::System.DateTime? deadline = null, global::System.Threading.CancellationToken cancellationToken = default(global::System.Threading.CancellationToken))
|
||||
{
|
||||
return SubmitNotificationAsync(request, new grpc::CallOptions(headers, deadline, cancellationToken));
|
||||
}
|
||||
/// <summary>
|
||||
/// Store-and-forward handoff of one notification for central delivery. The
|
||||
/// ack is idempotent on notification_id — a duplicate submit after a lost ack
|
||||
/// must not produce a second delivery.
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="options">The options for the call.</param>
|
||||
/// <returns>The call object.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual grpc::AsyncUnaryCall<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationSubmitAckDto> SubmitNotificationAsync(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationSubmitDto request, grpc::CallOptions options)
|
||||
{
|
||||
return CallInvoker.AsyncUnaryCall(__Method_SubmitNotification, null, options, request);
|
||||
}
|
||||
/// <summary>
|
||||
/// Notify.Status(id) round-trip for a notification that has already left the
|
||||
/// site buffer. `found = false` sends the caller back to the site-local buffer
|
||||
/// to decide Forwarding vs Unknown.
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="headers">The initial metadata to send with the call. This parameter is optional.</param>
|
||||
/// <param name="deadline">An optional deadline for the call. The call will be cancelled if deadline is hit.</param>
|
||||
/// <param name="cancellationToken">An optional token for canceling the call.</param>
|
||||
/// <returns>The response received from the server.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationStatusResponseDto QueryNotificationStatus(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationStatusQueryDto request, grpc::Metadata headers = null, global::System.DateTime? deadline = null, global::System.Threading.CancellationToken cancellationToken = default(global::System.Threading.CancellationToken))
|
||||
{
|
||||
return QueryNotificationStatus(request, new grpc::CallOptions(headers, deadline, cancellationToken));
|
||||
}
|
||||
/// <summary>
|
||||
/// Notify.Status(id) round-trip for a notification that has already left the
|
||||
/// site buffer. `found = false` sends the caller back to the site-local buffer
|
||||
/// to decide Forwarding vs Unknown.
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="options">The options for the call.</param>
|
||||
/// <returns>The response received from the server.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationStatusResponseDto QueryNotificationStatus(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationStatusQueryDto request, grpc::CallOptions options)
|
||||
{
|
||||
return CallInvoker.BlockingUnaryCall(__Method_QueryNotificationStatus, null, options, request);
|
||||
}
|
||||
/// <summary>
|
||||
/// Notify.Status(id) round-trip for a notification that has already left the
|
||||
/// site buffer. `found = false` sends the caller back to the site-local buffer
|
||||
/// to decide Forwarding vs Unknown.
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="headers">The initial metadata to send with the call. This parameter is optional.</param>
|
||||
/// <param name="deadline">An optional deadline for the call. The call will be cancelled if deadline is hit.</param>
|
||||
/// <param name="cancellationToken">An optional token for canceling the call.</param>
|
||||
/// <returns>The call object.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual grpc::AsyncUnaryCall<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationStatusResponseDto> QueryNotificationStatusAsync(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationStatusQueryDto request, grpc::Metadata headers = null, global::System.DateTime? deadline = null, global::System.Threading.CancellationToken cancellationToken = default(global::System.Threading.CancellationToken))
|
||||
{
|
||||
return QueryNotificationStatusAsync(request, new grpc::CallOptions(headers, deadline, cancellationToken));
|
||||
}
|
||||
/// <summary>
|
||||
/// Notify.Status(id) round-trip for a notification that has already left the
|
||||
/// site buffer. `found = false` sends the caller back to the site-local buffer
|
||||
/// to decide Forwarding vs Unknown.
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="options">The options for the call.</param>
|
||||
/// <returns>The call object.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual grpc::AsyncUnaryCall<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationStatusResponseDto> QueryNotificationStatusAsync(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationStatusQueryDto request, grpc::CallOptions options)
|
||||
{
|
||||
return CallInvoker.AsyncUnaryCall(__Method_QueryNotificationStatus, null, options, request);
|
||||
}
|
||||
/// <summary>
|
||||
/// Audit Log (#23) push telemetry. Reuses the SiteStreamService messages: the
|
||||
/// batch a site drains from its SQLite hot path is byte-identical whichever
|
||||
/// transport carries it.
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="headers">The initial metadata to send with the call. This parameter is optional.</param>
|
||||
/// <param name="deadline">An optional deadline for the call. The call will be cancelled if deadline is hit.</param>
|
||||
/// <param name="cancellationToken">An optional token for canceling the call.</param>
|
||||
/// <returns>The response received from the server.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.IngestAck IngestAuditEvents(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.AuditEventBatch request, grpc::Metadata headers = null, global::System.DateTime? deadline = null, global::System.Threading.CancellationToken cancellationToken = default(global::System.Threading.CancellationToken))
|
||||
{
|
||||
return IngestAuditEvents(request, new grpc::CallOptions(headers, deadline, cancellationToken));
|
||||
}
|
||||
/// <summary>
|
||||
/// Audit Log (#23) push telemetry. Reuses the SiteStreamService messages: the
|
||||
/// batch a site drains from its SQLite hot path is byte-identical whichever
|
||||
/// transport carries it.
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="options">The options for the call.</param>
|
||||
/// <returns>The response received from the server.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.IngestAck IngestAuditEvents(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.AuditEventBatch request, grpc::CallOptions options)
|
||||
{
|
||||
return CallInvoker.BlockingUnaryCall(__Method_IngestAuditEvents, null, options, request);
|
||||
}
|
||||
/// <summary>
|
||||
/// Audit Log (#23) push telemetry. Reuses the SiteStreamService messages: the
|
||||
/// batch a site drains from its SQLite hot path is byte-identical whichever
|
||||
/// transport carries it.
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="headers">The initial metadata to send with the call. This parameter is optional.</param>
|
||||
/// <param name="deadline">An optional deadline for the call. The call will be cancelled if deadline is hit.</param>
|
||||
/// <param name="cancellationToken">An optional token for canceling the call.</param>
|
||||
/// <returns>The call object.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual grpc::AsyncUnaryCall<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.IngestAck> IngestAuditEventsAsync(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.AuditEventBatch request, grpc::Metadata headers = null, global::System.DateTime? deadline = null, global::System.Threading.CancellationToken cancellationToken = default(global::System.Threading.CancellationToken))
|
||||
{
|
||||
return IngestAuditEventsAsync(request, new grpc::CallOptions(headers, deadline, cancellationToken));
|
||||
}
|
||||
/// <summary>
|
||||
/// Audit Log (#23) push telemetry. Reuses the SiteStreamService messages: the
|
||||
/// batch a site drains from its SQLite hot path is byte-identical whichever
|
||||
/// transport carries it.
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="options">The options for the call.</param>
|
||||
/// <returns>The call object.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual grpc::AsyncUnaryCall<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.IngestAck> IngestAuditEventsAsync(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.AuditEventBatch request, grpc::CallOptions options)
|
||||
{
|
||||
return CallInvoker.AsyncUnaryCall(__Method_IngestAuditEvents, null, options, request);
|
||||
}
|
||||
/// <summary>
|
||||
/// Audit Log (#23) M3 combined cached-call telemetry (audit row + SiteCalls
|
||||
/// operational upsert, written in one central transaction).
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="headers">The initial metadata to send with the call. This parameter is optional.</param>
|
||||
/// <param name="deadline">An optional deadline for the call. The call will be cancelled if deadline is hit.</param>
|
||||
/// <param name="cancellationToken">An optional token for canceling the call.</param>
|
||||
/// <returns>The response received from the server.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.IngestAck IngestCachedTelemetry(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.CachedTelemetryBatch request, grpc::Metadata headers = null, global::System.DateTime? deadline = null, global::System.Threading.CancellationToken cancellationToken = default(global::System.Threading.CancellationToken))
|
||||
{
|
||||
return IngestCachedTelemetry(request, new grpc::CallOptions(headers, deadline, cancellationToken));
|
||||
}
|
||||
/// <summary>
|
||||
/// Audit Log (#23) M3 combined cached-call telemetry (audit row + SiteCalls
|
||||
/// operational upsert, written in one central transaction).
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="options">The options for the call.</param>
|
||||
/// <returns>The response received from the server.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.IngestAck IngestCachedTelemetry(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.CachedTelemetryBatch request, grpc::CallOptions options)
|
||||
{
|
||||
return CallInvoker.BlockingUnaryCall(__Method_IngestCachedTelemetry, null, options, request);
|
||||
}
|
||||
/// <summary>
|
||||
/// Audit Log (#23) M3 combined cached-call telemetry (audit row + SiteCalls
|
||||
/// operational upsert, written in one central transaction).
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="headers">The initial metadata to send with the call. This parameter is optional.</param>
|
||||
/// <param name="deadline">An optional deadline for the call. The call will be cancelled if deadline is hit.</param>
|
||||
/// <param name="cancellationToken">An optional token for canceling the call.</param>
|
||||
/// <returns>The call object.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual grpc::AsyncUnaryCall<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.IngestAck> IngestCachedTelemetryAsync(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.CachedTelemetryBatch request, grpc::Metadata headers = null, global::System.DateTime? deadline = null, global::System.Threading.CancellationToken cancellationToken = default(global::System.Threading.CancellationToken))
|
||||
{
|
||||
return IngestCachedTelemetryAsync(request, new grpc::CallOptions(headers, deadline, cancellationToken));
|
||||
}
|
||||
/// <summary>
|
||||
/// Audit Log (#23) M3 combined cached-call telemetry (audit row + SiteCalls
|
||||
/// operational upsert, written in one central transaction).
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="options">The options for the call.</param>
|
||||
/// <returns>The call object.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual grpc::AsyncUnaryCall<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.IngestAck> IngestCachedTelemetryAsync(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.CachedTelemetryBatch request, grpc::CallOptions options)
|
||||
{
|
||||
return CallInvoker.AsyncUnaryCall(__Method_IngestCachedTelemetry, null, options, request);
|
||||
}
|
||||
/// <summary>
|
||||
/// Node-startup self-heal: the node's local deployed inventory in, fetch
|
||||
/// tokens for whatever it is missing or stale out.
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="headers">The initial metadata to send with the call. This parameter is optional.</param>
|
||||
/// <param name="deadline">An optional deadline for the call. The call will be cancelled if deadline is hit.</param>
|
||||
/// <param name="cancellationToken">An optional token for canceling the call.</param>
|
||||
/// <returns>The response received from the server.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.ReconcileSiteResponseDto ReconcileSite(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.ReconcileSiteRequestDto request, grpc::Metadata headers = null, global::System.DateTime? deadline = null, global::System.Threading.CancellationToken cancellationToken = default(global::System.Threading.CancellationToken))
|
||||
{
|
||||
return ReconcileSite(request, new grpc::CallOptions(headers, deadline, cancellationToken));
|
||||
}
|
||||
/// <summary>
|
||||
/// Node-startup self-heal: the node's local deployed inventory in, fetch
|
||||
/// tokens for whatever it is missing or stale out.
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="options">The options for the call.</param>
|
||||
/// <returns>The response received from the server.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.ReconcileSiteResponseDto ReconcileSite(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.ReconcileSiteRequestDto request, grpc::CallOptions options)
|
||||
{
|
||||
return CallInvoker.BlockingUnaryCall(__Method_ReconcileSite, null, options, request);
|
||||
}
|
||||
/// <summary>
|
||||
/// Node-startup self-heal: the node's local deployed inventory in, fetch
|
||||
/// tokens for whatever it is missing or stale out.
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="headers">The initial metadata to send with the call. This parameter is optional.</param>
|
||||
/// <param name="deadline">An optional deadline for the call. The call will be cancelled if deadline is hit.</param>
|
||||
/// <param name="cancellationToken">An optional token for canceling the call.</param>
|
||||
/// <returns>The call object.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual grpc::AsyncUnaryCall<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.ReconcileSiteResponseDto> ReconcileSiteAsync(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.ReconcileSiteRequestDto request, grpc::Metadata headers = null, global::System.DateTime? deadline = null, global::System.Threading.CancellationToken cancellationToken = default(global::System.Threading.CancellationToken))
|
||||
{
|
||||
return ReconcileSiteAsync(request, new grpc::CallOptions(headers, deadline, cancellationToken));
|
||||
}
|
||||
/// <summary>
|
||||
/// Node-startup self-heal: the node's local deployed inventory in, fetch
|
||||
/// tokens for whatever it is missing or stale out.
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="options">The options for the call.</param>
|
||||
/// <returns>The call object.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual grpc::AsyncUnaryCall<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.ReconcileSiteResponseDto> ReconcileSiteAsync(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.ReconcileSiteRequestDto request, grpc::CallOptions options)
|
||||
{
|
||||
return CallInvoker.AsyncUnaryCall(__Method_ReconcileSite, null, options, request);
|
||||
}
|
||||
/// <summary>
|
||||
/// Periodic site health report (30 s cadence). The ack makes delivery
|
||||
/// observable end-to-end so the sender can restore its per-interval counters
|
||||
/// when a report is lost.
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="headers">The initial metadata to send with the call. This parameter is optional.</param>
|
||||
/// <param name="deadline">An optional deadline for the call. The call will be cancelled if deadline is hit.</param>
|
||||
/// <param name="cancellationToken">An optional token for canceling the call.</param>
|
||||
/// <returns>The response received from the server.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.SiteHealthReportAckDto ReportSiteHealth(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.SiteHealthReportDto request, grpc::Metadata headers = null, global::System.DateTime? deadline = null, global::System.Threading.CancellationToken cancellationToken = default(global::System.Threading.CancellationToken))
|
||||
{
|
||||
return ReportSiteHealth(request, new grpc::CallOptions(headers, deadline, cancellationToken));
|
||||
}
|
||||
/// <summary>
|
||||
/// Periodic site health report (30 s cadence). The ack makes delivery
|
||||
/// observable end-to-end so the sender can restore its per-interval counters
|
||||
/// when a report is lost.
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="options">The options for the call.</param>
|
||||
/// <returns>The response received from the server.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.SiteHealthReportAckDto ReportSiteHealth(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.SiteHealthReportDto request, grpc::CallOptions options)
|
||||
{
|
||||
return CallInvoker.BlockingUnaryCall(__Method_ReportSiteHealth, null, options, request);
|
||||
}
|
||||
/// <summary>
|
||||
/// Periodic site health report (30 s cadence). The ack makes delivery
|
||||
/// observable end-to-end so the sender can restore its per-interval counters
|
||||
/// when a report is lost.
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="headers">The initial metadata to send with the call. This parameter is optional.</param>
|
||||
/// <param name="deadline">An optional deadline for the call. The call will be cancelled if deadline is hit.</param>
|
||||
/// <param name="cancellationToken">An optional token for canceling the call.</param>
|
||||
/// <returns>The call object.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual grpc::AsyncUnaryCall<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.SiteHealthReportAckDto> ReportSiteHealthAsync(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.SiteHealthReportDto request, grpc::Metadata headers = null, global::System.DateTime? deadline = null, global::System.Threading.CancellationToken cancellationToken = default(global::System.Threading.CancellationToken))
|
||||
{
|
||||
return ReportSiteHealthAsync(request, new grpc::CallOptions(headers, deadline, cancellationToken));
|
||||
}
|
||||
/// <summary>
|
||||
/// Periodic site health report (30 s cadence). The ack makes delivery
|
||||
/// observable end-to-end so the sender can restore its per-interval counters
|
||||
/// when a report is lost.
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="options">The options for the call.</param>
|
||||
/// <returns>The call object.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual grpc::AsyncUnaryCall<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.SiteHealthReportAckDto> ReportSiteHealthAsync(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.SiteHealthReportDto request, grpc::CallOptions options)
|
||||
{
|
||||
return CallInvoker.AsyncUnaryCall(__Method_ReportSiteHealth, null, options, request);
|
||||
}
|
||||
/// <summary>
|
||||
/// Application heartbeat. Returns Empty because the message is
|
||||
/// fire-and-forget: nothing on the site consumes a reply, and a failure here
|
||||
/// must never surface as a fault on the heartbeat timer path.
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="headers">The initial metadata to send with the call. This parameter is optional.</param>
|
||||
/// <param name="deadline">An optional deadline for the call. The call will be cancelled if deadline is hit.</param>
|
||||
/// <param name="cancellationToken">An optional token for canceling the call.</param>
|
||||
/// <returns>The response received from the server.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual global::Google.Protobuf.WellKnownTypes.Empty Heartbeat(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.HeartbeatDto request, grpc::Metadata headers = null, global::System.DateTime? deadline = null, global::System.Threading.CancellationToken cancellationToken = default(global::System.Threading.CancellationToken))
|
||||
{
|
||||
return Heartbeat(request, new grpc::CallOptions(headers, deadline, cancellationToken));
|
||||
}
|
||||
/// <summary>
|
||||
/// Application heartbeat. Returns Empty because the message is
|
||||
/// fire-and-forget: nothing on the site consumes a reply, and a failure here
|
||||
/// must never surface as a fault on the heartbeat timer path.
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="options">The options for the call.</param>
|
||||
/// <returns>The response received from the server.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual global::Google.Protobuf.WellKnownTypes.Empty Heartbeat(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.HeartbeatDto request, grpc::CallOptions options)
|
||||
{
|
||||
return CallInvoker.BlockingUnaryCall(__Method_Heartbeat, null, options, request);
|
||||
}
|
||||
/// <summary>
|
||||
/// Application heartbeat. Returns Empty because the message is
|
||||
/// fire-and-forget: nothing on the site consumes a reply, and a failure here
|
||||
/// must never surface as a fault on the heartbeat timer path.
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="headers">The initial metadata to send with the call. This parameter is optional.</param>
|
||||
/// <param name="deadline">An optional deadline for the call. The call will be cancelled if deadline is hit.</param>
|
||||
/// <param name="cancellationToken">An optional token for canceling the call.</param>
|
||||
/// <returns>The call object.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual grpc::AsyncUnaryCall<global::Google.Protobuf.WellKnownTypes.Empty> HeartbeatAsync(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.HeartbeatDto request, grpc::Metadata headers = null, global::System.DateTime? deadline = null, global::System.Threading.CancellationToken cancellationToken = default(global::System.Threading.CancellationToken))
|
||||
{
|
||||
return HeartbeatAsync(request, new grpc::CallOptions(headers, deadline, cancellationToken));
|
||||
}
|
||||
/// <summary>
|
||||
/// Application heartbeat. Returns Empty because the message is
|
||||
/// fire-and-forget: nothing on the site consumes a reply, and a failure here
|
||||
/// must never surface as a fault on the heartbeat timer path.
|
||||
/// </summary>
|
||||
/// <param name="request">The request to send to the server.</param>
|
||||
/// <param name="options">The options for the call.</param>
|
||||
/// <returns>The call object.</returns>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public virtual grpc::AsyncUnaryCall<global::Google.Protobuf.WellKnownTypes.Empty> HeartbeatAsync(global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.HeartbeatDto request, grpc::CallOptions options)
|
||||
{
|
||||
return CallInvoker.AsyncUnaryCall(__Method_Heartbeat, null, options, request);
|
||||
}
|
||||
/// <summary>Creates a new instance of client from given <c>ClientBaseConfiguration</c>.</summary>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
protected override CentralControlServiceClient NewInstance(ClientBaseConfiguration configuration)
|
||||
{
|
||||
return new CentralControlServiceClient(configuration);
|
||||
}
|
||||
}
|
||||
|
||||
/// <summary>Creates service definition that can be registered with a server</summary>
|
||||
/// <param name="serviceImpl">An object implementing the server-side handling logic.</param>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public static grpc::ServerServiceDefinition BindService(CentralControlServiceBase serviceImpl)
|
||||
{
|
||||
return grpc::ServerServiceDefinition.CreateBuilder()
|
||||
.AddMethod(__Method_SubmitNotification, serviceImpl.SubmitNotification)
|
||||
.AddMethod(__Method_QueryNotificationStatus, serviceImpl.QueryNotificationStatus)
|
||||
.AddMethod(__Method_IngestAuditEvents, serviceImpl.IngestAuditEvents)
|
||||
.AddMethod(__Method_IngestCachedTelemetry, serviceImpl.IngestCachedTelemetry)
|
||||
.AddMethod(__Method_ReconcileSite, serviceImpl.ReconcileSite)
|
||||
.AddMethod(__Method_ReportSiteHealth, serviceImpl.ReportSiteHealth)
|
||||
.AddMethod(__Method_Heartbeat, serviceImpl.Heartbeat).Build();
|
||||
}
|
||||
|
||||
/// <summary>Register service method with a service binder with or without implementation. Useful when customizing the service binding logic.
|
||||
/// Note: this method is part of an experimental API that can change or be removed without any prior notice.</summary>
|
||||
/// <param name="serviceBinder">Service methods will be bound by calling <c>AddMethod</c> on this object.</param>
|
||||
/// <param name="serviceImpl">An object implementing the server-side handling logic.</param>
|
||||
[global::System.CodeDom.Compiler.GeneratedCode("grpc_csharp_plugin", null)]
|
||||
public static void BindService(grpc::ServiceBinderBase serviceBinder, CentralControlServiceBase serviceImpl)
|
||||
{
|
||||
serviceBinder.AddMethod(__Method_SubmitNotification, serviceImpl == null ? null : new grpc::UnaryServerMethod<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationSubmitDto, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationSubmitAckDto>(serviceImpl.SubmitNotification));
|
||||
serviceBinder.AddMethod(__Method_QueryNotificationStatus, serviceImpl == null ? null : new grpc::UnaryServerMethod<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationStatusQueryDto, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.NotificationStatusResponseDto>(serviceImpl.QueryNotificationStatus));
|
||||
serviceBinder.AddMethod(__Method_IngestAuditEvents, serviceImpl == null ? null : new grpc::UnaryServerMethod<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.AuditEventBatch, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.IngestAck>(serviceImpl.IngestAuditEvents));
|
||||
serviceBinder.AddMethod(__Method_IngestCachedTelemetry, serviceImpl == null ? null : new grpc::UnaryServerMethod<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.CachedTelemetryBatch, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.IngestAck>(serviceImpl.IngestCachedTelemetry));
|
||||
serviceBinder.AddMethod(__Method_ReconcileSite, serviceImpl == null ? null : new grpc::UnaryServerMethod<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.ReconcileSiteRequestDto, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.ReconcileSiteResponseDto>(serviceImpl.ReconcileSite));
|
||||
serviceBinder.AddMethod(__Method_ReportSiteHealth, serviceImpl == null ? null : new grpc::UnaryServerMethod<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.SiteHealthReportDto, global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.SiteHealthReportAckDto>(serviceImpl.ReportSiteHealth));
|
||||
serviceBinder.AddMethod(__Method_Heartbeat, serviceImpl == null ? null : new grpc::UnaryServerMethod<global::ZB.MOM.WW.ScadaBridge.Communication.Grpc.HeartbeatDto, global::Google.Protobuf.WellKnownTypes.Empty>(serviceImpl.Heartbeat));
|
||||
}
|
||||
|
||||
}
|
||||
}
|
||||
#endregion
|
||||
@@ -1,11 +1,44 @@
|
||||
namespace ZB.MOM.WW.ScadaBridge.Communication;
|
||||
|
||||
/// <summary>
|
||||
/// Which transport carries the seven site→central control messages. Selected per node by
|
||||
/// <c>ScadaBridge:Communication:CentralTransport</c>; the migration ships with
|
||||
/// <see cref="Akka"/> as the default so nothing flips until a node opts in.
|
||||
/// </summary>
|
||||
public enum CentralTransportMode
|
||||
{
|
||||
/// <summary>Akka <c>ClusterClient</c> — the transport in production today, and the default.</summary>
|
||||
Akka = 0,
|
||||
|
||||
/// <summary>gRPC dial of the central <c>CentralControlService</c> (Phase 1A migration target).</summary>
|
||||
Grpc = 1,
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Configuration options for central-site communication, including per-pattern
|
||||
/// timeouts and transport heartbeat settings.
|
||||
/// </summary>
|
||||
public class CommunicationOptions
|
||||
{
|
||||
/// <summary>
|
||||
/// Which transport carries the site→central control messages. Default <see cref="CentralTransportMode.Akka"/>
|
||||
/// (ClusterClient) — coexistence rule: a node flips to gRPC only by setting this to <c>Grpc</c>,
|
||||
/// and rollback is flipping it back. Selecting <c>Grpc</c> requires <see cref="CentralGrpcEndpoints"/>.
|
||||
/// </summary>
|
||||
public CentralTransportMode CentralTransport { get; set; } = CentralTransportMode.Akka;
|
||||
|
||||
/// <summary>
|
||||
/// Central control-plane gRPC endpoints (preferred first), e.g.
|
||||
/// <c>["http://scadabridge-central-a:8083", "http://scadabridge-central-b:8083"]</c>. Dialled by
|
||||
/// <see cref="Grpc.CentralChannelProvider"/> with sticky failover/failback. Required when
|
||||
/// <see cref="CentralTransport"/> is <see cref="CentralTransportMode.Grpc"/>, ignored otherwise.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Sites reach central by container/host name, NOT via Traefik (which is HTTP/1 only; gRPC is
|
||||
/// h2c on the central node's dedicated <c>CentralGrpcPort</c>).
|
||||
/// </remarks>
|
||||
public List<string> CentralGrpcEndpoints { get; set; } = new();
|
||||
|
||||
/// <summary>Timeout for deployment commands (typically longest due to apply logic).</summary>
|
||||
public TimeSpan DeploymentTimeout { get; set; } = TimeSpan.FromMinutes(2);
|
||||
|
||||
|
||||
@@ -66,6 +66,19 @@ public sealed class CommunicationOptionsValidator : OptionsValidatorBase<Communi
|
||||
builder.RequireThat(options.GrpcMaxConcurrentStreams > 0,
|
||||
$"ScadaBridge:Communication:GrpcMaxConcurrentStreams must be positive (was {options.GrpcMaxConcurrentStreams}).");
|
||||
|
||||
// The gRPC site→central transport needs at least one central endpoint to dial. Only
|
||||
// enforced when that transport is selected — the default Akka path ignores the list, so a
|
||||
// node on ClusterClient must not be forced to declare gRPC endpoints it never uses.
|
||||
if (options.CentralTransport == CentralTransportMode.Grpc)
|
||||
{
|
||||
builder.RequireThat(
|
||||
options.CentralGrpcEndpoints.Count > 0
|
||||
&& options.CentralGrpcEndpoints.All(e => !string.IsNullOrWhiteSpace(e)),
|
||||
"ScadaBridge:Communication:CentralGrpcEndpoints must list at least one non-empty "
|
||||
+ "central gRPC endpoint when CentralTransport is Grpc "
|
||||
+ $"(was {options.CentralGrpcEndpoints.Count} entr{(options.CentralGrpcEndpoints.Count == 1 ? "y" : "ies")}).");
|
||||
}
|
||||
|
||||
// ── Aggregated live alarm cache (plan #10, Task 6) ───────────────────────
|
||||
// Linger drives a Timer dueTime; TimeSpan.Zero is valid (stop the aggregator
|
||||
// immediately when the last viewer leaves), only a negative value is invalid.
|
||||
|
||||
@@ -0,0 +1,274 @@
|
||||
using Google.Protobuf.WellKnownTypes;
|
||||
using Grpc.Core;
|
||||
using Grpc.Net.Client;
|
||||
using Microsoft.Extensions.Logging;
|
||||
|
||||
namespace ZB.MOM.WW.ScadaBridge.Communication.Grpc;
|
||||
|
||||
/// <summary>
|
||||
/// A pair (or more) of gRPC channels to the central cluster's control-plane nodes, with the
|
||||
/// sticky-failover + background-failback policy of the design (§3.5). The site holds one channel
|
||||
/// per central endpoint, prefers the first, and stays on it until a call proves it unreachable —
|
||||
/// only then flipping to the next, and only <see cref="StatusCode.Unavailable"/> / connect
|
||||
/// failures count (a <see cref="StatusCode.DeadlineExceeded"/> never flips or retries, because the
|
||||
/// call may have run).
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// <para>
|
||||
/// <b>Sticky.</b> All calls go to the current channel; a healthy preferred endpoint never
|
||||
/// ping-pongs. <see cref="ReportUnavailable"/> flips to the next endpoint (round-robin) when the
|
||||
/// caller sees the current one refuse a connection.
|
||||
/// </para>
|
||||
/// <para>
|
||||
/// <b>Failback.</b> While off the preferred endpoint a background probe (a cheap <c>Heartbeat</c>
|
||||
/// ping — always answered, even by a not-yet-ready node) re-checks the preferred one. On the first
|
||||
/// success new calls return to it; in-flight calls finish where they are. The probe is
|
||||
/// event-driven: it arms on a flip and stops the moment we are back on the preferred endpoint, so
|
||||
/// a steady healthy pair spends no cycles. Its cadence backs off exponentially — 1 s, doubling,
|
||||
/// capped at 60 s — while the preferred endpoint stays down, so a genuinely dead node is not
|
||||
/// probed hard.
|
||||
/// </para>
|
||||
/// <para>
|
||||
/// <b>Auth.</b> Every channel carries the site's own preshared key and its
|
||||
/// <c>x-scadabridge-site</c> identity via <see cref="ControlPlaneCredentials"/> — the same
|
||||
/// insecure-h2c call-credentials shape the streaming client uses. The <paramref name="handlerFactory"/>
|
||||
/// seam lets a test point a channel at an in-process <c>TestServer</c>; production uses a
|
||||
/// keepalive-configured <see cref="SocketsHttpHandler"/>.
|
||||
/// </para>
|
||||
/// </remarks>
|
||||
public sealed class CentralChannelProvider : IDisposable
|
||||
{
|
||||
private static readonly TimeSpan DefaultBackoffBase = TimeSpan.FromSeconds(1);
|
||||
private static readonly TimeSpan DefaultBackoffCap = TimeSpan.FromSeconds(60);
|
||||
private static readonly TimeSpan DefaultProbeDeadline = TimeSpan.FromSeconds(5);
|
||||
|
||||
private readonly IReadOnlyList<string> _endpoints;
|
||||
private readonly GrpcChannel[] _channels;
|
||||
private readonly CentralControlService.CentralControlServiceClient[] _clients;
|
||||
private readonly ILogger _logger;
|
||||
private readonly string _siteId;
|
||||
private readonly TimeSpan _backoffBase;
|
||||
private readonly TimeSpan _backoffCap;
|
||||
private readonly TimeSpan _probeDeadline;
|
||||
private readonly Timer? _failbackTimer;
|
||||
private readonly object _gate = new();
|
||||
|
||||
private volatile int _current; // preferred == 0
|
||||
private int _consecutiveProbeFailures;
|
||||
private bool _disposed;
|
||||
|
||||
/// <summary>Creates the provider and opens one channel per endpoint.</summary>
|
||||
/// <param name="endpoints">Central control-plane endpoints, preferred first (index 0). Must be non-empty.</param>
|
||||
/// <param name="pskProvider">Resolves this site's preshared key (site-side: a single-key provider).</param>
|
||||
/// <param name="siteId">This site's identity, sent as the <c>x-scadabridge-site</c> header.</param>
|
||||
/// <param name="options">Communication options supplying gRPC keepalive settings.</param>
|
||||
/// <param name="logger">Logger for flip/failback diagnostics.</param>
|
||||
/// <param name="handlerFactory">Test seam: per-endpoint <see cref="HttpMessageHandler"/>; null uses a production socket handler.</param>
|
||||
/// <param name="probeDeadline">Deadline for a failback probe. Null uses 5 s.</param>
|
||||
/// <param name="backoffBase">Initial failback-probe backoff. Null uses 1 s.</param>
|
||||
/// <param name="backoffCap">Maximum failback-probe backoff. Null uses 60 s.</param>
|
||||
public CentralChannelProvider(
|
||||
IReadOnlyList<string> endpoints,
|
||||
ISitePskProvider pskProvider,
|
||||
string siteId,
|
||||
CommunicationOptions options,
|
||||
ILogger logger,
|
||||
Func<string, HttpMessageHandler>? handlerFactory = null,
|
||||
TimeSpan? probeDeadline = null,
|
||||
TimeSpan? backoffBase = null,
|
||||
TimeSpan? backoffCap = null)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(endpoints);
|
||||
ArgumentNullException.ThrowIfNull(pskProvider);
|
||||
ArgumentException.ThrowIfNullOrWhiteSpace(siteId);
|
||||
ArgumentNullException.ThrowIfNull(options);
|
||||
ArgumentNullException.ThrowIfNull(logger);
|
||||
if (endpoints.Count == 0)
|
||||
{
|
||||
throw new ArgumentException("At least one central gRPC endpoint is required.", nameof(endpoints));
|
||||
}
|
||||
|
||||
_endpoints = endpoints;
|
||||
_logger = logger;
|
||||
_siteId = siteId;
|
||||
_backoffBase = backoffBase ?? DefaultBackoffBase;
|
||||
_backoffCap = backoffCap ?? DefaultBackoffCap;
|
||||
_probeDeadline = probeDeadline ?? DefaultProbeDeadline;
|
||||
|
||||
_channels = new GrpcChannel[endpoints.Count];
|
||||
_clients = new CentralControlService.CentralControlServiceClient[endpoints.Count];
|
||||
for (var i = 0; i < endpoints.Count; i++)
|
||||
{
|
||||
var channelOptions = new GrpcChannelOptions
|
||||
{
|
||||
HttpHandler = handlerFactory?.Invoke(endpoints[i]) ?? new SocketsHttpHandler
|
||||
{
|
||||
KeepAlivePingDelay = options.GrpcKeepAlivePingDelay,
|
||||
KeepAlivePingTimeout = options.GrpcKeepAlivePingTimeout,
|
||||
KeepAlivePingPolicy = HttpKeepAlivePingPolicy.Always,
|
||||
EnableMultipleHttp2Connections = true,
|
||||
},
|
||||
}.WithSiteCredentials(pskProvider, siteId);
|
||||
|
||||
_channels[i] = GrpcChannel.ForAddress(endpoints[i], channelOptions);
|
||||
_clients[i] = new CentralControlService.CentralControlServiceClient(_channels[i]);
|
||||
}
|
||||
|
||||
// Only a multi-endpoint pair can ever fail over, so a lone endpoint needs no probe.
|
||||
if (endpoints.Count > 1)
|
||||
{
|
||||
_failbackTimer = new Timer(_ => _ = FailbackTickAsync(), null, Timeout.Infinite, Timeout.Infinite);
|
||||
}
|
||||
}
|
||||
|
||||
/// <summary>The number of endpoints in the pair.</summary>
|
||||
public int EndpointCount => _endpoints.Count;
|
||||
|
||||
/// <summary>The index of the endpoint calls are currently routed to (preferred == 0).</summary>
|
||||
public int CurrentIndex => _current;
|
||||
|
||||
/// <summary>The endpoint address calls are currently routed to.</summary>
|
||||
public string CurrentEndpoint => _endpoints[_current];
|
||||
|
||||
/// <summary>
|
||||
/// The endpoint index and client calls should use right now. Captured together so a caller can
|
||||
/// tell <see cref="ReportUnavailable"/> exactly which endpoint failed even if a concurrent flip
|
||||
/// has already moved <see cref="CurrentIndex"/>.
|
||||
/// </summary>
|
||||
/// <returns>The current endpoint index and its client.</returns>
|
||||
public (int Index, CentralControlService.CentralControlServiceClient Client) Current()
|
||||
{
|
||||
var idx = _current;
|
||||
return (idx, _clients[idx]);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Reports that the endpoint at <paramref name="failedIndex"/> refused a connection (an
|
||||
/// <see cref="StatusCode.Unavailable"/> / connect failure). If it is still the current endpoint
|
||||
/// and another exists, flips to the next one and — when now off the preferred endpoint — arms
|
||||
/// the failback probe. Idempotent under a concurrent flip: a stale index is ignored.
|
||||
/// </summary>
|
||||
/// <param name="failedIndex">The endpoint index the caller's failed call used.</param>
|
||||
public void ReportUnavailable(int failedIndex)
|
||||
{
|
||||
if (_endpoints.Count < 2)
|
||||
{
|
||||
return; // nothing to fail over to
|
||||
}
|
||||
|
||||
lock (_gate)
|
||||
{
|
||||
if (_disposed || failedIndex != _current)
|
||||
{
|
||||
return; // a concurrent flip already moved us; do not double-flip
|
||||
}
|
||||
|
||||
var next = (failedIndex + 1) % _endpoints.Count;
|
||||
_current = next;
|
||||
_logger.LogWarning(
|
||||
"Central control-plane endpoint {Failed} is unavailable; site {SiteId} failed over to {Next}.",
|
||||
_endpoints[failedIndex], _siteId, _endpoints[next]);
|
||||
|
||||
if (_current != 0)
|
||||
{
|
||||
_consecutiveProbeFailures = 0;
|
||||
ArmFailback(_backoffBase);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private void ArmFailback(TimeSpan due)
|
||||
{
|
||||
if (_disposed)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
_failbackTimer?.Change(due, Timeout.InfiniteTimeSpan);
|
||||
}
|
||||
|
||||
private async Task FailbackTickAsync()
|
||||
{
|
||||
int currentAtTick = _current;
|
||||
if (_disposed || currentAtTick == 0)
|
||||
{
|
||||
return; // already back on the preferred endpoint (or shutting down)
|
||||
}
|
||||
|
||||
var preferred = _clients[0];
|
||||
try
|
||||
{
|
||||
await preferred.HeartbeatAsync(
|
||||
new HeartbeatDto
|
||||
{
|
||||
SiteId = _siteId,
|
||||
NodeHostname = "failback-probe",
|
||||
IsActive = false,
|
||||
Timestamp = Timestamp.FromDateTimeOffset(DateTimeOffset.UtcNow),
|
||||
},
|
||||
deadline: DateTime.UtcNow.Add(_probeDeadline)).ConfigureAwait(false);
|
||||
|
||||
// The preferred endpoint answered — return new calls to it.
|
||||
lock (_gate)
|
||||
{
|
||||
if (_disposed)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
_current = 0;
|
||||
_consecutiveProbeFailures = 0;
|
||||
}
|
||||
|
||||
_logger.LogInformation(
|
||||
"Central control-plane preferred endpoint {Preferred} is reachable again; site {SiteId} failed back.",
|
||||
_endpoints[0], _siteId);
|
||||
}
|
||||
catch (Exception ex)
|
||||
{
|
||||
lock (_gate)
|
||||
{
|
||||
if (_disposed || _current == 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
_consecutiveProbeFailures++;
|
||||
var backoff = NextBackoff(_consecutiveProbeFailures);
|
||||
_logger.LogDebug(ex,
|
||||
"Failback probe of preferred central endpoint {Preferred} failed; re-probing in {Backoff}.",
|
||||
_endpoints[0], backoff);
|
||||
ArmFailback(backoff);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private TimeSpan NextBackoff(int failures)
|
||||
{
|
||||
// 1 s, doubling, capped at 60 s. Guard the shift against overflow for a long outage.
|
||||
var exponent = Math.Min(failures - 1, 20);
|
||||
var scaled = _backoffBase.Ticks * (1L << exponent);
|
||||
var cap = _backoffCap.Ticks;
|
||||
return TimeSpan.FromTicks(scaled >= cap || scaled < 0 ? cap : scaled);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public void Dispose()
|
||||
{
|
||||
lock (_gate)
|
||||
{
|
||||
if (_disposed)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
_disposed = true;
|
||||
}
|
||||
|
||||
_failbackTimer?.Dispose();
|
||||
foreach (var channel in _channels)
|
||||
{
|
||||
channel.Dispose();
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,764 @@
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Audit;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Deployment;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Health;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Notification;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Types;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Types.Enums;
|
||||
using Timestamp = Google.Protobuf.WellKnownTypes.Timestamp;
|
||||
|
||||
namespace ZB.MOM.WW.ScadaBridge.Communication.Grpc;
|
||||
|
||||
/// <summary>
|
||||
/// Canonical bridge between the seven in-process messages the site sends to central
|
||||
/// and the wire format of <c>CentralControlService</c> (<c>Protos/central_control.proto</c>).
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// <para>
|
||||
/// The seven pairs mirror, one for one, the seven messages
|
||||
/// <c>SiteCommunicationActor</c> forwards to <c>/user/central-communication</c> over
|
||||
/// Akka <c>ClusterClient</c> today. Both transports carry the SAME message types
|
||||
/// end-to-end — central's handlers are untouched by the migration — so this mapper is
|
||||
/// the only place the two representations meet, and a field that does not survive a
|
||||
/// round-trip here is a field the gRPC transport silently drops.
|
||||
/// </para>
|
||||
/// <para>
|
||||
/// <b>Ingest is deliberately absent from the message list.</b> <c>IngestAuditEvents</c>
|
||||
/// and <c>IngestCachedTelemetry</c> reuse the <c>AuditEventBatch</c> /
|
||||
/// <c>CachedTelemetryBatch</c> / <c>IngestAck</c> messages already defined for the
|
||||
/// site-hosted <c>SiteStreamService</c>, so the per-row work is delegated to the
|
||||
/// existing <see cref="AuditEventDtoMapper"/> and <see cref="SiteCallDtoMapper"/>; only
|
||||
/// the batch/ack envelopes are assembled here.
|
||||
/// </para>
|
||||
///
|
||||
/// <para><b>Conventions, applied uniformly across every method below.</b></para>
|
||||
/// <list type="bullet">
|
||||
/// <item>
|
||||
/// <b>Nullable strings ↔ empty strings.</b> A proto3 scalar string cannot be absent,
|
||||
/// so a null .NET string is written as <see cref="string.Empty"/> and an empty wire
|
||||
/// string is read back as <see langword="null"/>. This is the convention already in
|
||||
/// force on <see cref="AuditEventDtoMapper"/>, and it is why no field on this wire
|
||||
/// may distinguish "null" from "deliberately empty".
|
||||
/// </item>
|
||||
/// <item>
|
||||
/// <b>Nullable <see cref="Guid"/> ↔ string.</b> Execution ids travel as their "D"
|
||||
/// string form; the empty string means <see langword="null"/>. A malformed non-empty
|
||||
/// value throws out of <c>FromDto</c> rather than degrading to null — a corrupt
|
||||
/// correlation id must not be laundered into "no correlation".
|
||||
/// </item>
|
||||
/// <item>
|
||||
/// <b>Nullable numbers and booleans ↔ protobuf wrapper types.</b>
|
||||
/// <c>Int32Value</c>/<c>Int64Value</c>/<c>DoubleValue</c>/<c>BoolValue</c> preserve
|
||||
/// true null. Several health gauges (<c>LocalDbOplogBacklog</c>,
|
||||
/// <c>LocalDbReplicationConnected</c>) are documented as "null means unknown, and
|
||||
/// that is NOT the same as zero/false"; collapsing them to a bare scalar would
|
||||
/// report a broken replication pair as healthy.
|
||||
/// </item>
|
||||
/// <item>
|
||||
/// <b>Nullable collections ↔ wrapper messages.</b> proto3 cannot express presence on
|
||||
/// a <c>repeated</c> or <c>map</c> field, so the three nullable
|
||||
/// <see cref="SiteHealthReport"/> collections travel inside single-field wrapper
|
||||
/// messages (<c>ConnectionEndpointMapDto</c>, <c>TagQualityMapDto</c>,
|
||||
/// <c>NodeStatusListDto</c>). An absent wrapper is null; a present-but-empty wrapper
|
||||
/// is an empty collection.
|
||||
/// </item>
|
||||
/// <item>
|
||||
/// <b><see cref="DateTimeOffset"/> normalizes to a UTC instant.</b> A protobuf
|
||||
/// <c>Timestamp</c> is an instant, not an offset-qualified local time, so the offset
|
||||
/// component is dropped and the value round-trips with <c>Offset == TimeSpan.Zero</c>.
|
||||
/// Every producer in this system stamps UTC (the repo-wide invariant; e.g.
|
||||
/// <c>Notify.Send</c> uses <c>DateTimeOffset.UtcNow</c>), so this is lossless in
|
||||
/// practice and the instant is preserved regardless.
|
||||
/// </item>
|
||||
/// </list>
|
||||
/// </remarks>
|
||||
public static class CentralControlDtoMapper
|
||||
{
|
||||
// -----------------------------------------------------------------------
|
||||
// Notification Outbox (#21)
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
/// <summary>Projects a <see cref="NotificationSubmit"/> onto its wire DTO.</summary>
|
||||
/// <param name="msg">The notification submission to project.</param>
|
||||
/// <returns>The wire-format DTO; null strings and null execution ids collapse to empty strings.</returns>
|
||||
public static NotificationSubmitDto ToDto(NotificationSubmit msg)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(msg);
|
||||
|
||||
return new NotificationSubmitDto
|
||||
{
|
||||
NotificationId = msg.NotificationId,
|
||||
ListName = msg.ListName,
|
||||
Subject = msg.Subject,
|
||||
Body = msg.Body,
|
||||
SourceSiteId = msg.SourceSiteId,
|
||||
SourceInstanceId = msg.SourceInstanceId ?? string.Empty,
|
||||
SourceScript = msg.SourceScript ?? string.Empty,
|
||||
SiteEnqueuedAt = Timestamp.FromDateTimeOffset(msg.SiteEnqueuedAt),
|
||||
OriginExecutionId = GuidToWire(msg.OriginExecutionId),
|
||||
OriginParentExecutionId = GuidToWire(msg.OriginParentExecutionId),
|
||||
SourceNode = msg.SourceNode ?? string.Empty,
|
||||
};
|
||||
}
|
||||
|
||||
/// <summary>Reconstructs a <see cref="NotificationSubmit"/> from its wire DTO.</summary>
|
||||
/// <param name="dto">The wire-format DTO to reconstruct.</param>
|
||||
/// <returns>The in-process message; empty strings rehydrate as null.</returns>
|
||||
public static NotificationSubmit FromDto(NotificationSubmitDto dto)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(dto);
|
||||
|
||||
return new NotificationSubmit(
|
||||
NotificationId: dto.NotificationId,
|
||||
ListName: dto.ListName,
|
||||
Subject: dto.Subject,
|
||||
Body: dto.Body,
|
||||
SourceSiteId: dto.SourceSiteId,
|
||||
SourceInstanceId: NullIfEmpty(dto.SourceInstanceId),
|
||||
SourceScript: NullIfEmpty(dto.SourceScript),
|
||||
SiteEnqueuedAt: dto.SiteEnqueuedAt.ToDateTimeOffset(),
|
||||
OriginExecutionId: GuidFromWire(dto.OriginExecutionId),
|
||||
OriginParentExecutionId: GuidFromWire(dto.OriginParentExecutionId),
|
||||
SourceNode: NullIfEmpty(dto.SourceNode));
|
||||
}
|
||||
|
||||
/// <summary>Projects a <see cref="NotificationSubmitAck"/> onto its wire DTO.</summary>
|
||||
/// <param name="msg">The ack to project.</param>
|
||||
/// <returns>The wire-format DTO; a null error collapses to an empty string.</returns>
|
||||
public static NotificationSubmitAckDto ToDto(NotificationSubmitAck msg)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(msg);
|
||||
|
||||
return new NotificationSubmitAckDto
|
||||
{
|
||||
NotificationId = msg.NotificationId,
|
||||
Accepted = msg.Accepted,
|
||||
Error = msg.Error ?? string.Empty,
|
||||
};
|
||||
}
|
||||
|
||||
/// <summary>Reconstructs a <see cref="NotificationSubmitAck"/> from its wire DTO.</summary>
|
||||
/// <param name="dto">The wire-format DTO to reconstruct.</param>
|
||||
/// <returns>The in-process ack; an empty error rehydrates as null.</returns>
|
||||
public static NotificationSubmitAck FromDto(NotificationSubmitAckDto dto)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(dto);
|
||||
|
||||
return new NotificationSubmitAck(
|
||||
NotificationId: dto.NotificationId,
|
||||
Accepted: dto.Accepted,
|
||||
Error: NullIfEmpty(dto.Error));
|
||||
}
|
||||
|
||||
/// <summary>Projects a <see cref="NotificationStatusQuery"/> onto its wire DTO.</summary>
|
||||
/// <param name="msg">The status query to project.</param>
|
||||
/// <returns>The wire-format DTO.</returns>
|
||||
public static NotificationStatusQueryDto ToDto(NotificationStatusQuery msg)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(msg);
|
||||
|
||||
return new NotificationStatusQueryDto
|
||||
{
|
||||
CorrelationId = msg.CorrelationId,
|
||||
NotificationId = msg.NotificationId,
|
||||
};
|
||||
}
|
||||
|
||||
/// <summary>Reconstructs a <see cref="NotificationStatusQuery"/> from its wire DTO.</summary>
|
||||
/// <param name="dto">The wire-format DTO to reconstruct.</param>
|
||||
/// <returns>The in-process query.</returns>
|
||||
public static NotificationStatusQuery FromDto(NotificationStatusQueryDto dto)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(dto);
|
||||
|
||||
return new NotificationStatusQuery(
|
||||
CorrelationId: dto.CorrelationId,
|
||||
NotificationId: dto.NotificationId);
|
||||
}
|
||||
|
||||
/// <summary>Projects a <see cref="NotificationStatusResponse"/> onto its wire DTO.</summary>
|
||||
/// <param name="msg">The status response to project.</param>
|
||||
/// <returns>The wire-format DTO; a null delivery timestamp leaves the field unset.</returns>
|
||||
public static NotificationStatusResponseDto ToDto(NotificationStatusResponse msg)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(msg);
|
||||
|
||||
var dto = new NotificationStatusResponseDto
|
||||
{
|
||||
CorrelationId = msg.CorrelationId,
|
||||
Found = msg.Found,
|
||||
Status = msg.Status,
|
||||
RetryCount = msg.RetryCount,
|
||||
LastError = msg.LastError ?? string.Empty,
|
||||
};
|
||||
|
||||
if (msg.DeliveredAt.HasValue)
|
||||
{
|
||||
dto.DeliveredAt = Timestamp.FromDateTimeOffset(msg.DeliveredAt.Value);
|
||||
}
|
||||
|
||||
return dto;
|
||||
}
|
||||
|
||||
/// <summary>Reconstructs a <see cref="NotificationStatusResponse"/> from its wire DTO.</summary>
|
||||
/// <param name="dto">The wire-format DTO to reconstruct.</param>
|
||||
/// <returns>The in-process response; an unset delivery timestamp rehydrates as null.</returns>
|
||||
public static NotificationStatusResponse FromDto(NotificationStatusResponseDto dto)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(dto);
|
||||
|
||||
return new NotificationStatusResponse(
|
||||
CorrelationId: dto.CorrelationId,
|
||||
Found: dto.Found,
|
||||
Status: dto.Status,
|
||||
RetryCount: dto.RetryCount,
|
||||
LastError: NullIfEmpty(dto.LastError),
|
||||
DeliveredAt: dto.DeliveredAt?.ToDateTimeOffset());
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// Audit Log (#23) ingest — envelopes only; rows go through the existing mappers
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
/// <summary>
|
||||
/// Projects an <see cref="IngestAuditEventsCommand"/> onto the shared
|
||||
/// <see cref="AuditEventBatch"/> wire message.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The per-row projection is <see cref="AuditEventDtoMapper.ToDto"/>, which is lossy
|
||||
/// by design: <c>ForwardState</c> is site-local storage state and <c>IngestedAtUtc</c>
|
||||
/// is stamped centrally at ingest, so neither travels.
|
||||
/// </remarks>
|
||||
/// <param name="cmd">The ingest command to project.</param>
|
||||
/// <returns>A batch carrying one DTO per audit event, in order.</returns>
|
||||
public static AuditEventBatch ToDto(IngestAuditEventsCommand cmd)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(cmd);
|
||||
|
||||
var batch = new AuditEventBatch();
|
||||
foreach (var evt in cmd.Events)
|
||||
{
|
||||
batch.Events.Add(AuditEventDtoMapper.ToDto(evt));
|
||||
}
|
||||
|
||||
return batch;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Reconstructs an <see cref="IngestAuditEventsCommand"/> from the shared
|
||||
/// <see cref="AuditEventBatch"/> wire message — the shape central's
|
||||
/// <c>CentralCommunicationActor</c> already handles.
|
||||
/// </summary>
|
||||
/// <param name="batch">The wire batch to reconstruct.</param>
|
||||
/// <returns>The in-process ingest command.</returns>
|
||||
public static IngestAuditEventsCommand FromDto(AuditEventBatch batch)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(batch);
|
||||
|
||||
var events = new List<ZB.MOM.WW.Audit.AuditEvent>(batch.Events.Count);
|
||||
foreach (var dto in batch.Events)
|
||||
{
|
||||
events.Add(AuditEventDtoMapper.FromDto(dto));
|
||||
}
|
||||
|
||||
return new IngestAuditEventsCommand(events);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Projects an <see cref="IngestCachedTelemetryCommand"/> onto the shared
|
||||
/// <see cref="CachedTelemetryBatch"/> wire message.
|
||||
/// </summary>
|
||||
/// <param name="cmd">The cached-telemetry ingest command to project.</param>
|
||||
/// <returns>A batch carrying one packet (audit row + operational row) per entry, in order.</returns>
|
||||
public static CachedTelemetryBatch ToDto(IngestCachedTelemetryCommand cmd)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(cmd);
|
||||
|
||||
var batch = new CachedTelemetryBatch();
|
||||
foreach (var entry in cmd.Entries)
|
||||
{
|
||||
batch.Packets.Add(new CachedTelemetryPacket
|
||||
{
|
||||
AuditEvent = AuditEventDtoMapper.ToDto(entry.Audit),
|
||||
Operational = SiteCallDtoMapper.ToDto(entry.SiteCall),
|
||||
});
|
||||
}
|
||||
|
||||
return batch;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Reconstructs an <see cref="IngestCachedTelemetryCommand"/> from the shared
|
||||
/// <see cref="CachedTelemetryBatch"/> wire message.
|
||||
/// </summary>
|
||||
/// <param name="batch">The wire batch to reconstruct.</param>
|
||||
/// <returns>The in-process dual-write ingest command.</returns>
|
||||
public static IngestCachedTelemetryCommand FromDto(CachedTelemetryBatch batch)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(batch);
|
||||
|
||||
var entries = new List<CachedTelemetryEntry>(batch.Packets.Count);
|
||||
foreach (var packet in batch.Packets)
|
||||
{
|
||||
entries.Add(new CachedTelemetryEntry(
|
||||
AuditEventDtoMapper.FromDto(packet.AuditEvent),
|
||||
SiteCallDtoMapper.FromDto(packet.Operational)));
|
||||
}
|
||||
|
||||
return new IngestCachedTelemetryCommand(entries);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Projects the accepted-id list of an ingest reply onto the shared
|
||||
/// <see cref="IngestAck"/> wire message. Shared by both ingest RPCs — the two
|
||||
/// central reply types differ only in which handler produced them.
|
||||
/// </summary>
|
||||
/// <param name="acceptedEventIds">Ids central considers durably persisted.</param>
|
||||
/// <returns>The wire ack carrying the ids in "D" string form, in order.</returns>
|
||||
public static IngestAck ToIngestAck(IReadOnlyList<Guid> acceptedEventIds)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(acceptedEventIds);
|
||||
|
||||
var ack = new IngestAck();
|
||||
foreach (var id in acceptedEventIds)
|
||||
{
|
||||
ack.AcceptedEventIds.Add(id.ToString());
|
||||
}
|
||||
|
||||
return ack;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Reads the accepted-id list back out of an <see cref="IngestAck"/>.
|
||||
/// </summary>
|
||||
/// <param name="ack">The wire ack to read.</param>
|
||||
/// <returns>The accepted event ids, in wire order.</returns>
|
||||
public static IReadOnlyList<Guid> FromIngestAck(IngestAck ack)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(ack);
|
||||
|
||||
var ids = new List<Guid>(ack.AcceptedEventIds.Count);
|
||||
foreach (var id in ack.AcceptedEventIds)
|
||||
{
|
||||
ids.Add(Guid.Parse(id));
|
||||
}
|
||||
|
||||
return ids;
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// Startup reconciliation
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
/// <summary>Projects a <see cref="ReconcileSiteRequest"/> onto its wire DTO.</summary>
|
||||
/// <param name="msg">The reconcile request to project.</param>
|
||||
/// <returns>The wire-format DTO carrying the node's local name→revision-hash inventory.</returns>
|
||||
public static ReconcileSiteRequestDto ToDto(ReconcileSiteRequest msg)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(msg);
|
||||
|
||||
var dto = new ReconcileSiteRequestDto
|
||||
{
|
||||
SiteIdentifier = msg.SiteIdentifier,
|
||||
NodeId = msg.NodeId,
|
||||
};
|
||||
|
||||
foreach (var (name, hash) in msg.LocalNameToRevisionHash)
|
||||
{
|
||||
dto.LocalNameToRevisionHash[name] = hash;
|
||||
}
|
||||
|
||||
return dto;
|
||||
}
|
||||
|
||||
/// <summary>Reconstructs a <see cref="ReconcileSiteRequest"/> from its wire DTO.</summary>
|
||||
/// <param name="dto">The wire-format DTO to reconstruct.</param>
|
||||
/// <returns>The in-process reconcile request.</returns>
|
||||
public static ReconcileSiteRequest FromDto(ReconcileSiteRequestDto dto)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(dto);
|
||||
|
||||
return new ReconcileSiteRequest(
|
||||
SiteIdentifier: dto.SiteIdentifier,
|
||||
NodeId: dto.NodeId,
|
||||
LocalNameToRevisionHash: new Dictionary<string, string>(dto.LocalNameToRevisionHash));
|
||||
}
|
||||
|
||||
/// <summary>Projects a <see cref="ReconcileSiteResponse"/> onto its wire DTO.</summary>
|
||||
/// <param name="msg">The reconcile response to project.</param>
|
||||
/// <returns>The wire-format DTO carrying the gap items, orphan names and fetch base URL.</returns>
|
||||
public static ReconcileSiteResponseDto ToDto(ReconcileSiteResponse msg)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(msg);
|
||||
|
||||
var dto = new ReconcileSiteResponseDto
|
||||
{
|
||||
CentralFetchBaseUrl = msg.CentralFetchBaseUrl,
|
||||
};
|
||||
|
||||
foreach (var item in msg.Gap)
|
||||
{
|
||||
dto.Gap.Add(new ReconcileGapItemDto
|
||||
{
|
||||
InstanceUniqueName = item.InstanceUniqueName,
|
||||
DeploymentId = item.DeploymentId,
|
||||
RevisionHash = item.RevisionHash,
|
||||
IsEnabled = item.IsEnabled,
|
||||
FetchToken = item.FetchToken,
|
||||
});
|
||||
}
|
||||
|
||||
dto.OrphanNames.AddRange(msg.OrphanNames);
|
||||
|
||||
return dto;
|
||||
}
|
||||
|
||||
/// <summary>Reconstructs a <see cref="ReconcileSiteResponse"/> from its wire DTO.</summary>
|
||||
/// <param name="dto">The wire-format DTO to reconstruct.</param>
|
||||
/// <returns>The in-process reconcile response.</returns>
|
||||
public static ReconcileSiteResponse FromDto(ReconcileSiteResponseDto dto)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(dto);
|
||||
|
||||
var gap = new List<ReconcileGapItem>(dto.Gap.Count);
|
||||
foreach (var item in dto.Gap)
|
||||
{
|
||||
gap.Add(new ReconcileGapItem(
|
||||
InstanceUniqueName: item.InstanceUniqueName,
|
||||
DeploymentId: item.DeploymentId,
|
||||
RevisionHash: item.RevisionHash,
|
||||
IsEnabled: item.IsEnabled,
|
||||
FetchToken: item.FetchToken));
|
||||
}
|
||||
|
||||
return new ReconcileSiteResponse(
|
||||
Gap: gap,
|
||||
OrphanNames: dto.OrphanNames.ToList(),
|
||||
CentralFetchBaseUrl: dto.CentralFetchBaseUrl);
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// Health Monitoring (#11)
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
/// <summary>Projects a <see cref="SiteHealthReport"/> onto its wire DTO.</summary>
|
||||
/// <remarks>
|
||||
/// The three nullable collections travel inside wrapper messages so a null stays
|
||||
/// distinguishable from an empty collection; the nullable gauges travel in protobuf
|
||||
/// wrapper types for the same reason.
|
||||
/// </remarks>
|
||||
/// <param name="msg">The health report to project.</param>
|
||||
/// <returns>The wire-format DTO.</returns>
|
||||
public static SiteHealthReportDto ToDto(SiteHealthReport msg)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(msg);
|
||||
|
||||
var dto = new SiteHealthReportDto
|
||||
{
|
||||
SiteId = msg.SiteId,
|
||||
SequenceNumber = msg.SequenceNumber,
|
||||
ReportTimestamp = Timestamp.FromDateTimeOffset(msg.ReportTimestamp),
|
||||
ScriptErrorCount = msg.ScriptErrorCount,
|
||||
AlarmEvaluationErrorCount = msg.AlarmEvaluationErrorCount,
|
||||
DeadLetterCount = msg.DeadLetterCount,
|
||||
DeployedInstanceCount = msg.DeployedInstanceCount,
|
||||
EnabledInstanceCount = msg.EnabledInstanceCount,
|
||||
DisabledInstanceCount = msg.DisabledInstanceCount,
|
||||
NodeRole = msg.NodeRole,
|
||||
NodeHostname = msg.NodeHostname,
|
||||
ParkedMessageCount = msg.ParkedMessageCount,
|
||||
SiteAuditWriteFailures = msg.SiteAuditWriteFailures,
|
||||
AuditRedactionFailure = msg.AuditRedactionFailure,
|
||||
SiteEventLogWriteFailures = msg.SiteEventLogWriteFailures,
|
||||
OldestParkedMessageAgeSeconds = msg.OldestParkedMessageAgeSeconds,
|
||||
ScriptQueueDepth = msg.ScriptQueueDepth,
|
||||
ScriptBusyThreads = msg.ScriptBusyThreads,
|
||||
ScriptOldestBusyAgeSeconds = msg.ScriptOldestBusyAgeSeconds,
|
||||
LocalDbReplicationConnected = msg.LocalDbReplicationConnected,
|
||||
LocalDbOplogBacklog = msg.LocalDbOplogBacklog,
|
||||
};
|
||||
|
||||
foreach (var (name, health) in msg.DataConnectionStatuses)
|
||||
{
|
||||
dto.DataConnectionStatuses[name] = ToDto(health);
|
||||
}
|
||||
|
||||
foreach (var (name, resolution) in msg.TagResolutionCounts)
|
||||
{
|
||||
dto.TagResolutionCounts[name] = new TagResolutionStatusDto
|
||||
{
|
||||
TotalSubscribed = resolution.TotalSubscribed,
|
||||
SuccessfullyResolved = resolution.SuccessfullyResolved,
|
||||
};
|
||||
}
|
||||
|
||||
foreach (var (name, depth) in msg.StoreAndForwardBufferDepths)
|
||||
{
|
||||
dto.StoreAndForwardBufferDepths[name] = depth;
|
||||
}
|
||||
|
||||
if (msg.DataConnectionEndpoints is { } endpoints)
|
||||
{
|
||||
var wrapper = new ConnectionEndpointMapDto();
|
||||
foreach (var (name, endpoint) in endpoints)
|
||||
{
|
||||
wrapper.Entries[name] = endpoint;
|
||||
}
|
||||
|
||||
dto.DataConnectionEndpoints = wrapper;
|
||||
}
|
||||
|
||||
if (msg.DataConnectionTagQuality is { } tagQuality)
|
||||
{
|
||||
var wrapper = new TagQualityMapDto();
|
||||
foreach (var (name, counts) in tagQuality)
|
||||
{
|
||||
wrapper.Entries[name] = new TagQualityCountsDto
|
||||
{
|
||||
Good = counts.Good,
|
||||
Bad = counts.Bad,
|
||||
Uncertain = counts.Uncertain,
|
||||
};
|
||||
}
|
||||
|
||||
dto.DataConnectionTagQuality = wrapper;
|
||||
}
|
||||
|
||||
if (msg.ClusterNodes is { } clusterNodes)
|
||||
{
|
||||
var wrapper = new NodeStatusListDto();
|
||||
foreach (var node in clusterNodes)
|
||||
{
|
||||
wrapper.Nodes.Add(new NodeStatusDto
|
||||
{
|
||||
Hostname = node.Hostname,
|
||||
IsOnline = node.IsOnline,
|
||||
Role = node.Role,
|
||||
});
|
||||
}
|
||||
|
||||
dto.ClusterNodes = wrapper;
|
||||
}
|
||||
|
||||
if (msg.SiteAuditBacklog is { } backlog)
|
||||
{
|
||||
var snapshot = new SiteAuditBacklogSnapshotDto
|
||||
{
|
||||
PendingCount = backlog.PendingCount,
|
||||
OnDiskBytes = backlog.OnDiskBytes,
|
||||
};
|
||||
|
||||
if (backlog.OldestPendingUtc.HasValue)
|
||||
{
|
||||
snapshot.OldestPendingUtc = Timestamp.FromDateTime(EnsureUtc(backlog.OldestPendingUtc.Value));
|
||||
}
|
||||
|
||||
dto.SiteAuditBacklog = snapshot;
|
||||
}
|
||||
|
||||
return dto;
|
||||
}
|
||||
|
||||
/// <summary>Reconstructs a <see cref="SiteHealthReport"/> from its wire DTO.</summary>
|
||||
/// <param name="dto">The wire-format DTO to reconstruct.</param>
|
||||
/// <returns>The in-process health report; absent wrappers rehydrate as null, not as empty.</returns>
|
||||
public static SiteHealthReport FromDto(SiteHealthReportDto dto)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(dto);
|
||||
|
||||
var statuses = new Dictionary<string, ConnectionHealth>(dto.DataConnectionStatuses.Count);
|
||||
foreach (var (name, health) in dto.DataConnectionStatuses)
|
||||
{
|
||||
statuses[name] = FromDto(health);
|
||||
}
|
||||
|
||||
var resolution = new Dictionary<string, TagResolutionStatus>(dto.TagResolutionCounts.Count);
|
||||
foreach (var (name, counts) in dto.TagResolutionCounts)
|
||||
{
|
||||
resolution[name] = new TagResolutionStatus(counts.TotalSubscribed, counts.SuccessfullyResolved);
|
||||
}
|
||||
|
||||
var bufferDepths = new Dictionary<string, int>(dto.StoreAndForwardBufferDepths);
|
||||
|
||||
Dictionary<string, string>? endpoints = null;
|
||||
if (dto.DataConnectionEndpoints is { } endpointWrapper)
|
||||
{
|
||||
endpoints = new Dictionary<string, string>(endpointWrapper.Entries);
|
||||
}
|
||||
|
||||
Dictionary<string, TagQualityCounts>? tagQuality = null;
|
||||
if (dto.DataConnectionTagQuality is { } tagQualityWrapper)
|
||||
{
|
||||
tagQuality = new Dictionary<string, TagQualityCounts>(tagQualityWrapper.Entries.Count);
|
||||
foreach (var (name, counts) in tagQualityWrapper.Entries)
|
||||
{
|
||||
tagQuality[name] = new TagQualityCounts(counts.Good, counts.Bad, counts.Uncertain);
|
||||
}
|
||||
}
|
||||
|
||||
List<NodeStatus>? clusterNodes = null;
|
||||
if (dto.ClusterNodes is { } nodeWrapper)
|
||||
{
|
||||
clusterNodes = new List<NodeStatus>(nodeWrapper.Nodes.Count);
|
||||
foreach (var node in nodeWrapper.Nodes)
|
||||
{
|
||||
clusterNodes.Add(new NodeStatus(node.Hostname, node.IsOnline, node.Role));
|
||||
}
|
||||
}
|
||||
|
||||
SiteAuditBacklogSnapshot? backlog = null;
|
||||
if (dto.SiteAuditBacklog is { } snapshot)
|
||||
{
|
||||
backlog = new SiteAuditBacklogSnapshot(
|
||||
PendingCount: snapshot.PendingCount,
|
||||
OldestPendingUtc: snapshot.OldestPendingUtc is null
|
||||
? null
|
||||
: DateTime.SpecifyKind(snapshot.OldestPendingUtc.ToDateTime(), DateTimeKind.Utc),
|
||||
OnDiskBytes: snapshot.OnDiskBytes);
|
||||
}
|
||||
|
||||
return new SiteHealthReport(
|
||||
SiteId: dto.SiteId,
|
||||
SequenceNumber: dto.SequenceNumber,
|
||||
ReportTimestamp: dto.ReportTimestamp.ToDateTimeOffset(),
|
||||
DataConnectionStatuses: statuses,
|
||||
TagResolutionCounts: resolution,
|
||||
ScriptErrorCount: dto.ScriptErrorCount,
|
||||
AlarmEvaluationErrorCount: dto.AlarmEvaluationErrorCount,
|
||||
StoreAndForwardBufferDepths: bufferDepths,
|
||||
DeadLetterCount: dto.DeadLetterCount,
|
||||
DeployedInstanceCount: dto.DeployedInstanceCount,
|
||||
EnabledInstanceCount: dto.EnabledInstanceCount,
|
||||
DisabledInstanceCount: dto.DisabledInstanceCount,
|
||||
NodeRole: dto.NodeRole,
|
||||
NodeHostname: dto.NodeHostname,
|
||||
DataConnectionEndpoints: endpoints,
|
||||
DataConnectionTagQuality: tagQuality,
|
||||
ParkedMessageCount: dto.ParkedMessageCount,
|
||||
ClusterNodes: clusterNodes,
|
||||
SiteAuditWriteFailures: dto.SiteAuditWriteFailures,
|
||||
AuditRedactionFailure: dto.AuditRedactionFailure,
|
||||
SiteAuditBacklog: backlog,
|
||||
SiteEventLogWriteFailures: dto.SiteEventLogWriteFailures,
|
||||
OldestParkedMessageAgeSeconds: dto.OldestParkedMessageAgeSeconds)
|
||||
{
|
||||
// Init-only members: SiteHealthReport surfaces the scheduler and LocalDb
|
||||
// gauges as init properties rather than positional parameters, so they
|
||||
// cannot be passed to the constructor above.
|
||||
ScriptQueueDepth = dto.ScriptQueueDepth,
|
||||
ScriptBusyThreads = dto.ScriptBusyThreads,
|
||||
ScriptOldestBusyAgeSeconds = dto.ScriptOldestBusyAgeSeconds,
|
||||
LocalDbReplicationConnected = dto.LocalDbReplicationConnected,
|
||||
LocalDbOplogBacklog = dto.LocalDbOplogBacklog,
|
||||
};
|
||||
}
|
||||
|
||||
/// <summary>Projects a <see cref="SiteHealthReportAck"/> onto its wire DTO.</summary>
|
||||
/// <param name="msg">The ack to project.</param>
|
||||
/// <returns>The wire-format DTO; a null error collapses to an empty string.</returns>
|
||||
public static SiteHealthReportAckDto ToDto(SiteHealthReportAck msg)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(msg);
|
||||
|
||||
return new SiteHealthReportAckDto
|
||||
{
|
||||
SiteId = msg.SiteId,
|
||||
SequenceNumber = msg.SequenceNumber,
|
||||
Accepted = msg.Accepted,
|
||||
Error = msg.Error ?? string.Empty,
|
||||
};
|
||||
}
|
||||
|
||||
/// <summary>Reconstructs a <see cref="SiteHealthReportAck"/> from its wire DTO.</summary>
|
||||
/// <param name="dto">The wire-format DTO to reconstruct.</param>
|
||||
/// <returns>The in-process ack; an empty error rehydrates as null.</returns>
|
||||
public static SiteHealthReportAck FromDto(SiteHealthReportAckDto dto)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(dto);
|
||||
|
||||
return new SiteHealthReportAck(
|
||||
SiteId: dto.SiteId,
|
||||
SequenceNumber: dto.SequenceNumber,
|
||||
Accepted: dto.Accepted,
|
||||
Error: NullIfEmpty(dto.Error));
|
||||
}
|
||||
|
||||
/// <summary>Projects a <see cref="HeartbeatMessage"/> onto its wire DTO.</summary>
|
||||
/// <param name="msg">The heartbeat to project.</param>
|
||||
/// <returns>The wire-format DTO.</returns>
|
||||
public static HeartbeatDto ToDto(HeartbeatMessage msg)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(msg);
|
||||
|
||||
return new HeartbeatDto
|
||||
{
|
||||
SiteId = msg.SiteId,
|
||||
NodeHostname = msg.NodeHostname,
|
||||
IsActive = msg.IsActive,
|
||||
Timestamp = Timestamp.FromDateTimeOffset(msg.Timestamp),
|
||||
};
|
||||
}
|
||||
|
||||
/// <summary>Reconstructs a <see cref="HeartbeatMessage"/> from its wire DTO.</summary>
|
||||
/// <param name="dto">The wire-format DTO to reconstruct.</param>
|
||||
/// <returns>The in-process heartbeat.</returns>
|
||||
public static HeartbeatMessage FromDto(HeartbeatDto dto)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(dto);
|
||||
|
||||
return new HeartbeatMessage(
|
||||
SiteId: dto.SiteId,
|
||||
NodeHostname: dto.NodeHostname,
|
||||
IsActive: dto.IsActive,
|
||||
Timestamp: dto.Timestamp.ToDateTimeOffset());
|
||||
}
|
||||
|
||||
/// <summary>Projects a <see cref="ConnectionHealth"/> onto its wire enum.</summary>
|
||||
/// <param name="health">The connection health to project.</param>
|
||||
/// <returns>The wire enum value. Never <c>ConnectionHealthUnspecified</c>.</returns>
|
||||
/// <exception cref="ArgumentOutOfRangeException">
|
||||
/// A <see cref="ConnectionHealth"/> value was added without extending this mapping.
|
||||
/// Throwing beats a silent default: an unmapped state would otherwise be reported as
|
||||
/// whichever value happened to be first.
|
||||
/// </exception>
|
||||
public static ConnectionHealthEnum ToDto(ConnectionHealth health) => health switch
|
||||
{
|
||||
ConnectionHealth.Connected => ConnectionHealthEnum.ConnectionHealthConnected,
|
||||
ConnectionHealth.Disconnected => ConnectionHealthEnum.ConnectionHealthDisconnected,
|
||||
ConnectionHealth.Connecting => ConnectionHealthEnum.ConnectionHealthConnecting,
|
||||
ConnectionHealth.Error => ConnectionHealthEnum.ConnectionHealthError,
|
||||
_ => throw new ArgumentOutOfRangeException(nameof(health), health, "Unmapped ConnectionHealth value"),
|
||||
};
|
||||
|
||||
/// <summary>Reconstructs a <see cref="ConnectionHealth"/> from its wire enum.</summary>
|
||||
/// <remarks>
|
||||
/// An unspecified or unknown wire value decodes to <see cref="ConnectionHealth.Error"/>,
|
||||
/// never to <see cref="ConnectionHealth.Connected"/>. A newer site sending a value this
|
||||
/// build has never heard of must not have it rendered as "healthy" on the central
|
||||
/// health page — the safe direction for an unknown connection state is "not working".
|
||||
/// </remarks>
|
||||
/// <param name="health">The wire enum value to reconstruct.</param>
|
||||
/// <returns>The in-process connection health.</returns>
|
||||
public static ConnectionHealth FromDto(ConnectionHealthEnum health) => health switch
|
||||
{
|
||||
ConnectionHealthEnum.ConnectionHealthConnected => ConnectionHealth.Connected,
|
||||
ConnectionHealthEnum.ConnectionHealthDisconnected => ConnectionHealth.Disconnected,
|
||||
ConnectionHealthEnum.ConnectionHealthConnecting => ConnectionHealth.Connecting,
|
||||
_ => ConnectionHealth.Error,
|
||||
};
|
||||
|
||||
private static string GuidToWire(Guid? value) =>
|
||||
value?.ToString() ?? string.Empty;
|
||||
|
||||
private static Guid? GuidFromWire(string? value) =>
|
||||
string.IsNullOrEmpty(value) ? null : Guid.Parse(value);
|
||||
|
||||
private static string? NullIfEmpty(string? value) =>
|
||||
string.IsNullOrEmpty(value) ? null : value;
|
||||
|
||||
// All ScadaBridge timestamps are UTC by invariant; Timestamp.FromDateTime requires
|
||||
// UTC kind. Specify (never convert) so a value read back from SQLite with Kind=Utc
|
||||
// passes through and a defensively-unspecified one is treated as the UTC it already
|
||||
// is. Mirrors AuditEventDtoMapper/SiteCallDtoMapper.EnsureUtc.
|
||||
private static DateTime EnsureUtc(DateTime value) =>
|
||||
value.Kind == DateTimeKind.Utc ? value : DateTime.SpecifyKind(value, DateTimeKind.Utc);
|
||||
}
|
||||
@@ -0,0 +1,299 @@
|
||||
using Akka.Actor;
|
||||
using Google.Protobuf.WellKnownTypes;
|
||||
using Grpc.Core;
|
||||
using Microsoft.Extensions.Logging;
|
||||
using Microsoft.Extensions.Options;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Audit;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Deployment;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Health;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Notification;
|
||||
using GrpcStatus = Grpc.Core.Status;
|
||||
|
||||
namespace ZB.MOM.WW.ScadaBridge.Communication.Grpc;
|
||||
|
||||
/// <summary>
|
||||
/// Central-hosted gRPC face of the seven site→central control messages
|
||||
/// (<c>Protos/central_control.proto</c>). Decodes each request onto the SAME in-process
|
||||
/// message type the Akka <c>ClusterClient</c> path already carries, <c>Ask</c>s
|
||||
/// <see cref="Actors.CentralCommunicationActor"/>, and encodes the reply back.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// <para>
|
||||
/// <b>Direction is inverted from <see cref="SiteStreamGrpcServer"/>.</b> That server runs on a
|
||||
/// site and central dials in; this one runs on CENTRAL and the site dials in. The two listen on
|
||||
/// the same port number (8083) on their respective nodes, which is symmetry, not a collision —
|
||||
/// a node is either central or a site, never both.
|
||||
/// </para>
|
||||
/// <para>
|
||||
/// <b>Zero handler logic lives here.</b> Every RPC lands on a receive
|
||||
/// <c>CentralCommunicationActor</c> already implements for the ClusterClient path, so the two
|
||||
/// transports cannot drift in behaviour: the actor is the single implementation, and this class
|
||||
/// is a codec plus an <c>Ask</c>. That is also why the service takes the actor through
|
||||
/// <see cref="SetReady"/> rather than resolving anything from DI — the actor is created by the
|
||||
/// host's Akka bootstrap, not by the container.
|
||||
/// </para>
|
||||
/// <para>
|
||||
/// <b>Fault semantics deliberately differ from <see cref="SiteStreamGrpcServer"/>'s ingest
|
||||
/// RPCs.</b> That server answers a failed audit ingest with an EMPTY <c>IngestAck</c>; this one
|
||||
/// fails the call with a non-OK status. Both leave the site's rows <c>Pending</c> for the next
|
||||
/// drain, so the outcome is the same — but this service replaces the ClusterClient path, whose
|
||||
/// documented behaviour is to propagate the fault (<c>CentralCommunicationActor</c>'s
|
||||
/// <c>HandleIngestAuditEvents</c> pipes a <c>Status.Failure</c> back), and preserving that keeps
|
||||
/// a lost batch visible as a failure rather than as a successful call that acked nothing.
|
||||
/// </para>
|
||||
/// <para>
|
||||
/// <b>Status mapping, and why it is not uniform.</b> A site transport may safely re-send a call
|
||||
/// to the peer central node only when the call provably never ran. So:
|
||||
/// </para>
|
||||
/// <list type="bullet">
|
||||
/// <item><see cref="StatusCode.Unavailable"/> — this node is not ready; nothing was dispatched,
|
||||
/// so a cross-node retry is safe and correct.</item>
|
||||
/// <item><see cref="StatusCode.DeadlineExceeded"/> — the <c>Ask</c> timed out. The message WAS
|
||||
/// delivered and may have been processed; retrying it on the other node would duplicate work.</item>
|
||||
/// <item><see cref="StatusCode.Internal"/> — the handler faulted (a piped
|
||||
/// <see cref="Akka.Actor.Status.Failure"/>, e.g. a database error inside reconcile). Same
|
||||
/// reasoning: it ran, so do not re-send it elsewhere.</item>
|
||||
/// </list>
|
||||
/// </remarks>
|
||||
public sealed class CentralControlGrpcService : CentralControlService.CentralControlServiceBase
|
||||
{
|
||||
private readonly ILogger<CentralControlGrpcService> _logger;
|
||||
private readonly CommunicationOptions _options;
|
||||
|
||||
// Null until the host's Akka bootstrap hands the actor over. Doubles as the readiness
|
||||
// flag: a call arriving before then cannot be served and is refused with Unavailable.
|
||||
// Volatile because SetReady runs on the startup thread while calls are served on
|
||||
// Kestrel's thread pool.
|
||||
private volatile IActorRef? _central;
|
||||
|
||||
/// <summary>
|
||||
/// Creates the service. <b>This must remain the only public constructor</b> — see
|
||||
/// <see cref="SetReady"/> for how the actor arrives, and the Host's
|
||||
/// <c>CentralControlAuthInterceptor</c> for the interceptor-side version of the same rule.
|
||||
/// </summary>
|
||||
/// <param name="logger">Logger for readiness and fault diagnostics.</param>
|
||||
/// <param name="options">Communication options supplying the per-RPC <c>Ask</c> timeouts.</param>
|
||||
public CentralControlGrpcService(
|
||||
ILogger<CentralControlGrpcService> logger,
|
||||
IOptions<CommunicationOptions> options)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(logger);
|
||||
ArgumentNullException.ThrowIfNull(options);
|
||||
|
||||
_logger = logger;
|
||||
_options = options.Value;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Hands the <c>CentralCommunicationActor</c> to the service and opens the gate. Mirrors
|
||||
/// <see cref="SiteStreamGrpcServer.SetReady"/>: the gRPC service is a DI singleton created
|
||||
/// before the actor system exists, so the actor arrives post-construction.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The contract is deliberately narrow, exactly as on the site side: it asserts that the
|
||||
/// actor exists and can receive, NOT that every downstream singleton proxy
|
||||
/// (<c>notification-outbox</c>, <c>audit-log-ingest</c>) has registered itself yet. Those
|
||||
/// register moments later in the same startup path, and the actor already answers a call
|
||||
/// that beats them with the same "not available, retry" reply it gives on the ClusterClient
|
||||
/// path — so gating readiness on them would add nothing but a longer window in which sites
|
||||
/// see <see cref="StatusCode.Unavailable"/>.
|
||||
/// </remarks>
|
||||
/// <param name="centralCommunicationActor">The central communication actor.</param>
|
||||
public void SetReady(IActorRef centralCommunicationActor)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(centralCommunicationActor);
|
||||
_central = centralCommunicationActor;
|
||||
}
|
||||
|
||||
/// <summary>Exposed for wiring assertions in tests.</summary>
|
||||
internal bool IsReady => _central is not null;
|
||||
|
||||
/// <inheritdoc />
|
||||
public override async Task<NotificationSubmitAckDto> SubmitNotification(
|
||||
NotificationSubmitDto request, ServerCallContext context)
|
||||
{
|
||||
var central = RequireReady(context);
|
||||
var ack = await AskAsync<NotificationSubmitAck>(
|
||||
central,
|
||||
CentralControlDtoMapper.FromDto(request),
|
||||
_options.NotificationForwardTimeout,
|
||||
context).ConfigureAwait(false);
|
||||
|
||||
return CentralControlDtoMapper.ToDto(ack);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public override async Task<NotificationStatusResponseDto> QueryNotificationStatus(
|
||||
NotificationStatusQueryDto request, ServerCallContext context)
|
||||
{
|
||||
var central = RequireReady(context);
|
||||
var response = await AskAsync<NotificationStatusResponse>(
|
||||
central,
|
||||
CentralControlDtoMapper.FromDto(request),
|
||||
_options.NotificationForwardTimeout,
|
||||
context).ConfigureAwait(false);
|
||||
|
||||
return CentralControlDtoMapper.ToDto(response);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public override async Task<IngestAck> IngestAuditEvents(
|
||||
AuditEventBatch request, ServerCallContext context)
|
||||
{
|
||||
// An empty batch is a no-op the actor need never see; answering it here also means a
|
||||
// site that drains an empty queue does not fail against a not-yet-ready central.
|
||||
if (request.Events.Count == 0)
|
||||
{
|
||||
return new IngestAck();
|
||||
}
|
||||
|
||||
var central = RequireReady(context);
|
||||
var reply = await AskAsync<IngestAuditEventsReply>(
|
||||
central,
|
||||
CentralControlDtoMapper.FromDto(request),
|
||||
SiteStreamGrpcServer.AuditIngestAskTimeout,
|
||||
context).ConfigureAwait(false);
|
||||
|
||||
return CentralControlDtoMapper.ToIngestAck(reply.AcceptedEventIds);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public override async Task<IngestAck> IngestCachedTelemetry(
|
||||
CachedTelemetryBatch request, ServerCallContext context)
|
||||
{
|
||||
if (request.Packets.Count == 0)
|
||||
{
|
||||
return new IngestAck();
|
||||
}
|
||||
|
||||
var central = RequireReady(context);
|
||||
var reply = await AskAsync<IngestCachedTelemetryReply>(
|
||||
central,
|
||||
CentralControlDtoMapper.FromDto(request),
|
||||
SiteStreamGrpcServer.AuditIngestAskTimeout,
|
||||
context).ConfigureAwait(false);
|
||||
|
||||
return CentralControlDtoMapper.ToIngestAck(reply.AcceptedEventIds);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public override async Task<ReconcileSiteResponseDto> ReconcileSite(
|
||||
ReconcileSiteRequestDto request, ServerCallContext context)
|
||||
{
|
||||
var central = RequireReady(context);
|
||||
var response = await AskAsync<ReconcileSiteResponse>(
|
||||
central,
|
||||
CentralControlDtoMapper.FromDto(request),
|
||||
_options.QueryTimeout,
|
||||
context).ConfigureAwait(false);
|
||||
|
||||
return CentralControlDtoMapper.ToDto(response);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public override async Task<SiteHealthReportAckDto> ReportSiteHealth(
|
||||
SiteHealthReportDto request, ServerCallContext context)
|
||||
{
|
||||
var central = RequireReady(context);
|
||||
var ack = await AskAsync<SiteHealthReportAck>(
|
||||
central,
|
||||
CentralControlDtoMapper.FromDto(request),
|
||||
_options.HealthReportTimeout,
|
||||
context).ConfigureAwait(false);
|
||||
|
||||
return CentralControlDtoMapper.ToDto(ack);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Application heartbeat — <b>always answers OK</b>, even when this node is not ready.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The heartbeat is fire-and-forget on both sides: nothing on the site consumes the reply,
|
||||
/// and the site's heartbeat timer must never take a fault (a failing heartbeat that raised
|
||||
/// an error would be a self-inflicted outage on a purely informational signal). So this is
|
||||
/// the one RPC that does not go through <see cref="RequireReady"/>: a heartbeat arriving
|
||||
/// before the actor exists is logged and dropped, exactly as the actor itself drops one
|
||||
/// that arrives before <c>ICentralHealthAggregator</c> is resolvable. Liveness is still
|
||||
/// detected — the aggregator's offline timeout fires when the heartbeats stop landing.
|
||||
/// </remarks>
|
||||
/// <param name="request">The heartbeat.</param>
|
||||
/// <param name="context">The gRPC call context.</param>
|
||||
/// <returns>An empty reply, always.</returns>
|
||||
public override Task<Empty> Heartbeat(HeartbeatDto request, ServerCallContext context)
|
||||
{
|
||||
var central = _central;
|
||||
if (central is null)
|
||||
{
|
||||
_logger.LogDebug(
|
||||
"Dropped a heartbeat from site {SiteId}: the central communication actor is not "
|
||||
+ "ready yet. Heartbeats are fire-and-forget, so the call still succeeds.",
|
||||
request.SiteId);
|
||||
return Task.FromResult(new Empty());
|
||||
}
|
||||
|
||||
// Tell, never Ask: the actor's HandleHeartbeat sends no reply.
|
||||
central.Tell(CentralControlDtoMapper.FromDto(request), ActorRefs.NoSender);
|
||||
return Task.FromResult(new Empty());
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Returns the central communication actor, or throws <see cref="StatusCode.Unavailable"/>
|
||||
/// when the host has not finished bringing the actor system up.
|
||||
/// </summary>
|
||||
private IActorRef RequireReady(ServerCallContext context)
|
||||
{
|
||||
var central = _central;
|
||||
if (central is not null)
|
||||
{
|
||||
return central;
|
||||
}
|
||||
|
||||
_logger.LogWarning(
|
||||
"Refused a control-plane call to {Method}: the central communication actor is not "
|
||||
+ "ready yet. Nothing was dispatched, so the caller may retry (including against "
|
||||
+ "the peer central node).",
|
||||
context.Method);
|
||||
|
||||
throw new RpcException(new GrpcStatus(
|
||||
StatusCode.Unavailable,
|
||||
"Central control plane is not ready: the actor system is still starting."));
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Asks the central actor and maps a fault onto the status code that tells the caller
|
||||
/// whether a cross-node retry is safe. See the class remarks for the mapping rationale.
|
||||
/// </summary>
|
||||
private async Task<TReply> AskAsync<TReply>(
|
||||
IActorRef central, object message, TimeSpan timeout, ServerCallContext context)
|
||||
{
|
||||
try
|
||||
{
|
||||
return await central.Ask<TReply>(message, timeout, context.CancellationToken)
|
||||
.ConfigureAwait(false);
|
||||
}
|
||||
catch (AskTimeoutException ex)
|
||||
{
|
||||
_logger.LogWarning(ex,
|
||||
"Control-plane call {Method} timed out after {Timeout} waiting for the central "
|
||||
+ "communication actor.",
|
||||
context.Method, timeout);
|
||||
throw new RpcException(new GrpcStatus(
|
||||
StatusCode.DeadlineExceeded,
|
||||
$"Central did not answer within {timeout}."));
|
||||
}
|
||||
catch (OperationCanceledException)
|
||||
{
|
||||
// The client gave up or its deadline expired; there is no one left to answer.
|
||||
throw new RpcException(new GrpcStatus(
|
||||
StatusCode.Cancelled, "The call was cancelled."));
|
||||
}
|
||||
catch (Exception ex)
|
||||
{
|
||||
_logger.LogError(ex,
|
||||
"Control-plane call {Method} faulted inside the central communication actor.",
|
||||
context.Method);
|
||||
throw new RpcException(new GrpcStatus(
|
||||
StatusCode.Internal, "Central failed to process the call."));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,258 @@
|
||||
using Akka.Actor;
|
||||
using Grpc.Core;
|
||||
using Microsoft.Extensions.Logging;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Audit;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Deployment;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Health;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Notification;
|
||||
using ZB.MOM.WW.ScadaBridge.Communication.Actors;
|
||||
using AkkaStatus = Akka.Actor.Status;
|
||||
|
||||
namespace ZB.MOM.WW.ScadaBridge.Communication.Grpc;
|
||||
|
||||
/// <summary>
|
||||
/// The <see cref="ICentralTransport"/> that carries the seven site→central sends over gRPC — the
|
||||
/// migration target for the Akka <c>ClusterClient</c> path. Each method encodes the message with
|
||||
/// <see cref="CentralControlDtoMapper"/>, dials <c>CentralControlService</c> through the sticky
|
||||
/// <see cref="CentralChannelProvider"/>, and delivers the decoded reply (or a transient-failure
|
||||
/// signal) to the waiting Ask.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// <para>
|
||||
/// The actor handlers are synchronous; each method here kicks off the RPC as a detached task and
|
||||
/// <c>Tell</c>s the result to <paramref name="replyTo"/> when it completes — <c>IActorRef.Tell</c>
|
||||
/// is thread-safe, so the reply lands at the Ask exactly as central's ClusterClient reply did. On
|
||||
/// any non-OK status the reply is <see cref="Status.Failure"/>, which the S&F / audit / health
|
||||
/// layers already treat as transient.
|
||||
/// </para>
|
||||
/// <para>
|
||||
/// <b>Cross-node retry only on provably-unsent failures.</b> An <see cref="StatusCode.Unavailable"/>
|
||||
/// (connection refused / node not ready) flips the channel pair and retries once on the peer.
|
||||
/// A <see cref="StatusCode.DeadlineExceeded"/> is NEVER retried across nodes — a deploy / write /
|
||||
/// failover may already have executed, and duplicating it is worse than surfacing a transient
|
||||
/// failure the layer above tolerates.
|
||||
/// </para>
|
||||
/// <para>
|
||||
/// <b>Per-call deadlines mirror today's Ask timeouts.</b> Notification submit/status →
|
||||
/// <c>NotificationForwardTimeout</c> (30 s, the value the S&F forwarder and the central service
|
||||
/// both use); health → <c>HealthReportTimeout</c> (10 s); reconcile → <c>QueryTimeout</c> (30 s);
|
||||
/// both ingest RPCs → <see cref="SiteStreamGrpcServer.AuditIngestAskTimeout"/> (the one shared 30 s
|
||||
/// constant). The heartbeat, fire-and-forget with no server-side Ask, is merely bounded by
|
||||
/// <c>HealthReportTimeout</c> and its failures are swallowed.
|
||||
/// </para>
|
||||
/// </remarks>
|
||||
public sealed class GrpcCentralTransport : ICentralTransport
|
||||
{
|
||||
private readonly CentralChannelProvider _channels;
|
||||
private readonly CommunicationOptions _options;
|
||||
private readonly ILogger<GrpcCentralTransport> _logger;
|
||||
|
||||
/// <summary>Creates the transport over a channel pair.</summary>
|
||||
/// <param name="channels">The sticky central channel pair.</param>
|
||||
/// <param name="options">Communication options supplying the per-call deadlines.</param>
|
||||
/// <param name="logger">Logger for failover/fault diagnostics.</param>
|
||||
public GrpcCentralTransport(
|
||||
CentralChannelProvider channels,
|
||||
CommunicationOptions options,
|
||||
ILogger<GrpcCentralTransport> logger)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(channels);
|
||||
ArgumentNullException.ThrowIfNull(options);
|
||||
ArgumentNullException.ThrowIfNull(logger);
|
||||
|
||||
_channels = channels;
|
||||
_options = options;
|
||||
_logger = logger;
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public void SubmitNotification(NotificationSubmit message, IActorRef replyTo)
|
||||
{
|
||||
var dto = CentralControlDtoMapper.ToDto(message);
|
||||
Dispatch(replyTo, _options.NotificationForwardTimeout,
|
||||
(c, o) => c.SubmitNotificationAsync(dto, o),
|
||||
ack => CentralControlDtoMapper.FromDto(ack));
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public void QueryNotificationStatus(NotificationStatusQuery message, IActorRef replyTo)
|
||||
{
|
||||
var dto = CentralControlDtoMapper.ToDto(message);
|
||||
Dispatch(replyTo, _options.NotificationForwardTimeout,
|
||||
(c, o) => c.QueryNotificationStatusAsync(dto, o),
|
||||
response => CentralControlDtoMapper.FromDto(response));
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public void IngestAuditEvents(IngestAuditEventsCommand message, IActorRef replyTo)
|
||||
{
|
||||
var batch = CentralControlDtoMapper.ToDto(message);
|
||||
Dispatch(replyTo, SiteStreamGrpcServer.AuditIngestAskTimeout,
|
||||
(c, o) => c.IngestAuditEventsAsync(batch, o),
|
||||
ack => new IngestAuditEventsReply(CentralControlDtoMapper.FromIngestAck(ack)));
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public void IngestCachedTelemetry(IngestCachedTelemetryCommand message, IActorRef replyTo)
|
||||
{
|
||||
var batch = CentralControlDtoMapper.ToDto(message);
|
||||
Dispatch(replyTo, SiteStreamGrpcServer.AuditIngestAskTimeout,
|
||||
(c, o) => c.IngestCachedTelemetryAsync(batch, o),
|
||||
ack => new IngestCachedTelemetryReply(CentralControlDtoMapper.FromIngestAck(ack)));
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public void ReconcileSite(ReconcileSiteRequest message, IActorRef replyTo)
|
||||
{
|
||||
var dto = CentralControlDtoMapper.ToDto(message);
|
||||
Dispatch(replyTo, _options.QueryTimeout,
|
||||
(c, o) => c.ReconcileSiteAsync(dto, o),
|
||||
response => CentralControlDtoMapper.FromDto(response));
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public void ReportSiteHealth(SiteHealthReport message, IActorRef replyTo)
|
||||
{
|
||||
var dto = CentralControlDtoMapper.ToDto(message);
|
||||
Dispatch(replyTo, _options.HealthReportTimeout,
|
||||
(c, o) => c.ReportSiteHealthAsync(dto, o),
|
||||
ack => CentralControlDtoMapper.FromDto(ack));
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public void SendHeartbeat(HeartbeatMessage message, IActorRef self)
|
||||
{
|
||||
_ = SendHeartbeatAsync(message);
|
||||
}
|
||||
|
||||
private async Task SendHeartbeatAsync(HeartbeatMessage message)
|
||||
{
|
||||
var (index, client) = _channels.Current();
|
||||
var dto = CentralControlDtoMapper.ToDto(message);
|
||||
try
|
||||
{
|
||||
var options = new CallOptions(deadline: DateTime.UtcNow.Add(_options.HealthReportTimeout));
|
||||
using var call = client.HeartbeatAsync(dto, options);
|
||||
await call.ResponseAsync.ConfigureAwait(false);
|
||||
}
|
||||
catch (RpcException ex) when (IsConnectFailure(ex))
|
||||
{
|
||||
// Nudge the pair so the next call tries the peer, but never fault: a heartbeat
|
||||
// failure must not surface on the site's heartbeat timer path.
|
||||
_channels.ReportUnavailable(index);
|
||||
_logger.LogDebug(ex, "Heartbeat to central endpoint {Endpoint} was unavailable.", CurrentEndpointSafe());
|
||||
}
|
||||
catch (Exception ex)
|
||||
{
|
||||
_logger.LogDebug(ex, "Heartbeat to central failed (swallowed — heartbeats are fire-and-forget).");
|
||||
}
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Runs a unary RPC on the current channel, delivers the decoded reply to
|
||||
/// <paramref name="replyTo"/>, and applies the sticky-failover / no-retry-on-deadline policy.
|
||||
/// </summary>
|
||||
private void Dispatch<TWire>(
|
||||
IActorRef replyTo,
|
||||
TimeSpan timeout,
|
||||
Func<CentralControlService.CentralControlServiceClient, CallOptions, AsyncUnaryCall<TWire>> call,
|
||||
Func<TWire, object> decode)
|
||||
{
|
||||
_ = DispatchAsync(replyTo, timeout, call, decode);
|
||||
}
|
||||
|
||||
private async Task DispatchAsync<TWire>(
|
||||
IActorRef replyTo,
|
||||
TimeSpan timeout,
|
||||
Func<CentralControlService.CentralControlServiceClient, CallOptions, AsyncUnaryCall<TWire>> call,
|
||||
Func<TWire, object> decode)
|
||||
{
|
||||
var (index, client) = _channels.Current();
|
||||
try
|
||||
{
|
||||
var reply = await InvokeAsync(client, timeout, call).ConfigureAwait(false);
|
||||
replyTo.Tell(decode(reply));
|
||||
}
|
||||
catch (RpcException ex) when (IsConnectFailure(ex))
|
||||
{
|
||||
// Provably unsent: the connection was refused / the node was not ready. Fail over
|
||||
// to the peer and retry ONCE. This is the only status we retry across nodes.
|
||||
_channels.ReportUnavailable(index);
|
||||
var (retryIndex, retryClient) = _channels.Current();
|
||||
if (retryIndex != index)
|
||||
{
|
||||
try
|
||||
{
|
||||
var reply = await InvokeAsync(retryClient, timeout, call).ConfigureAwait(false);
|
||||
replyTo.Tell(decode(reply));
|
||||
return;
|
||||
}
|
||||
catch (Exception retryEx)
|
||||
{
|
||||
_logger.LogWarning(retryEx,
|
||||
"Central control-plane call failed on both endpoints; surfacing as transient.");
|
||||
replyTo.Tell(new AkkaStatus.Failure(retryEx));
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
replyTo.Tell(new AkkaStatus.Failure(ex));
|
||||
}
|
||||
catch (Exception ex)
|
||||
{
|
||||
// DeadlineExceeded / Internal / PermissionDenied / a PSK-resolution throw — do NOT
|
||||
// retry across nodes (the call may have run). Surface as the transient failure the
|
||||
// layer above already tolerates.
|
||||
replyTo.Tell(new AkkaStatus.Failure(ex));
|
||||
}
|
||||
}
|
||||
|
||||
private static async Task<TWire> InvokeAsync<TWire>(
|
||||
CentralControlService.CentralControlServiceClient client,
|
||||
TimeSpan timeout,
|
||||
Func<CentralControlService.CentralControlServiceClient, CallOptions, AsyncUnaryCall<TWire>> call)
|
||||
{
|
||||
var options = new CallOptions(deadline: DateTime.UtcNow.Add(timeout));
|
||||
using var asyncCall = call(client, options);
|
||||
return await asyncCall.ResponseAsync.ConfigureAwait(false);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// A failure that provably never reached a server — the only class safe to retry on the peer.
|
||||
/// <see cref="StatusCode.DeadlineExceeded"/> is deliberately excluded (the call may have run).
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Two shapes qualify: a server-signalled <see cref="StatusCode.Unavailable"/> (e.g. a node
|
||||
/// that returns Unavailable while it is still starting), and a client-side failure to even
|
||||
/// start the call — Grpc.Net surfaces a refused/failed connection as
|
||||
/// <see cref="StatusCode.Internal"/> "Error starting gRPC call" with the transport exception
|
||||
/// attached, and there the request never left the client. Anything else — including a deadline,
|
||||
/// a permission denial, or a generic server-side Internal after the call reached the server —
|
||||
/// is NOT retried across nodes.
|
||||
/// </remarks>
|
||||
private static bool IsConnectFailure(RpcException ex)
|
||||
{
|
||||
if (ex.StatusCode == StatusCode.Unavailable)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
// "Error starting gRPC call" is Grpc.Net's marker for a call that could not be sent — a
|
||||
// refused/failed connection carrying an HttpRequestException. Provably unsent.
|
||||
return ex.StatusCode == StatusCode.Internal
|
||||
&& (ex.Status.DebugException is HttpRequestException
|
||||
|| ex.Status.Detail.StartsWith("Error starting gRPC call", StringComparison.Ordinal));
|
||||
}
|
||||
|
||||
private string CurrentEndpointSafe()
|
||||
{
|
||||
try
|
||||
{
|
||||
return _channels.CurrentEndpoint;
|
||||
}
|
||||
catch
|
||||
{
|
||||
return "(unknown)";
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -21,15 +21,17 @@ namespace ZB.MOM.WW.ScadaBridge.Communication.Grpc;
|
||||
/// Mirrors the sibling <see cref="AuditEventDtoMapper"/>.
|
||||
/// </para>
|
||||
/// <para>
|
||||
/// Two directions are provided. <see cref="FromDto"/> rehydrates the central
|
||||
/// Three directions are provided. <see cref="FromDto"/> rehydrates the central
|
||||
/// <see cref="SiteCall"/> entity central writes into the <c>SiteCalls</c> table.
|
||||
/// <see cref="ToDto"/> projects a site-local <see cref="SiteCallOperational"/>
|
||||
/// onto the wire — used by the Site Call Audit <c>PullSiteCalls</c>
|
||||
/// reconciliation handler (the central→site self-heal pull). The
|
||||
/// <see cref="SiteCall"/> entity itself is never mapped back onto the wire:
|
||||
/// sites emit operational state from <see cref="SiteCallOperational"/>, never
|
||||
/// from the central <see cref="SiteCall"/>, so a <c>SiteCall</c>→DTO method
|
||||
/// would be dead code.
|
||||
/// <see cref="ToDto(SiteCallOperational)"/> projects a site-local
|
||||
/// <see cref="SiteCallOperational"/> onto the wire — used by the Site Call Audit
|
||||
/// <c>PullSiteCalls</c> reconciliation handler (the central→site self-heal pull).
|
||||
/// <see cref="ToDto(SiteCall)"/> projects the entity form back onto the wire; it
|
||||
/// exists for the gRPC central control plane, whose site-side transport receives
|
||||
/// an already-decoded <c>IngestCachedTelemetryCommand</c> (which carries
|
||||
/// <see cref="SiteCall"/>, not <see cref="SiteCallOperational"/>) and must
|
||||
/// re-encode it. It was previously documented here as necessarily dead code —
|
||||
/// true only while ClusterClient was the sole path from that command to central.
|
||||
/// </para>
|
||||
/// <para>
|
||||
/// String nullability convention: proto3 scalar strings cannot be absent, so the
|
||||
@@ -120,6 +122,51 @@ public static class SiteCallDtoMapper
|
||||
return dto;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Projects a <see cref="SiteCall"/> entity onto its wire-format DTO — the
|
||||
/// inverse of <see cref="FromDto"/>, so the pair round-trips exactly.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// <see cref="SiteCall.IngestedAtUtc"/> is deliberately NOT written: it is
|
||||
/// central-set inside the dual-write transaction and the value carried on the
|
||||
/// wire is informational only (<see cref="FromDto"/> stamps a placeholder that
|
||||
/// the ingest actor overwrites). Every other field survives; null
|
||||
/// <c>SourceNode</c>/<c>LastError</c> collapse to empty strings while the
|
||||
/// nullable <c>HttpStatus</c>/<c>TerminalAtUtc</c> stay unset on the wire.
|
||||
/// </remarks>
|
||||
/// <param name="siteCall">The central operational-state entity to project.</param>
|
||||
/// <returns>A populated <see cref="SiteCallOperationalDto"/> ready for transmission.</returns>
|
||||
public static SiteCallOperationalDto ToDto(SiteCall siteCall)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(siteCall);
|
||||
|
||||
var dto = new SiteCallOperationalDto
|
||||
{
|
||||
TrackedOperationId = siteCall.TrackedOperationId.ToString(),
|
||||
Channel = siteCall.Channel,
|
||||
Target = siteCall.Target,
|
||||
SourceSite = siteCall.SourceSite,
|
||||
SourceNode = siteCall.SourceNode ?? string.Empty,
|
||||
Status = siteCall.Status,
|
||||
RetryCount = siteCall.RetryCount,
|
||||
LastError = siteCall.LastError ?? string.Empty,
|
||||
CreatedAtUtc = Timestamp.FromDateTime(EnsureUtc(siteCall.CreatedAtUtc)),
|
||||
UpdatedAtUtc = Timestamp.FromDateTime(EnsureUtc(siteCall.UpdatedAtUtc)),
|
||||
};
|
||||
|
||||
if (siteCall.HttpStatus.HasValue)
|
||||
{
|
||||
dto.HttpStatus = siteCall.HttpStatus.Value;
|
||||
}
|
||||
|
||||
if (siteCall.TerminalAtUtc.HasValue)
|
||||
{
|
||||
dto.TerminalAtUtc = Timestamp.FromDateTime(EnsureUtc(siteCall.TerminalAtUtc.Value));
|
||||
}
|
||||
|
||||
return dto;
|
||||
}
|
||||
|
||||
// All ScadaBridge timestamps are UTC by invariant; Timestamp.FromDateTime
|
||||
// requires UTC kind. Specify (never convert) so a row read back from SQLite
|
||||
// with Kind=Utc passes through and a defensively-unspecified value is
|
||||
|
||||
@@ -0,0 +1,43 @@
|
||||
namespace ZB.MOM.WW.ScadaBridge.Communication.Grpc;
|
||||
|
||||
/// <summary>
|
||||
/// Site-side <see cref="ISitePskProvider"/> over a single fixed key — the one key a site node
|
||||
/// presents on every control-plane call it makes to central (<c>CommunicationOptions.GrpcPsk</c>).
|
||||
/// Central's provider resolves a key <em>per site</em>; a site has exactly one, so it ignores the
|
||||
/// requested <c>siteId</c> and returns its own key.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// <b>Fail-closed.</b> An empty key throws, matching the contract on <see cref="ISitePskProvider"/>
|
||||
/// and the interceptor's own posture: a node shipped without a key must not degrade to an
|
||||
/// unauthenticated dial.
|
||||
/// </remarks>
|
||||
public sealed class StaticSitePskProvider : ISitePskProvider
|
||||
{
|
||||
private readonly string _key;
|
||||
|
||||
/// <summary>Creates the provider bound to a site's own preshared key.</summary>
|
||||
/// <param name="key">The site's <c>GrpcPsk</c>. Empty is permitted at construction but throws on use.</param>
|
||||
public StaticSitePskProvider(string key)
|
||||
{
|
||||
_key = key ?? string.Empty;
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public ValueTask<string> GetAsync(string siteId, CancellationToken ct)
|
||||
{
|
||||
if (string.IsNullOrEmpty(_key))
|
||||
{
|
||||
throw new InvalidOperationException(
|
||||
"No gRPC preshared key is configured for this site (ScadaBridge:Communication:GrpcPsk). "
|
||||
+ "The control plane is fail-closed: an unauthenticated dial does not happen.");
|
||||
}
|
||||
|
||||
return new ValueTask<string>(_key);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public void Invalidate(string siteId)
|
||||
{
|
||||
// A single static key never changes for the process lifetime; nothing to drop.
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,247 @@
|
||||
syntax = "proto3";
|
||||
option csharp_namespace = "ZB.MOM.WW.ScadaBridge.Communication.Grpc";
|
||||
package scadabridge.centralcontrol.v1;
|
||||
|
||||
import "google/protobuf/empty.proto";
|
||||
import "google/protobuf/timestamp.proto";
|
||||
import "google/protobuf/wrappers.proto";
|
||||
|
||||
// The two ingest RPCs deliberately REUSE the batch/ack messages already defined
|
||||
// for the site-hosted SiteStreamService rather than redeclaring them. The site
|
||||
// telemetry actor builds AuditEventBatch / CachedTelemetryBatch today and hands
|
||||
// them to ISiteStreamAuditClient; duplicating the shapes here would fork one
|
||||
// wire contract into two that must be kept in lockstep by hand.
|
||||
import "Protos/sitestream.proto";
|
||||
|
||||
// Central-hosted control plane (Phase 1A of the ClusterClient→gRPC migration).
|
||||
//
|
||||
// Direction: SITE is the client, CENTRAL is the server — the inverse of
|
||||
// SiteStreamService, where central dials the site. That asymmetry is deliberate
|
||||
// and mirrors the direction the Akka ClusterClient traffic flows today: these
|
||||
// seven calls are exactly the seven messages SiteCommunicationActor sends to
|
||||
// /user/central-communication.
|
||||
//
|
||||
// Every call is gated by ControlPlaneAuthInterceptor: `authorization: Bearer <psk>`
|
||||
// plus the `x-scadabridge-site` metadata header naming which site's preshared key
|
||||
// central must verify against.
|
||||
service CentralControlService {
|
||||
// Store-and-forward handoff of one notification for central delivery. The
|
||||
// ack is idempotent on notification_id — a duplicate submit after a lost ack
|
||||
// must not produce a second delivery.
|
||||
rpc SubmitNotification(NotificationSubmitDto) returns (NotificationSubmitAckDto);
|
||||
|
||||
// Notify.Status(id) round-trip for a notification that has already left the
|
||||
// site buffer. `found = false` sends the caller back to the site-local buffer
|
||||
// to decide Forwarding vs Unknown.
|
||||
rpc QueryNotificationStatus(NotificationStatusQueryDto) returns (NotificationStatusResponseDto);
|
||||
|
||||
// Audit Log (#23) push telemetry. Reuses the SiteStreamService messages: the
|
||||
// batch a site drains from its SQLite hot path is byte-identical whichever
|
||||
// transport carries it.
|
||||
rpc IngestAuditEvents(sitestream.AuditEventBatch) returns (sitestream.IngestAck);
|
||||
|
||||
// Audit Log (#23) M3 combined cached-call telemetry (audit row + SiteCalls
|
||||
// operational upsert, written in one central transaction).
|
||||
rpc IngestCachedTelemetry(sitestream.CachedTelemetryBatch) returns (sitestream.IngestAck);
|
||||
|
||||
// Node-startup self-heal: the node's local deployed inventory in, fetch
|
||||
// tokens for whatever it is missing or stale out.
|
||||
rpc ReconcileSite(ReconcileSiteRequestDto) returns (ReconcileSiteResponseDto);
|
||||
|
||||
// Periodic site health report (30 s cadence). The ack makes delivery
|
||||
// observable end-to-end so the sender can restore its per-interval counters
|
||||
// when a report is lost.
|
||||
rpc ReportSiteHealth(SiteHealthReportDto) returns (SiteHealthReportAckDto);
|
||||
|
||||
// Application heartbeat. Returns Empty because the message is
|
||||
// fire-and-forget: nothing on the site consumes a reply, and a failure here
|
||||
// must never surface as a fault on the heartbeat timer path.
|
||||
rpc Heartbeat(HeartbeatDto) returns (google.protobuf.Empty);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Notification Outbox (#21)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// Site -> Central: submit a buffered notification for central delivery.
|
||||
// Mirrors Commons NotificationSubmit.
|
||||
message NotificationSubmitDto {
|
||||
string notification_id = 1; // GUID string, the idempotency key
|
||||
string list_name = 2;
|
||||
string subject = 3;
|
||||
string body = 4;
|
||||
string source_site_id = 5;
|
||||
string source_instance_id = 6; // empty string represents null
|
||||
string source_script = 7; // empty string represents null
|
||||
google.protobuf.Timestamp site_enqueued_at = 8;
|
||||
string origin_execution_id = 9; // GUID string; empty represents null
|
||||
string origin_parent_execution_id = 10; // GUID string; empty represents null
|
||||
string source_node = 11; // empty string represents null
|
||||
}
|
||||
|
||||
// Central -> Site: ack sent after the Notifications row is persisted.
|
||||
message NotificationSubmitAckDto {
|
||||
string notification_id = 1;
|
||||
bool accepted = 2;
|
||||
string error = 3; // empty string represents null
|
||||
}
|
||||
|
||||
// Site -> Central: Notify.Status(id) lookup against the central outbox.
|
||||
message NotificationStatusQueryDto {
|
||||
string correlation_id = 1;
|
||||
string notification_id = 2;
|
||||
}
|
||||
|
||||
// Central -> Site: current central delivery state for a queried notification.
|
||||
message NotificationStatusResponseDto {
|
||||
string correlation_id = 1;
|
||||
bool found = 2;
|
||||
string status = 3;
|
||||
int32 retry_count = 4;
|
||||
string last_error = 5; // empty string represents null
|
||||
google.protobuf.Timestamp delivered_at = 6; // absent when null
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Startup reconciliation (Deployment Manager)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// Site -> Central: the node's local deployed inventory at startup.
|
||||
message ReconcileSiteRequestDto {
|
||||
string site_identifier = 1;
|
||||
string node_id = 2;
|
||||
// Instance unique name -> revision hash of the config the node currently holds.
|
||||
map<string, string> local_name_to_revision_hash = 3;
|
||||
}
|
||||
|
||||
// Central -> Site: the gap the node must (re)fetch, plus orphans to log.
|
||||
message ReconcileSiteResponseDto {
|
||||
repeated ReconcileGapItemDto gap = 1;
|
||||
repeated string orphan_names = 2;
|
||||
string central_fetch_base_url = 3;
|
||||
}
|
||||
|
||||
// One instance the node must (re)fetch, with a freshly-minted short-TTL token.
|
||||
message ReconcileGapItemDto {
|
||||
string instance_unique_name = 1;
|
||||
string deployment_id = 2;
|
||||
string revision_hash = 3;
|
||||
bool is_enabled = 4;
|
||||
string fetch_token = 5;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Health Monitoring (#11)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// Wire form of the Commons ConnectionHealth enum.
|
||||
//
|
||||
// CONNECTION_HEALTH_UNSPECIFIED exists only to keep the proto3 zero value from
|
||||
// meaning something. Mapping Connected onto 0 would make an absent/garbled
|
||||
// value decode as "healthy", which is precisely the wrong direction to fail;
|
||||
// the mapper decodes UNSPECIFIED as Error instead and never emits it.
|
||||
enum ConnectionHealthEnum {
|
||||
CONNECTION_HEALTH_UNSPECIFIED = 0;
|
||||
CONNECTION_HEALTH_CONNECTED = 1;
|
||||
CONNECTION_HEALTH_DISCONNECTED = 2;
|
||||
CONNECTION_HEALTH_CONNECTING = 3;
|
||||
CONNECTION_HEALTH_ERROR = 4;
|
||||
}
|
||||
|
||||
message TagResolutionStatusDto {
|
||||
int32 total_subscribed = 1;
|
||||
int32 successfully_resolved = 2;
|
||||
}
|
||||
|
||||
message TagQualityCountsDto {
|
||||
int32 good = 1;
|
||||
int32 bad = 2;
|
||||
int32 uncertain = 3;
|
||||
}
|
||||
|
||||
message NodeStatusDto {
|
||||
string hostname = 1;
|
||||
bool is_online = 2;
|
||||
string role = 3;
|
||||
}
|
||||
|
||||
// Point-in-time snapshot of the site-local SQLite audit queue.
|
||||
message SiteAuditBacklogSnapshotDto {
|
||||
int32 pending_count = 1;
|
||||
google.protobuf.Timestamp oldest_pending_utc = 2; // absent when the queue is empty
|
||||
int64 on_disk_bytes = 3;
|
||||
}
|
||||
|
||||
// The three collection wrappers below exist so null and empty stay
|
||||
// distinguishable. proto3 cannot express presence on a `repeated` or `map`
|
||||
// field — an unset one and an empty one are the same bytes — but the
|
||||
// corresponding SiteHealthReport members are genuinely nullable
|
||||
// (SiteHealthCollector emits `ClusterNodes: _clusterNodes?.ToList()`), and the
|
||||
// central health surface reads null as "this producer doesn't report the
|
||||
// signal" rather than "the signal is empty". Wrapping in a message restores
|
||||
// message presence and makes the distinction survive the round-trip.
|
||||
|
||||
message ConnectionEndpointMapDto {
|
||||
map<string, string> entries = 1;
|
||||
}
|
||||
|
||||
message TagQualityMapDto {
|
||||
map<string, TagQualityCountsDto> entries = 1;
|
||||
}
|
||||
|
||||
message NodeStatusListDto {
|
||||
repeated NodeStatusDto nodes = 1;
|
||||
}
|
||||
|
||||
// Site -> Central: periodic site health report. Mirrors Commons SiteHealthReport.
|
||||
// Additive-only evolution: field numbers are never reused.
|
||||
message SiteHealthReportDto {
|
||||
string site_id = 1;
|
||||
int64 sequence_number = 2;
|
||||
google.protobuf.Timestamp report_timestamp = 3;
|
||||
map<string, ConnectionHealthEnum> data_connection_statuses = 4;
|
||||
map<string, TagResolutionStatusDto> tag_resolution_counts = 5;
|
||||
int32 script_error_count = 6;
|
||||
int32 alarm_evaluation_error_count = 7;
|
||||
map<string, int32> store_and_forward_buffer_depths = 8;
|
||||
int32 dead_letter_count = 9;
|
||||
int32 deployed_instance_count = 10;
|
||||
int32 enabled_instance_count = 11;
|
||||
int32 disabled_instance_count = 12;
|
||||
string node_role = 13;
|
||||
string node_hostname = 14;
|
||||
ConnectionEndpointMapDto data_connection_endpoints = 15; // absent when null
|
||||
TagQualityMapDto data_connection_tag_quality = 16; // absent when null
|
||||
int32 parked_message_count = 17;
|
||||
NodeStatusListDto cluster_nodes = 18; // absent when null
|
||||
int32 site_audit_write_failures = 19;
|
||||
int32 audit_redaction_failure = 20;
|
||||
SiteAuditBacklogSnapshotDto site_audit_backlog = 21; // absent when no data yet
|
||||
int64 site_event_log_write_failures = 22;
|
||||
google.protobuf.DoubleValue oldest_parked_message_age_seconds = 23; // absent when nothing parked
|
||||
int32 script_queue_depth = 24;
|
||||
int32 script_busy_threads = 25;
|
||||
google.protobuf.DoubleValue script_oldest_busy_age_seconds = 26; // absent when the pool is idle
|
||||
// Nullable on purpose: absent means "replication not wired on this node",
|
||||
// which is NOT the same as false ("wired but currently disconnected").
|
||||
google.protobuf.BoolValue local_db_replication_connected = 27;
|
||||
// Absent means UNKNOWN, never zero — a failed backlog read rendered as 0
|
||||
// would report a broken replication pair as perfectly healthy.
|
||||
google.protobuf.Int64Value local_db_oplog_backlog = 28;
|
||||
}
|
||||
|
||||
// Central -> Site: health report ack, so a lost report is observable.
|
||||
message SiteHealthReportAckDto {
|
||||
string site_id = 1;
|
||||
int64 sequence_number = 2;
|
||||
bool accepted = 3;
|
||||
string error = 4; // empty string represents null
|
||||
}
|
||||
|
||||
// Site -> Central: application heartbeat (fire-and-forget; reply is Empty).
|
||||
message HeartbeatDto {
|
||||
string site_id = 1;
|
||||
string node_hostname = 2;
|
||||
bool is_active = 3;
|
||||
google.protobuf.Timestamp timestamp = 4;
|
||||
}
|
||||
@@ -32,20 +32,30 @@
|
||||
<ProjectReference Include="../ZB.MOM.WW.ScadaBridge.HealthMonitoring/ZB.MOM.WW.ScadaBridge.HealthMonitoring.csproj" />
|
||||
</ItemGroup>
|
||||
|
||||
<!-- gRPC proto generation. The compiled C# is checked in under
|
||||
SiteStreamGrpc/ (Sitestream.cs + SitestreamGrpc.cs) because protoc
|
||||
segfaults inside our linux_arm64 Docker build image. To regenerate
|
||||
after schema changes:
|
||||
1. Temporarily uncomment the Protobuf ItemGroup below.
|
||||
2. Delete SiteStreamGrpc/*.cs.
|
||||
<!-- gRPC proto generation. The compiled C# is checked in — SiteStreamGrpc/
|
||||
(Sitestream.cs + SitestreamGrpc.cs) for Protos/sitestream.proto, and
|
||||
CentralControlGrpc/ (CentralControl.cs + CentralControlGrpc.cs) for
|
||||
Protos/central_control.proto — because protoc segfaults inside our
|
||||
linux_arm64 Docker build image. To regenerate after schema changes run
|
||||
`docker/regen-proto.sh [sitestream|centralcontrol|all]`, which does all
|
||||
of the following and always leaves this file as it found it:
|
||||
1. Temporarily uncomment the Protobuf ItemGroup below (just the line
|
||||
for the proto you changed — the other file's checked-in C# is
|
||||
already compiled, so enabling both at once duplicates types).
|
||||
2. Delete the matching checked-in *.cs.
|
||||
3. `dotnet build` (on macOS) — Grpc.Tools writes fresh files to obj/.
|
||||
4. Copy obj/Debug/net10.0/Protos/*.cs into SiteStreamGrpc/.
|
||||
4. Copy obj/Debug/net10.0/Protos/*.cs into the matching folder.
|
||||
5. Re-comment the ItemGroup.
|
||||
Eventually we should switch the Docker build image to one with a
|
||||
working protoc on arm64. -->
|
||||
central_control.proto imports sitestream.proto, so protoc resolves it
|
||||
from the project-relative path without sitestream.proto needing its own
|
||||
Protobuf item.
|
||||
An ACTIVE Protobuf item must never be committed — it breaks the Docker
|
||||
image build. Eventually we should switch the Docker build image to one
|
||||
with a working protoc on arm64. -->
|
||||
<!--
|
||||
<ItemGroup>
|
||||
<Protobuf Include="Protos\sitestream.proto" GrpcServices="Both" />
|
||||
<Protobuf Include="Protos\central_control.proto" GrpcServices="Both" />
|
||||
</ItemGroup>
|
||||
-->
|
||||
|
||||
|
||||
@@ -8,6 +8,7 @@ using Microsoft.Extensions.Options;
|
||||
using ZB.MOM.WW.ScadaBridge.ClusterInfrastructure;
|
||||
using ZB.MOM.WW.ScadaBridge.Communication;
|
||||
using ZB.MOM.WW.ScadaBridge.Communication.Actors;
|
||||
using ZB.MOM.WW.ScadaBridge.Communication.Grpc;
|
||||
using ZB.MOM.WW.ScadaBridge.Host.Actors;
|
||||
using ZB.MOM.WW.ScadaBridge.SiteRuntime;
|
||||
using ZB.MOM.WW.ScadaBridge.SiteRuntime.Actors;
|
||||
@@ -436,6 +437,19 @@ akka {{
|
||||
ClusterClientReceptionist.Get(_actorSystem).RegisterService(centralCommActor);
|
||||
_logger.LogInformation("CentralCommunicationActor registered with ClusterClientReceptionist");
|
||||
|
||||
// Hand the same actor to the central-hosted gRPC control plane (T1A.2) and open its
|
||||
// readiness gate — the gRPC face Asks this exact actor, so both transports resolve to
|
||||
// one handler implementation. Mirrors SiteStreamGrpcServer.SetReady on the site side:
|
||||
// the service is a DI singleton created before the actor system exists, so the actor
|
||||
// arrives here post-construction. Null on a host that did not register the service
|
||||
// (e.g. an in-process test harness), so the wiring is a guarded no-op there.
|
||||
var centralControlGrpc = _serviceProvider
|
||||
.GetService<ZB.MOM.WW.ScadaBridge.Communication.Grpc.CentralControlGrpcService>();
|
||||
centralControlGrpc?.SetReady(centralCommActor);
|
||||
_logger.LogInformation(
|
||||
"CentralControlGrpcService readiness set (service bound: {Bound})",
|
||||
centralControlGrpc is not null);
|
||||
|
||||
// Wire up the CommunicationService with the actor reference
|
||||
var commService = _serviceProvider.GetService<CommunicationService>();
|
||||
commService?.SetCommunicationActor(centralCommActor);
|
||||
@@ -819,13 +833,45 @@ akka {{
|
||||
_logger, role: siteRole);
|
||||
var dmProxy = dm.Proxy;
|
||||
|
||||
// Select the site→central transport behind the coexistence flag (default Akka
|
||||
// ClusterClient). When gRPC is chosen the site dials CentralControlService directly with
|
||||
// a sticky-failover channel pair, presenting its own preshared key; the ClusterClient
|
||||
// below is then not created at all.
|
||||
ICentralTransport? centralTransport = null;
|
||||
if (_communicationOptions.CentralTransport == CentralTransportMode.Grpc)
|
||||
{
|
||||
var loggerFactory = _serviceProvider.GetRequiredService<ILoggerFactory>();
|
||||
var channelProvider = new CentralChannelProvider(
|
||||
_communicationOptions.CentralGrpcEndpoints,
|
||||
new StaticSitePskProvider(_communicationOptions.GrpcPsk),
|
||||
_nodeOptions.SiteId!,
|
||||
_communicationOptions,
|
||||
loggerFactory.CreateLogger<CentralChannelProvider>());
|
||||
_trackedDisposables.Add(channelProvider);
|
||||
centralTransport = new GrpcCentralTransport(
|
||||
channelProvider,
|
||||
_communicationOptions,
|
||||
loggerFactory.CreateLogger<GrpcCentralTransport>());
|
||||
_logger.LogInformation(
|
||||
"Site→central transport: gRPC to {Count} central endpoint(s) for site {SiteId}",
|
||||
_communicationOptions.CentralGrpcEndpoints.Count, _nodeOptions.SiteId);
|
||||
}
|
||||
else
|
||||
{
|
||||
_logger.LogInformation(
|
||||
"Site→central transport: Akka ClusterClient (default) for site {SiteId}",
|
||||
_nodeOptions.SiteId);
|
||||
}
|
||||
|
||||
// Create SiteCommunicationActor for receiving messages from central
|
||||
var siteCommActor = _actorSystem.ActorOf(
|
||||
Props.Create(() => new SiteCommunicationActor(
|
||||
_nodeOptions.SiteId!,
|
||||
_communicationOptions,
|
||||
dmProxy,
|
||||
activeNodeCheck)),
|
||||
activeNodeCheck,
|
||||
null,
|
||||
centralTransport)),
|
||||
"site-communication");
|
||||
|
||||
// Register local handlers with SiteCommunicationActor
|
||||
@@ -944,8 +990,12 @@ akka {{
|
||||
"Site actors registered. DeploymentManager singleton scoped to role={SiteRole}, SiteCommunicationActor created.",
|
||||
siteRole);
|
||||
|
||||
// Create ClusterClient to central if contact points are configured
|
||||
if (_communicationOptions.CentralContactPoints.Count > 0)
|
||||
// Create ClusterClient to central if contact points are configured — but only on the Akka
|
||||
// transport. On the gRPC transport the SiteCommunicationActor already holds a
|
||||
// GrpcCentralTransport and never receives RegisterCentralClient, so a ClusterClient here
|
||||
// would be dead weight (and keep an unwanted cross-cluster Akka association alive).
|
||||
if (_communicationOptions.CentralTransport == CentralTransportMode.Akka
|
||||
&& _communicationOptions.CentralContactPoints.Count > 0)
|
||||
{
|
||||
var contacts = _communicationOptions.CentralContactPoints
|
||||
.Select(cp => ActorPath.Parse($"{cp}/system/receptionist"))
|
||||
|
||||
@@ -0,0 +1,245 @@
|
||||
using System.Security.Cryptography;
|
||||
using System.Text;
|
||||
using Grpc.Core;
|
||||
using Grpc.Core.Interceptors;
|
||||
using Microsoft.Extensions.Logging;
|
||||
using ZB.MOM.WW.ScadaBridge.Communication.Grpc;
|
||||
|
||||
namespace ZB.MOM.WW.ScadaBridge.Host;
|
||||
|
||||
/// <summary>
|
||||
/// Gates the central-hosted gRPC control plane (<c>CentralControlService</c>) with each site's
|
||||
/// preshared key. The sibling of <see cref="ControlPlaneAuthInterceptor"/>, but with the
|
||||
/// verification model inverted: a site checks one bearer token against its own single key, whereas
|
||||
/// central must check the presented token against the key belonging to the SITE that sent it.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// <para>
|
||||
/// <b>Why a separate class rather than a second constructor on
|
||||
/// <see cref="ControlPlaneAuthInterceptor"/>.</b> <c>Grpc.AspNetCore</c> registers a
|
||||
/// type-registered interceptor through <c>InterceptorRegistration.GetFactory()</c>, which throws
|
||||
/// <c>"Multiple constructors accepting all given argument types have been found"</c> the moment a
|
||||
/// second public constructor is applicable. That throw lands inside the pipeline on every call and
|
||||
/// surfaces as <c>Unknown / "Exception was thrown by handler"</c> — the node boots healthy and
|
||||
/// every gated call dies looking like a handler bug, with correct/wrong/no key all producing the
|
||||
/// identical error. It shipped once with a fully green suite and was caught only on the rig.
|
||||
/// Central's model genuinely differs (per-site key by header, not the site's one own-key), so it
|
||||
/// gets its own class with its own single public constructor rather than a variant ctor on the
|
||||
/// site interceptor. Both classes are pinned by a reflection test asserting exactly one public
|
||||
/// constructor.
|
||||
/// </para>
|
||||
/// <para>
|
||||
/// <b>Fail-closed on every branch.</b> A gated call is refused with
|
||||
/// <see cref="StatusCode.PermissionDenied"/> when: the <c>x-scadabridge-site</c> header is
|
||||
/// missing or blank; no key can be resolved for that site (<see cref="ISitePskProvider"/> throws);
|
||||
/// or the presented bearer token does not match. There is no pass-through — an unresolvable or
|
||||
/// absent identity never degrades to "let it in". Non-gated services (should any share the
|
||||
/// listener) return immediately, matching the site interceptor's shape.
|
||||
/// </para>
|
||||
/// <para>
|
||||
/// The comparison is <see cref="CryptographicOperations.FixedTimeEquals"/> over UTF-8, the same
|
||||
/// constant-time compare the site interceptor and LocalDb sync use.
|
||||
/// </para>
|
||||
/// </remarks>
|
||||
public sealed class CentralControlAuthInterceptor : Interceptor
|
||||
{
|
||||
/// <summary>
|
||||
/// Service prefixes gated by default — the one central-hosted control-plane service. Taken
|
||||
/// from the generated <c>package scadabridge.centralcontrol.v1; service CentralControlService</c>.
|
||||
/// </summary>
|
||||
public static readonly IReadOnlyList<string> DefaultGatedPrefixes =
|
||||
new[] { $"/{CentralControlService.Descriptor.FullName}/" };
|
||||
|
||||
private readonly IReadOnlyList<string> _gatedPrefixes;
|
||||
private readonly ISitePskProvider _pskProvider;
|
||||
private readonly ILogger<CentralControlAuthInterceptor> _logger;
|
||||
|
||||
/// <summary>
|
||||
/// Creates the interceptor gating <see cref="DefaultGatedPrefixes"/>.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// <b>This must remain the ONLY public constructor</b> — see the class remarks for why a
|
||||
/// second one silently disables the gate. Pinned by
|
||||
/// <c>CentralControlAuthInterceptorTests.TheInterceptorHasExactlyOnePublicConstructor</c>.
|
||||
/// </remarks>
|
||||
/// <param name="pskProvider">Resolves each site's preshared key.</param>
|
||||
/// <param name="logger">Logger for denial diagnostics.</param>
|
||||
public CentralControlAuthInterceptor(
|
||||
ISitePskProvider pskProvider,
|
||||
ILogger<CentralControlAuthInterceptor> logger)
|
||||
: this(pskProvider, logger, DefaultGatedPrefixes)
|
||||
{
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Creates the interceptor gating an explicit prefix set. <b>Internal</b> — a public second
|
||||
/// constructor would reintroduce the ambiguous-constructor defect described on the class.
|
||||
/// </summary>
|
||||
/// <param name="pskProvider">Resolves each site's preshared key.</param>
|
||||
/// <param name="logger">Logger for denial diagnostics.</param>
|
||||
/// <param name="gatedPrefixes">Method-path prefixes to gate.</param>
|
||||
internal CentralControlAuthInterceptor(
|
||||
ISitePskProvider pskProvider,
|
||||
ILogger<CentralControlAuthInterceptor> logger,
|
||||
IReadOnlyList<string> gatedPrefixes)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(pskProvider);
|
||||
ArgumentNullException.ThrowIfNull(logger);
|
||||
ArgumentNullException.ThrowIfNull(gatedPrefixes);
|
||||
|
||||
_pskProvider = pskProvider;
|
||||
_logger = logger;
|
||||
_gatedPrefixes = gatedPrefixes;
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public override async Task<TResponse> UnaryServerHandler<TRequest, TResponse>(
|
||||
TRequest request,
|
||||
ServerCallContext context,
|
||||
UnaryServerMethod<TRequest, TResponse> continuation)
|
||||
{
|
||||
await AuthorizeAsync(context).ConfigureAwait(false);
|
||||
return await continuation(request, context).ConfigureAwait(false);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public override async Task DuplexStreamingServerHandler<TRequest, TResponse>(
|
||||
IAsyncStreamReader<TRequest> requestStream,
|
||||
IServerStreamWriter<TResponse> responseStream,
|
||||
ServerCallContext context,
|
||||
DuplexStreamingServerMethod<TRequest, TResponse> continuation)
|
||||
{
|
||||
await AuthorizeAsync(context).ConfigureAwait(false);
|
||||
await continuation(requestStream, responseStream, context).ConfigureAwait(false);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public override async Task<TResponse> ClientStreamingServerHandler<TRequest, TResponse>(
|
||||
IAsyncStreamReader<TRequest> requestStream,
|
||||
ServerCallContext context,
|
||||
ClientStreamingServerMethod<TRequest, TResponse> continuation)
|
||||
{
|
||||
await AuthorizeAsync(context).ConfigureAwait(false);
|
||||
return await continuation(requestStream, context).ConfigureAwait(false);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public override async Task ServerStreamingServerHandler<TRequest, TResponse>(
|
||||
TRequest request,
|
||||
IServerStreamWriter<TResponse> responseStream,
|
||||
ServerCallContext context,
|
||||
ServerStreamingServerMethod<TRequest, TResponse> continuation)
|
||||
{
|
||||
await AuthorizeAsync(context).ConfigureAwait(false);
|
||||
await continuation(request, responseStream, context).ConfigureAwait(false);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Throws <see cref="RpcException"/> with <see cref="StatusCode.PermissionDenied"/> unless a
|
||||
/// gated call carries a valid <c>x-scadabridge-site</c> header AND a bearer token matching
|
||||
/// that site's resolved key. Non-gated calls return immediately.
|
||||
/// </summary>
|
||||
private async Task AuthorizeAsync(ServerCallContext context)
|
||||
{
|
||||
if (!IsGated(context.Method))
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
var siteId = ExtractSiteId(context.RequestHeaders);
|
||||
if (string.IsNullOrWhiteSpace(siteId))
|
||||
{
|
||||
_logger.LogWarning(
|
||||
"Rejected a central control-plane call to {Method}: the required "
|
||||
+ "'{Header}' metadata header is missing or blank, so there is no per-site key to "
|
||||
+ "verify against.",
|
||||
context.Method, ControlPlaneCredentials.SiteHeader);
|
||||
throw Denied("missing site identity header");
|
||||
}
|
||||
|
||||
string expected;
|
||||
try
|
||||
{
|
||||
expected = await _pskProvider.GetAsync(siteId, context.CancellationToken)
|
||||
.ConfigureAwait(false);
|
||||
}
|
||||
catch (OperationCanceledException)
|
||||
{
|
||||
throw;
|
||||
}
|
||||
catch (Exception ex)
|
||||
{
|
||||
// Fail-closed: an unresolvable key is a denial, never a pass-through. The provider
|
||||
// has already logged the specific cause (missing secret / missing config entry).
|
||||
_logger.LogWarning(ex,
|
||||
"Rejected a central control-plane call to {Method}: no preshared key could be "
|
||||
+ "resolved for site {SiteId}.",
|
||||
context.Method, siteId);
|
||||
throw Denied("no key configured for the presented site");
|
||||
}
|
||||
|
||||
var presented = ExtractBearerToken(context.RequestHeaders);
|
||||
if (presented is null || !FixedTimeEquals(presented, expected))
|
||||
{
|
||||
_logger.LogWarning(
|
||||
"Rejected a central control-plane call to {Method} from site {SiteId}: {Reason}.",
|
||||
context.Method, siteId,
|
||||
presented is null ? "no bearer token presented" : "bearer token did not match");
|
||||
throw Denied("control plane authentication failed");
|
||||
}
|
||||
}
|
||||
|
||||
private static RpcException Denied(string reason)
|
||||
=> new(new Status(StatusCode.PermissionDenied, $"Control plane authentication failed: {reason}."));
|
||||
|
||||
private bool IsGated(string method)
|
||||
{
|
||||
foreach (var prefix in _gatedPrefixes)
|
||||
{
|
||||
if (method.StartsWith(prefix, StringComparison.Ordinal))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
private static string? ExtractSiteId(Metadata headers)
|
||||
{
|
||||
foreach (var entry in headers)
|
||||
{
|
||||
if (string.Equals(entry.Key, ControlPlaneCredentials.SiteHeader,
|
||||
StringComparison.OrdinalIgnoreCase))
|
||||
{
|
||||
return entry.Value;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
private static string? ExtractBearerToken(Metadata headers)
|
||||
{
|
||||
// gRPC lowercases header keys on the wire; compare case-insensitively so a hand-built
|
||||
// Metadata in a test behaves the same as a real request.
|
||||
foreach (var entry in headers)
|
||||
{
|
||||
if (!string.Equals(entry.Key, ControlPlaneCredentials.AuthorizationHeader,
|
||||
StringComparison.OrdinalIgnoreCase))
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
var value = entry.Value;
|
||||
if (value is not null && value.StartsWith("Bearer ", StringComparison.OrdinalIgnoreCase))
|
||||
{
|
||||
return value["Bearer ".Length..];
|
||||
}
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
private static bool FixedTimeEquals(string presented, string expected)
|
||||
=> CryptographicOperations.FixedTimeEquals(
|
||||
Encoding.UTF8.GetBytes(presented), Encoding.UTF8.GetBytes(expected));
|
||||
}
|
||||
@@ -21,6 +21,15 @@ public class NodeOptions
|
||||
/// <summary>Gets or sets the gRPC port for the site stream server.</summary>
|
||||
public int GrpcPort { get; set; } = 8083;
|
||||
/// <summary>
|
||||
/// HTTP/2 (h2c) port the CENTRAL node listens on for the site→central
|
||||
/// <c>CentralControlService</c> gRPC control plane. Default 8083 — deliberately symmetric
|
||||
/// with the site <see cref="GrpcPort"/>, since a node is either central or a site and the
|
||||
/// two never share a process. This listener is distinct from central's <c>:5000</c> HTTP/1
|
||||
/// surface (Central UI, Management/Inbound API), which stays exactly as-is: gRPC does NOT go
|
||||
/// through Traefik (HTTP/1 only). Ignored on site nodes.
|
||||
/// </summary>
|
||||
public int CentralGrpcPort { get; set; } = 8083;
|
||||
/// <summary>
|
||||
/// HTTP/1.1 port serving the Prometheus /metrics scrape endpoint on site nodes.
|
||||
/// Defaults to 8084 — deliberately distinct from <see cref="RemotingPort"/> (8082)
|
||||
/// and <see cref="GrpcPort"/> (8083) so the Kestrel metrics listener never contends
|
||||
|
||||
@@ -23,6 +23,7 @@ public sealed class NodeOptionsValidator : OptionsValidatorBase<NodeOptions>
|
||||
RequirePort(builder, options.RemotingPort, nameof(NodeOptions.RemotingPort));
|
||||
RequirePort(builder, options.GrpcPort, nameof(NodeOptions.GrpcPort));
|
||||
RequirePort(builder, options.MetricsPort, nameof(NodeOptions.MetricsPort));
|
||||
RequirePort(builder, options.CentralGrpcPort, nameof(NodeOptions.CentralGrpcPort));
|
||||
}
|
||||
|
||||
// 0 stays valid (dynamic-port request); reject only out-of-TCP-range values.
|
||||
|
||||
@@ -97,6 +97,49 @@ try
|
||||
// Windows Service support (no-op when not running as a Windows Service)
|
||||
builder.Host.UseWindowsService();
|
||||
|
||||
// Explicit Kestrel h2c listener for the central-hosted gRPC control plane
|
||||
// (CentralControlService): HTTP/2-only, on its own port (default 8083, symmetric
|
||||
// with the site GrpcPort — a node is either central or a site, never both). gRPC
|
||||
// does NOT go through Traefik (HTTP/1 only); sites reach this port by container name.
|
||||
//
|
||||
// WARNING — the trap the rig caught (T1A.2 shipped it, fixed here): calling
|
||||
// options.Listen*/ListenAnyIP puts Kestrel into EXPLICIT-ENDPOINTS mode, which
|
||||
// SUPPRESSES the URLs from ASPNETCORE_URLS/--urls entirely — it is NOT additive.
|
||||
// Central's ENTIRE HTTP/1 surface (Central UI, Management + Inbound API, and the
|
||||
// /health/* endpoints Traefik + IActiveNodeGate depend on) lives on that URL
|
||||
// (http://+:5000 on the rig, a different port in production). If we bind only the
|
||||
// gRPC port, :5000 vanishes and central is reachable over gRPC while the UI, the
|
||||
// management API and health checks are all dead — with no startup error. Unit tests
|
||||
// use TestServer and never bind real Kestrel, so only a live node exposes this.
|
||||
// The site branch has the same ConfigureKestrel shape but no ASPNETCORE_URLS surface
|
||||
// to lose (it binds every port it needs — gRPC + metrics — explicitly). Central must
|
||||
// therefore RE-BIND its HTTP port(s) here alongside the gRPC port.
|
||||
var centralGrpcPort = configuration.GetValue<int>("ScadaBridge:Node:CentralGrpcPort", 8083);
|
||||
// "urls" is WebHostDefaults.ServerUrlsKey — the host setting ASPNETCORE_URLS/--urls
|
||||
// populate. Read it as a literal so this needs no extra Hosting using.
|
||||
var httpUrls = configuration["ASPNETCORE_URLS"]
|
||||
?? builder.WebHost.GetSetting("urls")
|
||||
?? "http://+:5000";
|
||||
var httpPorts = ParseHttpBindPorts(httpUrls);
|
||||
builder.WebHost.ConfigureKestrel(options =>
|
||||
{
|
||||
// The HTTP/1.1 (+ HTTP/2) surface from the configured URLs, re-declared so it
|
||||
// survives the explicit-endpoints switch above.
|
||||
foreach (var httpPort in httpPorts)
|
||||
{
|
||||
options.ListenAnyIP(httpPort, listenOptions =>
|
||||
{
|
||||
listenOptions.Protocols = Microsoft.AspNetCore.Server.Kestrel.Core.HttpProtocols.Http1AndHttp2;
|
||||
});
|
||||
}
|
||||
|
||||
// The gRPC control plane, HTTP/2 h2c only, on its own port.
|
||||
options.ListenAnyIP(centralGrpcPort, listenOptions =>
|
||||
{
|
||||
listenOptions.Protocols = Microsoft.AspNetCore.Server.Kestrel.Core.HttpProtocols.Http2;
|
||||
});
|
||||
});
|
||||
|
||||
// Shared components
|
||||
builder.Services.AddClusterInfrastructure();
|
||||
builder.Services.AddCommunication();
|
||||
@@ -108,6 +151,22 @@ try
|
||||
// node uses for its single key). Registered before the clients that consume it.
|
||||
builder.Services.AddSingleton<
|
||||
ZB.MOM.WW.ScadaBridge.Communication.Grpc.ISitePskProvider, SitePskProvider>();
|
||||
|
||||
// Central-hosted gRPC control plane (T1A.2). CentralControlAuthInterceptor is the
|
||||
// per-site sibling of the site's ControlPlaneAuthInterceptor: it verifies the Bearer
|
||||
// token against the key for the site named in the required x-scadabridge-site header,
|
||||
// resolved through the ISitePskProvider registered just above. Registered BY TYPE on
|
||||
// AddGrpc exactly as the site branch registers its interceptors — NOT as a DI singleton,
|
||||
// which would let DI hand the instance back and bypass Grpc.AspNetCore's own activation
|
||||
// path (the shape that once hid a two-public-constructor defect until the rig caught it).
|
||||
// CentralControlGrpcService decodes each request onto the same in-process message the
|
||||
// ClusterClient path carries and Asks CentralCommunicationActor — zero handler logic here.
|
||||
builder.Services.AddGrpc(options =>
|
||||
{
|
||||
options.Interceptors.Add<CentralControlAuthInterceptor>();
|
||||
});
|
||||
builder.Services.AddSingleton<
|
||||
ZB.MOM.WW.ScadaBridge.Communication.Grpc.CentralControlGrpcService>();
|
||||
builder.Services.AddHealthMonitoring();
|
||||
builder.Services.AddCentralHealthAggregation();
|
||||
builder.Services.AddExternalSystemGateway();
|
||||
@@ -462,6 +521,14 @@ try
|
||||
// Requires endpoint routing (app.UseRouting() above).
|
||||
app.MapZbMetrics();
|
||||
|
||||
// Central-hosted gRPC control plane (T1A.2) — the site→central CentralControlService.
|
||||
// Runs on the dedicated h2c listener configured above (default :8083), gated by
|
||||
// CentralControlAuthInterceptor and readiness-gated by the service itself (Unavailable
|
||||
// until AkkaHostedService hands the CentralCommunicationActor over via SetReady). It
|
||||
// shares the app's endpoint routing with the HTTP/1 surface; Kestrel steers each
|
||||
// connection to the right pipeline by listener/protocol.
|
||||
app.MapGrpcService<ZB.MOM.WW.ScadaBridge.Communication.Grpc.CentralControlGrpcService>();
|
||||
|
||||
app.MapStaticAssets();
|
||||
app.MapCentralUI<ZB.MOM.WW.ScadaBridge.Host.Components.App>();
|
||||
app.MapInboundAPI();
|
||||
@@ -603,4 +670,56 @@ finally
|
||||
/// <summary>
|
||||
/// Exposes the auto-generated Program class for test infrastructure (e.g. WebApplicationFactory).
|
||||
/// </summary>
|
||||
public partial class Program { }
|
||||
public partial class Program
|
||||
{
|
||||
/// <summary>
|
||||
/// Extracts the distinct TCP ports from an ASP.NET Core server-URLs string (the
|
||||
/// <c>ASPNETCORE_URLS</c> / <c>--urls</c> value, e.g. <c>"http://+:5000"</c> or a
|
||||
/// semicolon-separated list). Used to re-declare central's HTTP surface after the gRPC
|
||||
/// listener switches Kestrel into explicit-endpoints mode — see the call site's warning.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Bind hosts (<c>+</c>, <c>*</c>, <c>0.0.0.0</c>, <c>[::]</c>, a hostname) are irrelevant
|
||||
/// here because the caller re-binds via <c>ListenAnyIP</c>; only the port matters. A URL
|
||||
/// with no explicit port falls back to the scheme default (80/443). Unparseable entries
|
||||
/// are skipped rather than throwing — a bad URL should not take the node down at boot.
|
||||
/// </remarks>
|
||||
/// <param name="serverUrls">The server-URLs string; may be null/empty.</param>
|
||||
/// <returns>The distinct ports, in first-seen order.</returns>
|
||||
internal static IReadOnlyList<int> ParseHttpBindPorts(string? serverUrls)
|
||||
{
|
||||
var ports = new List<int>();
|
||||
if (string.IsNullOrWhiteSpace(serverUrls))
|
||||
{
|
||||
return ports;
|
||||
}
|
||||
|
||||
foreach (var raw in serverUrls.Split(';', StringSplitOptions.RemoveEmptyEntries | StringSplitOptions.TrimEntries))
|
||||
{
|
||||
int port;
|
||||
// Uri can't parse the wildcard hosts Kestrel accepts (+, *), so normalize them
|
||||
// to a placeholder host before parsing; the host is discarded anyway.
|
||||
var normalized = raw.Replace("://+", "://placeholder").Replace("://*", "://placeholder");
|
||||
if (Uri.TryCreate(normalized, UriKind.Absolute, out var uri))
|
||||
{
|
||||
port = uri.Port; // Uri fills the scheme default (80/443) when none is given.
|
||||
}
|
||||
else
|
||||
{
|
||||
// Last-ditch: pull the port after the final ':' (handles odd inputs Uri rejects).
|
||||
var colon = raw.LastIndexOf(':');
|
||||
if (colon < 0 || !int.TryParse(raw.AsSpan(colon + 1), out port))
|
||||
{
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
if (port is > 0 and <= 65535 && !ports.Contains(port))
|
||||
{
|
||||
ports.Add(port);
|
||||
}
|
||||
}
|
||||
|
||||
return ports;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,67 @@
|
||||
using Akka.Actor;
|
||||
using Akka.Cluster.Tools.Client;
|
||||
using Akka.TestKit.Xunit2;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Notification;
|
||||
using ZB.MOM.WW.ScadaBridge.Communication.Actors;
|
||||
|
||||
namespace ZB.MOM.WW.ScadaBridge.Communication.Tests.Actors;
|
||||
|
||||
/// <summary>
|
||||
/// T1A.3: the regression-prone bit of the Akka transport — that a forwarded
|
||||
/// <see cref="ClusterClient.Send"/> carries the caller's sender, so central's reply routes
|
||||
/// straight back to the waiting Ask rather than to the site communication actor — plus the
|
||||
/// no-ClusterClient-yet fallback that keeps the S&F layer treating the send as transient.
|
||||
/// </summary>
|
||||
public class AkkaCentralTransportTests : TestKit
|
||||
{
|
||||
[Fact]
|
||||
public void SubmitNotification_ForwardsToClusterClient_WithReplyToAsSender()
|
||||
{
|
||||
var transport = new AkkaCentralTransport();
|
||||
var clusterClient = CreateTestProbe();
|
||||
var replyTo = CreateTestProbe();
|
||||
transport.SetCentralClient(clusterClient.Ref);
|
||||
|
||||
var submit = new NotificationSubmit(
|
||||
"notif-1", "Operators", "S", "B", "site1", null, null, DateTimeOffset.UtcNow);
|
||||
transport.SubmitNotification(submit, replyTo.Ref);
|
||||
|
||||
// The ClusterClient receives a Send addressed to the central actor...
|
||||
var send = clusterClient.ExpectMsg<ClusterClient.Send>();
|
||||
Assert.Equal("/user/central-communication", send.Path);
|
||||
Assert.IsType<NotificationSubmit>(send.Message);
|
||||
|
||||
// ...and replying to it lands at replyTo, proving the sender was forwarded (not the
|
||||
// transport / actor). This is the routing the waiting Ask relies on.
|
||||
clusterClient.Reply(new NotificationSubmitAck("notif-1", Accepted: true, Error: null));
|
||||
replyTo.ExpectMsg<NotificationSubmitAck>(ack => ack.NotificationId == "notif-1" && ack.Accepted);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void SubmitNotification_WithNoClusterClient_RepliesNonAcceptedToReplyTo()
|
||||
{
|
||||
var transport = new AkkaCentralTransport();
|
||||
var replyTo = CreateTestProbe();
|
||||
|
||||
transport.SubmitNotification(
|
||||
new NotificationSubmit("notif-2", "Operators", "S", "B", "site1", null, null, DateTimeOffset.UtcNow),
|
||||
replyTo.Ref);
|
||||
|
||||
replyTo.ExpectMsg<NotificationSubmitAck>(ack => ack.NotificationId == "notif-2" && !ack.Accepted);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void IngestAuditEvents_WithNoClusterClient_FaultsTheReplyTo()
|
||||
{
|
||||
// The audit drain treats a faulted Ask as transient and keeps rows Pending — so the
|
||||
// no-client path must be a Status.Failure, not a silent drop.
|
||||
var transport = new AkkaCentralTransport();
|
||||
var replyTo = CreateTestProbe();
|
||||
|
||||
transport.IngestAuditEvents(
|
||||
new Commons.Messages.Audit.IngestAuditEventsCommand(new List<ZB.MOM.WW.Audit.AuditEvent>()),
|
||||
replyTo.Ref);
|
||||
|
||||
replyTo.ExpectMsg<Status.Failure>();
|
||||
}
|
||||
}
|
||||
+166
@@ -0,0 +1,166 @@
|
||||
using Akka.Actor;
|
||||
using Akka.TestKit.Xunit2;
|
||||
using NSubstitute;
|
||||
using ZB.MOM.WW.Audit;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Audit;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Deployment;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Health;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Notification;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Types.Enums;
|
||||
using ZB.MOM.WW.ScadaBridge.Communication.Actors;
|
||||
|
||||
namespace ZB.MOM.WW.ScadaBridge.Communication.Tests.Actors;
|
||||
|
||||
/// <summary>
|
||||
/// T1A.3: the site communication actor delegates each of the seven site→central sends to the
|
||||
/// injected <see cref="ICentralTransport"/>, preserving the current <c>Sender</c> as the reply
|
||||
/// target — and a transport failure surfaces to that sender exactly as the S&F / audit / health
|
||||
/// layers already expect, while a heartbeat transport fault never faults the actor.
|
||||
/// </summary>
|
||||
public class SiteCommunicationActorTransportTests : TestKit
|
||||
{
|
||||
private readonly CommunicationOptions _options = new();
|
||||
|
||||
private (IActorRef actor, ICentralTransport transport) NewActor()
|
||||
{
|
||||
var transport = Substitute.For<ICentralTransport>();
|
||||
var dmProbe = CreateTestProbe();
|
||||
var actor = Sys.ActorOf(Props.Create(() =>
|
||||
new SiteCommunicationActor("site1", _options, dmProbe.Ref, () => false, null, transport)));
|
||||
return (actor, transport);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void NotificationSubmit_DelegatesToTransport_WithSenderAsReplyTo()
|
||||
{
|
||||
var (actor, transport) = NewActor();
|
||||
var submit = new NotificationSubmit(
|
||||
"notif-1", "Operators", "Subj", "Body", "site1", "inst1", "alarmScript", DateTimeOffset.UtcNow);
|
||||
|
||||
actor.Tell(submit, TestActor);
|
||||
|
||||
AwaitAssert(() => transport.Received(1).SubmitNotification(
|
||||
Arg.Is<NotificationSubmit>(m => m.NotificationId == "notif-1"), Arg.Is(TestActor)));
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void NotificationStatusQuery_DelegatesToTransport_WithSenderAsReplyTo()
|
||||
{
|
||||
var (actor, transport) = NewActor();
|
||||
|
||||
actor.Tell(new NotificationStatusQuery("corr-1", "notif-1"), TestActor);
|
||||
|
||||
AwaitAssert(() => transport.Received(1).QueryNotificationStatus(
|
||||
Arg.Is<NotificationStatusQuery>(m => m.NotificationId == "notif-1"), Arg.Is(TestActor)));
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void IngestAuditEvents_DelegatesToTransport_WithSenderAsReplyTo()
|
||||
{
|
||||
var (actor, transport) = NewActor();
|
||||
|
||||
actor.Tell(new IngestAuditEventsCommand(new List<AuditEvent>()), TestActor);
|
||||
|
||||
AwaitAssert(() => transport.Received(1).IngestAuditEvents(Arg.Any<IngestAuditEventsCommand>(), Arg.Is(TestActor)));
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void IngestCachedTelemetry_DelegatesToTransport_WithSenderAsReplyTo()
|
||||
{
|
||||
var (actor, transport) = NewActor();
|
||||
|
||||
actor.Tell(new IngestCachedTelemetryCommand(new List<CachedTelemetryEntry>()), TestActor);
|
||||
|
||||
AwaitAssert(() => transport.Received(1).IngestCachedTelemetry(Arg.Any<IngestCachedTelemetryCommand>(), Arg.Is(TestActor)));
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void ReconcileSite_DelegatesToTransport_WithSenderAsReplyTo()
|
||||
{
|
||||
var (actor, transport) = NewActor();
|
||||
|
||||
actor.Tell(
|
||||
new ReconcileSiteRequest("site1", "node-a", new Dictionary<string, string>()), TestActor);
|
||||
|
||||
AwaitAssert(() => transport.Received(1).ReconcileSite(
|
||||
Arg.Is<ReconcileSiteRequest>(m => m.NodeId == "node-a"), Arg.Is(TestActor)));
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void ReportSiteHealth_DelegatesToTransport_WithSenderAsReplyTo()
|
||||
{
|
||||
var (actor, transport) = NewActor();
|
||||
var report = MinimalHealthReport(sequence: 7);
|
||||
|
||||
actor.Tell(report, TestActor);
|
||||
|
||||
AwaitAssert(() => transport.Received(1).ReportSiteHealth(
|
||||
Arg.Is<SiteHealthReport>(m => m.SequenceNumber == 7), Arg.Is(TestActor)));
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void TransportFailureReply_RoutesBackToTheWaitingSender()
|
||||
{
|
||||
// The seam preserves the reply-routing the S&F layer depends on: when the transport
|
||||
// answers the captured replyTo with a Status.Failure (its transient-failure signal on a
|
||||
// non-OK status), that failure reaches the original sender — here the test actor — so the
|
||||
// waiting Ask faults, exactly as it did on the ClusterClient path.
|
||||
var transport = Substitute.For<ICentralTransport>();
|
||||
transport
|
||||
.When(t => t.SubmitNotification(Arg.Any<NotificationSubmit>(), Arg.Any<IActorRef>()))
|
||||
.Do(ci => ci.Arg<IActorRef>().Tell(
|
||||
new Status.Failure(new InvalidOperationException("central unavailable"))));
|
||||
|
||||
var dmProbe = CreateTestProbe();
|
||||
var actor = Sys.ActorOf(Props.Create(() =>
|
||||
new SiteCommunicationActor("site1", _options, dmProbe.Ref, () => false, null, transport)));
|
||||
|
||||
actor.Tell(new NotificationSubmit(
|
||||
"notif-x", "Operators", "S", "B", "site1", null, null, DateTimeOffset.UtcNow), TestActor);
|
||||
|
||||
var failure = ExpectMsg<Status.Failure>();
|
||||
Assert.IsType<InvalidOperationException>(failure.Cause);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void HeartbeatTransportThrow_DoesNotFaultTheActor()
|
||||
{
|
||||
// A transport whose SendHeartbeat throws must not fault the actor — heartbeats are
|
||||
// fire-and-forget and their failure is swallowed. Prove the actor still serves messages
|
||||
// after a heartbeat that threw.
|
||||
var transport = Substitute.For<ICentralTransport>();
|
||||
transport
|
||||
.When(t => t.SendHeartbeat(Arg.Any<HeartbeatMessage>(), Arg.Any<IActorRef>()))
|
||||
.Do(_ => throw new InvalidOperationException("boom"));
|
||||
|
||||
var dmProbe = CreateTestProbe();
|
||||
var actor = Sys.ActorOf(Props.Create(() =>
|
||||
new SiteCommunicationActor(
|
||||
"site1",
|
||||
new CommunicationOptions { ApplicationHeartbeatInterval = TimeSpan.FromMilliseconds(50) },
|
||||
dmProbe.Ref, () => true, null, transport)));
|
||||
|
||||
// Let the heartbeat timer fire a few times (each throws inside the transport).
|
||||
Thread.Sleep(250);
|
||||
|
||||
// The actor is still alive and delegating: a subsequent send is handled normally.
|
||||
actor.Tell(new NotificationSubmit(
|
||||
"after-heartbeat", "Operators", "S", "B", "site1", null, null, DateTimeOffset.UtcNow), TestActor);
|
||||
AwaitAssert(() => transport.Received(1).SubmitNotification(
|
||||
Arg.Is<NotificationSubmit>(m => m.NotificationId == "after-heartbeat"), Arg.Is(TestActor)));
|
||||
}
|
||||
|
||||
private static SiteHealthReport MinimalHealthReport(long sequence) => new(
|
||||
SiteId: "site1",
|
||||
SequenceNumber: sequence,
|
||||
ReportTimestamp: DateTimeOffset.UtcNow,
|
||||
DataConnectionStatuses: new Dictionary<string, ConnectionHealth>(),
|
||||
TagResolutionCounts: new Dictionary<string, TagResolutionStatus>(),
|
||||
ScriptErrorCount: 0,
|
||||
AlarmEvaluationErrorCount: 0,
|
||||
StoreAndForwardBufferDepths: new Dictionary<string, int>(),
|
||||
DeadLetterCount: 0,
|
||||
DeployedInstanceCount: 0,
|
||||
EnabledInstanceCount: 0,
|
||||
DisabledInstanceCount: 0);
|
||||
}
|
||||
@@ -104,4 +104,57 @@ public class CommunicationOptionsValidatorTests
|
||||
Assert.True(result.Failed);
|
||||
Assert.Contains("LiveAlarmCachePublishCoalesce", result.FailureMessage);
|
||||
}
|
||||
|
||||
// ── T1A.3: gRPC central transport endpoints (required only when selected) ────
|
||||
|
||||
[Fact]
|
||||
public void GrpcTransport_WithNoEndpoints_IsRejected()
|
||||
{
|
||||
var result = Validate(new CommunicationOptions
|
||||
{
|
||||
CentralTransport = CentralTransportMode.Grpc,
|
||||
CentralGrpcEndpoints = new List<string>(),
|
||||
});
|
||||
Assert.True(result.Failed);
|
||||
Assert.Contains("CentralGrpcEndpoints", result.FailureMessage);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void GrpcTransport_WithBlankEndpoint_IsRejected()
|
||||
{
|
||||
var result = Validate(new CommunicationOptions
|
||||
{
|
||||
CentralTransport = CentralTransportMode.Grpc,
|
||||
CentralGrpcEndpoints = new List<string> { " " },
|
||||
});
|
||||
Assert.True(result.Failed);
|
||||
Assert.Contains("CentralGrpcEndpoints", result.FailureMessage);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void GrpcTransport_WithEndpoints_IsValid()
|
||||
{
|
||||
var result = Validate(new CommunicationOptions
|
||||
{
|
||||
CentralTransport = CentralTransportMode.Grpc,
|
||||
CentralGrpcEndpoints = new List<string>
|
||||
{
|
||||
"http://scadabridge-central-a:8083",
|
||||
"http://scadabridge-central-b:8083",
|
||||
},
|
||||
});
|
||||
Assert.True(result.Succeeded, result.FailureMessage);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void AkkaTransport_IgnoresMissingGrpcEndpoints()
|
||||
{
|
||||
// The default transport must not be forced to declare gRPC endpoints it never dials.
|
||||
var result = Validate(new CommunicationOptions
|
||||
{
|
||||
CentralTransport = CentralTransportMode.Akka,
|
||||
CentralGrpcEndpoints = new List<string>(),
|
||||
});
|
||||
Assert.True(result.Succeeded, result.FailureMessage);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,643 @@
|
||||
using Google.Protobuf;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Entities.Audit;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Audit;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Deployment;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Health;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Notification;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Types;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Types.Audit;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Types.Enums;
|
||||
using ZB.MOM.WW.ScadaBridge.Communication.Grpc;
|
||||
|
||||
namespace ZB.MOM.WW.ScadaBridge.Communication.Tests.Grpc;
|
||||
|
||||
/// <summary>
|
||||
/// Round-trip golden tests for <see cref="CentralControlDtoMapper"/> — the DTO bridge
|
||||
/// for the seven <c>CentralControlService</c> RPCs (Phase 1A of the ClusterClient→gRPC
|
||||
/// migration).
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// <para>
|
||||
/// Every message gets a pair: a FULLY-POPULATED case that proves no field is dropped,
|
||||
/// and a MINIMAL case that proves every nullable/optional/empty-collection member comes
|
||||
/// back as null-or-empty rather than as a zero-valued stand-in. The minimal cases are
|
||||
/// the ones that matter: a per-field unit test happily passes while a whole optional
|
||||
/// branch is silently never written, which is exactly the class of bug the round-trip
|
||||
/// guard in PLAN-05 T8 caught five times over.
|
||||
/// </para>
|
||||
/// <para>
|
||||
/// When a round-trip here fails, the mapper is wrong — not the test.
|
||||
/// </para>
|
||||
/// </remarks>
|
||||
public class CentralControlDtoMapperTests
|
||||
{
|
||||
private static readonly DateTimeOffset SampleInstant =
|
||||
new(2026, 7, 22, 9, 30, 15, 250, TimeSpan.Zero);
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// NotificationSubmit / Ack
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
[Fact]
|
||||
public void NotificationSubmit_RoundTrip_FullyPopulated_PreservesEveryField()
|
||||
{
|
||||
var original = new NotificationSubmit(
|
||||
NotificationId: Guid.NewGuid().ToString(),
|
||||
ListName: "plant-ops",
|
||||
Subject: "Tank 4 overfill",
|
||||
Body: "Level exceeded 95% for 5 minutes.",
|
||||
SourceSiteId: "site-a",
|
||||
SourceInstanceId: "Tank04",
|
||||
SourceScript: "OnLevelHigh",
|
||||
SiteEnqueuedAt: SampleInstant,
|
||||
OriginExecutionId: Guid.NewGuid(),
|
||||
OriginParentExecutionId: Guid.NewGuid(),
|
||||
SourceNode: "node-b");
|
||||
|
||||
var roundTripped = CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(original)));
|
||||
|
||||
// Every member is a scalar, so record value-equality is a true deep compare.
|
||||
Assert.Equal(original, roundTripped);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void NotificationSubmit_RoundTrip_AllOptionalsNull_StayNull()
|
||||
{
|
||||
var original = new NotificationSubmit(
|
||||
NotificationId: Guid.NewGuid().ToString(),
|
||||
ListName: "plant-ops",
|
||||
Subject: "s",
|
||||
Body: "b",
|
||||
SourceSiteId: "site-a",
|
||||
SourceInstanceId: null,
|
||||
SourceScript: null,
|
||||
SiteEnqueuedAt: SampleInstant,
|
||||
OriginExecutionId: null,
|
||||
OriginParentExecutionId: null,
|
||||
SourceNode: null);
|
||||
|
||||
var roundTripped = CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(original)));
|
||||
|
||||
Assert.Equal(original, roundTripped);
|
||||
Assert.Null(roundTripped.SourceInstanceId);
|
||||
Assert.Null(roundTripped.SourceScript);
|
||||
Assert.Null(roundTripped.OriginExecutionId);
|
||||
Assert.Null(roundTripped.OriginParentExecutionId);
|
||||
Assert.Null(roundTripped.SourceNode);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void NotificationSubmit_NonUtcOffset_NormalizesToTheSameInstant()
|
||||
{
|
||||
// Documented contract: a protobuf Timestamp is an instant, so the offset
|
||||
// component of a DateTimeOffset does not survive. Every producer stamps UTC
|
||||
// (Notify.Send uses DateTimeOffset.UtcNow), so the instant is what matters.
|
||||
var melbourne = new DateTimeOffset(2026, 7, 22, 19, 30, 15, TimeSpan.FromHours(10));
|
||||
|
||||
var original = new NotificationSubmit(
|
||||
"id", "list", "s", "b", "site-a", null, null, melbourne);
|
||||
|
||||
var roundTripped = CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(original)));
|
||||
|
||||
Assert.Equal(melbourne.UtcDateTime, roundTripped.SiteEnqueuedAt.UtcDateTime);
|
||||
Assert.Equal(TimeSpan.Zero, roundTripped.SiteEnqueuedAt.Offset);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void NotificationSubmitAck_RoundTrip_Accepted_And_Rejected()
|
||||
{
|
||||
var accepted = new NotificationSubmitAck("n1", Accepted: true, Error: null);
|
||||
var rejected = new NotificationSubmitAck("n2", Accepted: false, Error: "list not found");
|
||||
|
||||
Assert.Equal(accepted, CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(accepted))));
|
||||
Assert.Equal(rejected, CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(rejected))));
|
||||
Assert.Null(CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(accepted))).Error);
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// NotificationStatusQuery / Response
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
[Fact]
|
||||
public void NotificationStatusQuery_RoundTrip_PreservesEveryField()
|
||||
{
|
||||
var original = new NotificationStatusQuery("corr-1", Guid.NewGuid().ToString());
|
||||
|
||||
Assert.Equal(original, CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(original))));
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void NotificationStatusResponse_RoundTrip_FullyPopulated_PreservesEveryField()
|
||||
{
|
||||
var original = new NotificationStatusResponse(
|
||||
CorrelationId: "corr-1",
|
||||
Found: true,
|
||||
Status: "Delivered",
|
||||
RetryCount: 3,
|
||||
LastError: "smtp 421",
|
||||
DeliveredAt: SampleInstant);
|
||||
|
||||
Assert.Equal(original, CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(original))));
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void NotificationStatusResponse_RoundTrip_NotFound_LeavesOptionalsNull()
|
||||
{
|
||||
var original = new NotificationStatusResponse(
|
||||
CorrelationId: "corr-1",
|
||||
Found: false,
|
||||
Status: "Unknown",
|
||||
RetryCount: 0,
|
||||
LastError: null,
|
||||
DeliveredAt: null);
|
||||
|
||||
var roundTripped = CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(original)));
|
||||
|
||||
Assert.Equal(original, roundTripped);
|
||||
Assert.Null(roundTripped.LastError);
|
||||
Assert.Null(roundTripped.DeliveredAt);
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// Audit ingest envelopes (rows delegate to AuditEventDtoMapper / SiteCallDtoMapper)
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
[Fact]
|
||||
public void IngestAuditEventsCommand_RoundTrip_PreservesOrderAndRows()
|
||||
{
|
||||
var first = NewAuditEvent(sourceScript: "OnDemand");
|
||||
var second = NewAuditEvent(sourceScript: "OnTrigger");
|
||||
var original = new IngestAuditEventsCommand([first, second]);
|
||||
|
||||
var roundTripped = CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(original)));
|
||||
|
||||
Assert.Equal(2, roundTripped.Events.Count);
|
||||
Assert.Equal(first.AsRow().EventId, roundTripped.Events[0].AsRow().EventId);
|
||||
Assert.Equal("OnDemand", roundTripped.Events[0].AsRow().SourceScript);
|
||||
Assert.Equal(second.AsRow().EventId, roundTripped.Events[1].AsRow().EventId);
|
||||
Assert.Equal("OnTrigger", roundTripped.Events[1].AsRow().SourceScript);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void IngestAuditEventsCommand_RoundTrip_EmptyBatch_StaysEmpty()
|
||||
{
|
||||
var roundTripped = CentralControlDtoMapper.FromDto(
|
||||
Wire(CentralControlDtoMapper.ToDto(new IngestAuditEventsCommand([]))));
|
||||
|
||||
Assert.Empty(roundTripped.Events);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void IngestCachedTelemetryCommand_RoundTrip_PreservesBothHalvesOfEachPacket()
|
||||
{
|
||||
var audit = NewAuditEvent(sourceScript: "OnDemand");
|
||||
var siteCall = NewSiteCall();
|
||||
var original = new IngestCachedTelemetryCommand([new CachedTelemetryEntry(audit, siteCall)]);
|
||||
|
||||
var roundTripped = CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(original)));
|
||||
|
||||
var entry = Assert.Single(roundTripped.Entries);
|
||||
Assert.Equal(audit.AsRow().EventId, entry.Audit.AsRow().EventId);
|
||||
|
||||
// IngestedAtUtc is central-set inside the dual-write transaction and is
|
||||
// deliberately off the wire, so it is the one member excluded from the compare.
|
||||
Assert.Equal(siteCall with { IngestedAtUtc = default }, entry.SiteCall with { IngestedAtUtc = default });
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void IngestCachedTelemetryCommand_RoundTrip_NullableSiteCallFields_StayNull()
|
||||
{
|
||||
var siteCall = NewSiteCall() with
|
||||
{
|
||||
SourceNode = null,
|
||||
LastError = null,
|
||||
HttpStatus = null,
|
||||
TerminalAtUtc = null,
|
||||
};
|
||||
var original = new IngestCachedTelemetryCommand(
|
||||
[new CachedTelemetryEntry(NewAuditEvent(), siteCall)]);
|
||||
|
||||
var roundTripped = CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(original)));
|
||||
|
||||
var entry = Assert.Single(roundTripped.Entries);
|
||||
Assert.Null(entry.SiteCall.SourceNode);
|
||||
Assert.Null(entry.SiteCall.LastError);
|
||||
Assert.Null(entry.SiteCall.HttpStatus);
|
||||
Assert.Null(entry.SiteCall.TerminalAtUtc);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void IngestCachedTelemetryCommand_RoundTrip_EmptyBatch_StaysEmpty()
|
||||
{
|
||||
var roundTripped = CentralControlDtoMapper.FromDto(
|
||||
Wire(CentralControlDtoMapper.ToDto(new IngestCachedTelemetryCommand([]))));
|
||||
|
||||
Assert.Empty(roundTripped.Entries);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void IngestAck_RoundTrip_PreservesIdsAndOrder()
|
||||
{
|
||||
var ids = new[] { Guid.NewGuid(), Guid.NewGuid(), Guid.NewGuid() };
|
||||
|
||||
var roundTripped = CentralControlDtoMapper.FromIngestAck(
|
||||
Wire(CentralControlDtoMapper.ToIngestAck(ids)));
|
||||
|
||||
Assert.Equal(ids, roundTripped);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void IngestAck_RoundTrip_NothingAccepted_StaysEmpty()
|
||||
{
|
||||
// An empty ack is meaningful — "zero rows persisted" leaves the site's rows
|
||||
// Pending — so it must not be indistinguishable from a dropped field.
|
||||
Assert.Empty(CentralControlDtoMapper.FromIngestAck(
|
||||
Wire(CentralControlDtoMapper.ToIngestAck([]))));
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// Startup reconciliation
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
[Fact]
|
||||
public void ReconcileSiteRequest_RoundTrip_PreservesInventoryMap()
|
||||
{
|
||||
var original = new ReconcileSiteRequest(
|
||||
SiteIdentifier: "site-a",
|
||||
NodeId: "node-a",
|
||||
LocalNameToRevisionHash: new Dictionary<string, string>
|
||||
{
|
||||
["Plant.Tank04"] = "hash-1",
|
||||
["Plant.Pump01"] = "hash-2",
|
||||
});
|
||||
|
||||
var roundTripped = CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(original)));
|
||||
|
||||
Assert.Equal(original.SiteIdentifier, roundTripped.SiteIdentifier);
|
||||
Assert.Equal(original.NodeId, roundTripped.NodeId);
|
||||
Assert.Equal(
|
||||
original.LocalNameToRevisionHash.OrderBy(kv => kv.Key),
|
||||
roundTripped.LocalNameToRevisionHash.OrderBy(kv => kv.Key));
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void ReconcileSiteRequest_RoundTrip_EmptyInventory_StaysEmpty()
|
||||
{
|
||||
// A node with nothing deployed is the normal first-boot case, and central
|
||||
// must see an empty inventory rather than a missing one.
|
||||
var roundTripped = CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(
|
||||
new ReconcileSiteRequest("site-a", "node-a", new Dictionary<string, string>()))));
|
||||
|
||||
Assert.Empty(roundTripped.LocalNameToRevisionHash);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void ReconcileSiteResponse_RoundTrip_PreservesGapOrphansAndUrl()
|
||||
{
|
||||
var original = new ReconcileSiteResponse(
|
||||
Gap:
|
||||
[
|
||||
new ReconcileGapItem("Plant.Tank04", "dep-1", "hash-1", IsEnabled: true, "token-1"),
|
||||
new ReconcileGapItem("Plant.Pump01", "dep-2", "hash-2", IsEnabled: false, "token-2"),
|
||||
],
|
||||
OrphanNames: ["Plant.Retired01", "Plant.Retired02"],
|
||||
CentralFetchBaseUrl: "http://scadabridge-central-a:5000");
|
||||
|
||||
var roundTripped = CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(original)));
|
||||
|
||||
Assert.Equal(original.Gap, roundTripped.Gap);
|
||||
Assert.Equal(original.OrphanNames, roundTripped.OrphanNames);
|
||||
Assert.Equal(original.CentralFetchBaseUrl, roundTripped.CentralFetchBaseUrl);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void ReconcileSiteResponse_RoundTrip_NoGap_StaysEmpty()
|
||||
{
|
||||
var roundTripped = CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(
|
||||
new ReconcileSiteResponse([], [], "http://central:5000"))));
|
||||
|
||||
Assert.Empty(roundTripped.Gap);
|
||||
Assert.Empty(roundTripped.OrphanNames);
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// Site health
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
[Fact]
|
||||
public void SiteHealthReport_RoundTrip_FullyPopulated_PreservesEveryField()
|
||||
{
|
||||
var original = FullyPopulatedHealthReport();
|
||||
|
||||
var roundTripped = CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(original)));
|
||||
|
||||
AssertHealthReportsEqual(original, roundTripped);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void SiteHealthReport_RoundTrip_MinimalReport_LeavesEveryOptionalNullOrEmpty()
|
||||
{
|
||||
// The shape a producer emits before any reporter has run: no endpoints map, no
|
||||
// tag-quality map, no cluster-node list, no audit backlog, no nullable gauges.
|
||||
var original = new SiteHealthReport(
|
||||
SiteId: "site-a",
|
||||
SequenceNumber: 1,
|
||||
ReportTimestamp: SampleInstant,
|
||||
DataConnectionStatuses: new Dictionary<string, ConnectionHealth>(),
|
||||
TagResolutionCounts: new Dictionary<string, TagResolutionStatus>(),
|
||||
ScriptErrorCount: 0,
|
||||
AlarmEvaluationErrorCount: 0,
|
||||
StoreAndForwardBufferDepths: new Dictionary<string, int>(),
|
||||
DeadLetterCount: 0,
|
||||
DeployedInstanceCount: 0,
|
||||
EnabledInstanceCount: 0,
|
||||
DisabledInstanceCount: 0);
|
||||
|
||||
var roundTripped = CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(original)));
|
||||
|
||||
AssertHealthReportsEqual(original, roundTripped);
|
||||
|
||||
Assert.Empty(roundTripped.DataConnectionStatuses);
|
||||
Assert.Empty(roundTripped.TagResolutionCounts);
|
||||
Assert.Empty(roundTripped.StoreAndForwardBufferDepths);
|
||||
Assert.Null(roundTripped.DataConnectionEndpoints);
|
||||
Assert.Null(roundTripped.DataConnectionTagQuality);
|
||||
Assert.Null(roundTripped.ClusterNodes);
|
||||
Assert.Null(roundTripped.SiteAuditBacklog);
|
||||
Assert.Null(roundTripped.OldestParkedMessageAgeSeconds);
|
||||
Assert.Null(roundTripped.ScriptOldestBusyAgeSeconds);
|
||||
Assert.Null(roundTripped.LocalDbReplicationConnected);
|
||||
Assert.Null(roundTripped.LocalDbOplogBacklog);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void SiteHealthReport_RoundTrip_EmptyNullableCollections_StayEmptyNotNull()
|
||||
{
|
||||
// The distinction the three wrapper messages exist for. Null means "this node
|
||||
// does not report the signal"; empty means "it reports it, and there is
|
||||
// nothing in it". proto3 cannot express presence on repeated/map fields, so
|
||||
// collapsing one into the other here would be invisible until an operator
|
||||
// misread the health page.
|
||||
var original = FullyPopulatedHealthReport() with
|
||||
{
|
||||
DataConnectionEndpoints = new Dictionary<string, string>(),
|
||||
DataConnectionTagQuality = new Dictionary<string, TagQualityCounts>(),
|
||||
ClusterNodes = [],
|
||||
};
|
||||
|
||||
var roundTripped = CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(original)));
|
||||
|
||||
Assert.NotNull(roundTripped.DataConnectionEndpoints);
|
||||
Assert.Empty(roundTripped.DataConnectionEndpoints);
|
||||
Assert.NotNull(roundTripped.DataConnectionTagQuality);
|
||||
Assert.Empty(roundTripped.DataConnectionTagQuality);
|
||||
Assert.NotNull(roundTripped.ClusterNodes);
|
||||
Assert.Empty(roundTripped.ClusterNodes);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void SiteHealthReport_RoundTrip_FalseAndZeroGauges_StayFalseAndZero()
|
||||
{
|
||||
// The mirror of the null case: LocalDbReplicationConnected=false ("configured
|
||||
// but disconnected") and LocalDbOplogBacklog=0 ("healthy, nothing queued") are
|
||||
// real values that must not decay into null on the wire.
|
||||
var original = FullyPopulatedHealthReport() with
|
||||
{
|
||||
LocalDbReplicationConnected = false,
|
||||
LocalDbOplogBacklog = 0,
|
||||
OldestParkedMessageAgeSeconds = 0d,
|
||||
ScriptOldestBusyAgeSeconds = 0d,
|
||||
};
|
||||
|
||||
var roundTripped = CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(original)));
|
||||
|
||||
Assert.False(roundTripped.LocalDbReplicationConnected);
|
||||
Assert.Equal(0L, roundTripped.LocalDbOplogBacklog);
|
||||
Assert.Equal(0d, roundTripped.OldestParkedMessageAgeSeconds);
|
||||
Assert.Equal(0d, roundTripped.ScriptOldestBusyAgeSeconds);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void SiteHealthReport_RoundTrip_BacklogWithEmptyQueue_KeepsNullOldestPending()
|
||||
{
|
||||
var original = FullyPopulatedHealthReport() with
|
||||
{
|
||||
SiteAuditBacklog = new SiteAuditBacklogSnapshot(0, null, 4096),
|
||||
};
|
||||
|
||||
var roundTripped = CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(original)));
|
||||
|
||||
Assert.Equal(new SiteAuditBacklogSnapshot(0, null, 4096), roundTripped.SiteAuditBacklog);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void SiteHealthReportAck_RoundTrip_Accepted_And_Rejected()
|
||||
{
|
||||
var accepted = new SiteHealthReportAck("site-a", 42, Accepted: true);
|
||||
var rejected = new SiteHealthReportAck("site-a", 43, Accepted: false, Error: "aggregator down");
|
||||
|
||||
Assert.Equal(accepted, CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(accepted))));
|
||||
Assert.Equal(rejected, CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(rejected))));
|
||||
Assert.Null(CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(accepted))).Error);
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// Heartbeat + ConnectionHealth enum
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
[Theory]
|
||||
[InlineData(true)]
|
||||
[InlineData(false)]
|
||||
public void HeartbeatMessage_RoundTrip_PreservesEveryField(bool isActive)
|
||||
{
|
||||
var original = new HeartbeatMessage("site-a", "scadabridge-site-a-node-b", isActive, SampleInstant);
|
||||
|
||||
Assert.Equal(original, CentralControlDtoMapper.FromDto(Wire(CentralControlDtoMapper.ToDto(original))));
|
||||
}
|
||||
|
||||
[Theory]
|
||||
[InlineData(ConnectionHealth.Connected)]
|
||||
[InlineData(ConnectionHealth.Disconnected)]
|
||||
[InlineData(ConnectionHealth.Connecting)]
|
||||
[InlineData(ConnectionHealth.Error)]
|
||||
public void ConnectionHealth_RoundTrip_EveryValueSurvives(ConnectionHealth health)
|
||||
{
|
||||
Assert.Equal(health, CentralControlDtoMapper.FromDto(CentralControlDtoMapper.ToDto(health)));
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void ConnectionHealth_Unspecified_DecodesToErrorNotConnected()
|
||||
{
|
||||
// Fail-safe direction: a wire value this build does not recognise must never
|
||||
// render as "healthy" on the central health page.
|
||||
Assert.Equal(
|
||||
ConnectionHealth.Error,
|
||||
CentralControlDtoMapper.FromDto(ConnectionHealthEnum.ConnectionHealthUnspecified));
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void ConnectionHealth_EveryEnumMemberIsMapped()
|
||||
{
|
||||
// Guards the ArgumentOutOfRangeException path: adding a ConnectionHealth value
|
||||
// without extending the mapper should fail here, not silently on a rig.
|
||||
foreach (var health in Enum.GetValues<ConnectionHealth>())
|
||||
{
|
||||
var wire = CentralControlDtoMapper.ToDto(health);
|
||||
Assert.NotEqual(ConnectionHealthEnum.ConnectionHealthUnspecified, wire);
|
||||
}
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// Fixtures
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
private static SiteHealthReport FullyPopulatedHealthReport() =>
|
||||
new(
|
||||
SiteId: "site-a",
|
||||
SequenceNumber: 987654321L,
|
||||
ReportTimestamp: SampleInstant,
|
||||
DataConnectionStatuses: new Dictionary<string, ConnectionHealth>
|
||||
{
|
||||
["opc-main"] = ConnectionHealth.Connected,
|
||||
["mx-gateway"] = ConnectionHealth.Error,
|
||||
["opc-backup"] = ConnectionHealth.Connecting,
|
||||
["legacy"] = ConnectionHealth.Disconnected,
|
||||
},
|
||||
TagResolutionCounts: new Dictionary<string, TagResolutionStatus>
|
||||
{
|
||||
["opc-main"] = new(TotalSubscribed: 120, SuccessfullyResolved: 118),
|
||||
},
|
||||
ScriptErrorCount: 4,
|
||||
AlarmEvaluationErrorCount: 2,
|
||||
StoreAndForwardBufferDepths: new Dictionary<string, int>
|
||||
{
|
||||
["erp"] = 17,
|
||||
["mes"] = 0,
|
||||
},
|
||||
DeadLetterCount: 5,
|
||||
DeployedInstanceCount: 30,
|
||||
EnabledInstanceCount: 28,
|
||||
DisabledInstanceCount: 2,
|
||||
NodeRole: "Active",
|
||||
NodeHostname: "scadabridge-site-a-node-a",
|
||||
DataConnectionEndpoints: new Dictionary<string, string>
|
||||
{
|
||||
["opc-main"] = "opc.tcp://opcua:4840",
|
||||
},
|
||||
DataConnectionTagQuality: new Dictionary<string, TagQualityCounts>
|
||||
{
|
||||
["opc-main"] = new(Good: 100, Bad: 3, Uncertain: 15),
|
||||
},
|
||||
ParkedMessageCount: 6,
|
||||
ClusterNodes:
|
||||
[
|
||||
new NodeStatus("scadabridge-site-a-node-a", IsOnline: true, "Active"),
|
||||
new NodeStatus("scadabridge-site-a-node-b", IsOnline: false, "Standby"),
|
||||
],
|
||||
SiteAuditWriteFailures: 7,
|
||||
AuditRedactionFailure: 8,
|
||||
SiteAuditBacklog: new SiteAuditBacklogSnapshot(
|
||||
PendingCount: 42,
|
||||
OldestPendingUtc: new DateTime(2026, 7, 22, 8, 0, 0, DateTimeKind.Utc),
|
||||
OnDiskBytes: 1_234_567L),
|
||||
SiteEventLogWriteFailures: 9L,
|
||||
OldestParkedMessageAgeSeconds: 3600.5d)
|
||||
{
|
||||
ScriptQueueDepth = 11,
|
||||
ScriptBusyThreads = 3,
|
||||
ScriptOldestBusyAgeSeconds = 12.25d,
|
||||
LocalDbReplicationConnected = true,
|
||||
LocalDbOplogBacklog = 250L,
|
||||
};
|
||||
|
||||
/// <summary>
|
||||
/// Compares two reports member by member. <see cref="SiteHealthReport"/> is a record,
|
||||
/// but its dictionary/list members compare by reference under record equality, so
|
||||
/// <c>Assert.Equal(a, b)</c> would pass trivially for the scalars and never look
|
||||
/// inside the collections — the exact blind spot these goldens exist to close.
|
||||
/// </summary>
|
||||
private static void AssertHealthReportsEqual(SiteHealthReport expected, SiteHealthReport actual)
|
||||
{
|
||||
Assert.Equal(expected.SiteId, actual.SiteId);
|
||||
Assert.Equal(expected.SequenceNumber, actual.SequenceNumber);
|
||||
Assert.Equal(expected.ReportTimestamp, actual.ReportTimestamp);
|
||||
Assert.Equal(
|
||||
expected.DataConnectionStatuses.OrderBy(kv => kv.Key),
|
||||
actual.DataConnectionStatuses.OrderBy(kv => kv.Key));
|
||||
Assert.Equal(
|
||||
expected.TagResolutionCounts.OrderBy(kv => kv.Key),
|
||||
actual.TagResolutionCounts.OrderBy(kv => kv.Key));
|
||||
Assert.Equal(expected.ScriptErrorCount, actual.ScriptErrorCount);
|
||||
Assert.Equal(expected.AlarmEvaluationErrorCount, actual.AlarmEvaluationErrorCount);
|
||||
Assert.Equal(
|
||||
expected.StoreAndForwardBufferDepths.OrderBy(kv => kv.Key),
|
||||
actual.StoreAndForwardBufferDepths.OrderBy(kv => kv.Key));
|
||||
Assert.Equal(expected.DeadLetterCount, actual.DeadLetterCount);
|
||||
Assert.Equal(expected.DeployedInstanceCount, actual.DeployedInstanceCount);
|
||||
Assert.Equal(expected.EnabledInstanceCount, actual.EnabledInstanceCount);
|
||||
Assert.Equal(expected.DisabledInstanceCount, actual.DisabledInstanceCount);
|
||||
Assert.Equal(expected.NodeRole, actual.NodeRole);
|
||||
Assert.Equal(expected.NodeHostname, actual.NodeHostname);
|
||||
Assert.Equal(
|
||||
expected.DataConnectionEndpoints?.OrderBy(kv => kv.Key),
|
||||
actual.DataConnectionEndpoints?.OrderBy(kv => kv.Key));
|
||||
Assert.Equal(
|
||||
expected.DataConnectionTagQuality?.OrderBy(kv => kv.Key),
|
||||
actual.DataConnectionTagQuality?.OrderBy(kv => kv.Key));
|
||||
Assert.Equal(expected.ParkedMessageCount, actual.ParkedMessageCount);
|
||||
Assert.Equal(expected.ClusterNodes, actual.ClusterNodes);
|
||||
Assert.Equal(expected.SiteAuditWriteFailures, actual.SiteAuditWriteFailures);
|
||||
Assert.Equal(expected.AuditRedactionFailure, actual.AuditRedactionFailure);
|
||||
Assert.Equal(expected.SiteAuditBacklog, actual.SiteAuditBacklog);
|
||||
Assert.Equal(expected.SiteEventLogWriteFailures, actual.SiteEventLogWriteFailures);
|
||||
Assert.Equal(expected.OldestParkedMessageAgeSeconds, actual.OldestParkedMessageAgeSeconds);
|
||||
Assert.Equal(expected.ScriptQueueDepth, actual.ScriptQueueDepth);
|
||||
Assert.Equal(expected.ScriptBusyThreads, actual.ScriptBusyThreads);
|
||||
Assert.Equal(expected.ScriptOldestBusyAgeSeconds, actual.ScriptOldestBusyAgeSeconds);
|
||||
Assert.Equal(expected.LocalDbReplicationConnected, actual.LocalDbReplicationConnected);
|
||||
Assert.Equal(expected.LocalDbOplogBacklog, actual.LocalDbOplogBacklog);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Serializes a wire message and parses it back, so every round-trip below crosses a
|
||||
/// real protobuf encode/decode rather than only exercising the mapper's object graph.
|
||||
/// This is what proves the three collection wrapper messages keep their presence
|
||||
/// when empty — an empty message encodes as a tag with zero-length payload, and a
|
||||
/// mapper that used a bare <c>repeated</c>/<c>map</c> field instead would decode an
|
||||
/// empty collection back as null with no test-visible difference at the object level.
|
||||
/// </summary>
|
||||
/// <typeparam name="T">Generated protobuf message type.</typeparam>
|
||||
/// <param name="message">The message to send through an encode/decode cycle.</param>
|
||||
/// <returns>An independent instance parsed from <paramref name="message"/>'s bytes.</returns>
|
||||
private static T Wire<T>(T message) where T : IMessage<T>, new() =>
|
||||
new MessageParser<T>(() => new T()).ParseFrom(message.ToByteArray());
|
||||
|
||||
private static ZB.MOM.WW.Audit.AuditEvent NewAuditEvent(string? sourceScript = null) =>
|
||||
ScadaBridgeAuditEventFactory.Create(
|
||||
channel: AuditChannel.ApiOutbound,
|
||||
kind: AuditKind.ApiCallCached,
|
||||
status: AuditStatus.Forwarded,
|
||||
eventId: Guid.NewGuid(),
|
||||
occurredAtUtc: new DateTime(2026, 7, 22, 9, 0, 0, DateTimeKind.Utc),
|
||||
target: "ERP.GetOrder",
|
||||
sourceSiteId: "site-a",
|
||||
sourceNode: "node-a",
|
||||
sourceScript: sourceScript);
|
||||
|
||||
private static SiteCall NewSiteCall() => new()
|
||||
{
|
||||
TrackedOperationId = TrackedOperationId.New(),
|
||||
Channel = "ApiOutbound",
|
||||
Target = "ERP.GetOrder",
|
||||
SourceSite = "site-a",
|
||||
SourceNode = "node-a",
|
||||
Status = "Delivered",
|
||||
RetryCount = 2,
|
||||
LastError = "transient 503",
|
||||
HttpStatus = 200,
|
||||
CreatedAtUtc = new DateTime(2026, 7, 22, 8, 0, 0, DateTimeKind.Utc),
|
||||
UpdatedAtUtc = new DateTime(2026, 7, 22, 8, 5, 0, DateTimeKind.Utc),
|
||||
TerminalAtUtc = new DateTime(2026, 7, 22, 8, 10, 0, DateTimeKind.Utc),
|
||||
IngestedAtUtc = new DateTime(2026, 7, 22, 8, 10, 1, DateTimeKind.Utc),
|
||||
};
|
||||
}
|
||||
+181
@@ -0,0 +1,181 @@
|
||||
using Akka.Actor;
|
||||
using Akka.TestKit;
|
||||
using Akka.TestKit.Xunit2;
|
||||
using Grpc.Core;
|
||||
using Microsoft.Extensions.Logging.Abstractions;
|
||||
using Microsoft.Extensions.Options;
|
||||
using NSubstitute;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Audit;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Notification;
|
||||
using ZB.MOM.WW.ScadaBridge.Communication;
|
||||
using ZB.MOM.WW.ScadaBridge.Communication.Grpc;
|
||||
|
||||
namespace ZB.MOM.WW.ScadaBridge.Communication.Tests.Grpc;
|
||||
|
||||
/// <summary>
|
||||
/// Behaviour of <see cref="CentralControlGrpcService"/> that does not need a real gRPC pipeline:
|
||||
/// the readiness gate, the fire-and-forget heartbeat contract, the ingest-timeout status
|
||||
/// mapping, and the shared audit-ingest timeout constant. The auth interceptor + real transport
|
||||
/// are proven separately in <c>CentralControlEndToEndTests</c> (Host.Tests).
|
||||
/// </summary>
|
||||
public class CentralControlGrpcServiceTests : TestKit
|
||||
{
|
||||
private static ServerCallContext NewContext(CancellationToken ct = default)
|
||||
{
|
||||
var context = Substitute.For<ServerCallContext>();
|
||||
context.CancellationToken.Returns(ct);
|
||||
return context;
|
||||
}
|
||||
|
||||
private CentralControlGrpcService CreateService(CommunicationOptions? options = null)
|
||||
=> new(
|
||||
NullLogger<CentralControlGrpcService>.Instance,
|
||||
Options.Create(options ?? new CommunicationOptions()));
|
||||
|
||||
[Fact]
|
||||
public async Task BeforeSetReady_AUnaryCall_IsUnavailable()
|
||||
{
|
||||
// Nothing was dispatched, so Unavailable is the right status — it tells a site transport
|
||||
// the call never ran and a cross-node retry is safe.
|
||||
var service = CreateService();
|
||||
|
||||
var ex = await Assert.ThrowsAsync<RpcException>(
|
||||
() => service.SubmitNotification(new NotificationSubmitDto(), NewContext()));
|
||||
|
||||
Assert.Equal(StatusCode.Unavailable, ex.StatusCode);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task BeforeSetReady_ANonEmptyIngestBatch_IsUnavailable()
|
||||
{
|
||||
var service = CreateService();
|
||||
var batch = new AuditEventBatch();
|
||||
batch.Events.Add(NewAuditDto());
|
||||
|
||||
var ex = await Assert.ThrowsAsync<RpcException>(
|
||||
() => service.IngestAuditEvents(batch, NewContext()));
|
||||
|
||||
Assert.Equal(StatusCode.Unavailable, ex.StatusCode);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task AfterSetReady_AUnaryCall_ReachesTheActor()
|
||||
{
|
||||
var stub = Sys.ActorOf(Props.Create(() => new StubActor()));
|
||||
var service = CreateService();
|
||||
service.SetReady(stub);
|
||||
|
||||
var ack = await service.SubmitNotification(NewNotificationDto(), NewContext());
|
||||
|
||||
Assert.True(ack.Accepted);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task Heartbeat_BeforeSetReady_SucceedsAndIsDropped()
|
||||
{
|
||||
// Fire-and-forget end-to-end: a heartbeat must never fault the site's timer, so even
|
||||
// with no actor wired the call returns OK rather than Unavailable.
|
||||
var service = CreateService();
|
||||
|
||||
var reply = await service.Heartbeat(NewHeartbeatDto(), NewContext());
|
||||
|
||||
Assert.NotNull(reply);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task Heartbeat_AfterSetReady_TellsTheActor_AndNeverAsks()
|
||||
{
|
||||
// A black-hole actor that never replies would hang an Ask forever; the heartbeat still
|
||||
// returns immediately, proving it is a Tell, not an Ask.
|
||||
var blackHole = Sys.ActorOf(Props.Create(() => new NeverRepliesActor()));
|
||||
var service = CreateService();
|
||||
service.SetReady(blackHole);
|
||||
|
||||
var reply = await service.Heartbeat(NewHeartbeatDto(), NewContext());
|
||||
|
||||
Assert.NotNull(reply);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task WhenTheActorNeverReplies_TheCall_IsDeadlineExceeded_NotUnavailable()
|
||||
{
|
||||
// The message WAS delivered (Unavailable would wrongly invite a duplicate retry on the
|
||||
// peer node); a timeout is DeadlineExceeded, which callers never cross-node-retry.
|
||||
var blackHole = Sys.ActorOf(Props.Create(() => new NeverRepliesActor()));
|
||||
var service = CreateService(new CommunicationOptions
|
||||
{
|
||||
NotificationForwardTimeout = TimeSpan.FromMilliseconds(200),
|
||||
});
|
||||
service.SetReady(blackHole);
|
||||
|
||||
var ex = await Assert.ThrowsAsync<RpcException>(
|
||||
() => service.SubmitNotification(NewNotificationDto(), NewContext()));
|
||||
|
||||
Assert.Equal(StatusCode.DeadlineExceeded, ex.StatusCode);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task EmptyIngestBatch_ShortCircuits_EvenBeforeSetReady()
|
||||
{
|
||||
// An empty batch is a no-op the actor need never see; it must not depend on readiness.
|
||||
var service = CreateService();
|
||||
|
||||
var ack = await service.IngestAuditEvents(new AuditEventBatch(), NewContext());
|
||||
|
||||
Assert.Empty(ack.AcceptedEventIds);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void TheAuditIngestTimeout_IsTheOneSharedConstant()
|
||||
{
|
||||
// The plan calls SiteStreamGrpcServer.AuditIngestAskTimeout "one source of truth" shared
|
||||
// between the two audit-ingest transports; the central service must not re-declare 30s.
|
||||
Assert.Equal(TimeSpan.FromSeconds(30), SiteStreamGrpcServer.AuditIngestAskTimeout);
|
||||
}
|
||||
|
||||
// The mapper reads SiteEnqueuedAt unconditionally, so a DTO that reaches mapping must carry
|
||||
// a timestamp. Only DTOs that get past the readiness/auth gate map, so the negative tests
|
||||
// above can pass a bare DTO.
|
||||
private static NotificationSubmitDto NewNotificationDto() => new()
|
||||
{
|
||||
NotificationId = Guid.NewGuid().ToString(),
|
||||
ListName = "ops",
|
||||
SiteEnqueuedAt = Google.Protobuf.WellKnownTypes.Timestamp.FromDateTime(
|
||||
DateTime.SpecifyKind(new DateTime(2026, 5, 20, 10, 0, 0), DateTimeKind.Utc)),
|
||||
};
|
||||
|
||||
private static HeartbeatDto NewHeartbeatDto() => new()
|
||||
{
|
||||
SiteId = "site-a",
|
||||
NodeHostname = "node-a",
|
||||
IsActive = true,
|
||||
Timestamp = Google.Protobuf.WellKnownTypes.Timestamp.FromDateTime(
|
||||
DateTime.SpecifyKind(new DateTime(2026, 5, 20, 10, 0, 0), DateTimeKind.Utc)),
|
||||
};
|
||||
|
||||
private static AuditEventDto NewAuditDto() => new()
|
||||
{
|
||||
EventId = Guid.NewGuid().ToString(),
|
||||
OccurredAtUtc = Google.Protobuf.WellKnownTypes.Timestamp.FromDateTime(
|
||||
DateTime.SpecifyKind(new DateTime(2026, 5, 20, 10, 0, 0), DateTimeKind.Utc)),
|
||||
Channel = "ApiOutbound",
|
||||
Kind = "ApiCall",
|
||||
Status = "Delivered",
|
||||
SourceSiteId = "site-a",
|
||||
};
|
||||
|
||||
private sealed class StubActor : ReceiveActor
|
||||
{
|
||||
public StubActor()
|
||||
{
|
||||
Receive<NotificationSubmit>(msg =>
|
||||
Sender.Tell(new NotificationSubmitAck(msg.NotificationId, Accepted: true, Error: null)));
|
||||
}
|
||||
}
|
||||
|
||||
/// <summary>Swallows every message and never replies, so an Ask against it times out.</summary>
|
||||
private sealed class NeverRepliesActor : ReceiveActor
|
||||
{
|
||||
public NeverRepliesActor() => ReceiveAny(_ => { });
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,197 @@
|
||||
using Grpc.Core;
|
||||
using Microsoft.Extensions.Logging.Abstractions;
|
||||
using ZB.MOM.WW.ScadaBridge.Communication.Grpc;
|
||||
|
||||
namespace ZB.MOM.WW.ScadaBridge.Host.Tests;
|
||||
|
||||
/// <summary>
|
||||
/// Central's inbound gate for the site→central gRPC control plane (T1A.2). The mirror of
|
||||
/// <see cref="ControlPlaneAuthInterceptorTests"/>, but central's verification model is inverted:
|
||||
/// it looks up the key for the site named in the <c>x-scadabridge-site</c> header and checks the
|
||||
/// bearer token against THAT, rather than against one own-key.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Central genuinely needs a distinct verification model, so it is a separate class with its own
|
||||
/// single public constructor — not a variant ctor on <see cref="ControlPlaneAuthInterceptor"/>,
|
||||
/// whose two-public-constructor form once silently disabled the gate. The one-public-ctor
|
||||
/// invariant is pinned below exactly as the sibling pins it.
|
||||
/// </remarks>
|
||||
public class CentralControlAuthInterceptorTests
|
||||
{
|
||||
private const string ControlMethod =
|
||||
"/scadabridge.centralcontrol.v1.CentralControlService/SubmitNotification";
|
||||
private const string SiteStreamMethod = "/sitestream.SiteStreamService/SubscribeInstance";
|
||||
|
||||
private const string SiteA = "site-a";
|
||||
private const string SiteAKey = "site-a-preshared-key";
|
||||
private const string SiteB = "site-b";
|
||||
private const string SiteBKey = "site-b-preshared-key";
|
||||
|
||||
/// <summary>An <see cref="ISitePskProvider"/> backed by a fixed site→key map; throws for unknown sites.</summary>
|
||||
private sealed class MapPskProvider(IReadOnlyDictionary<string, string> keys) : ISitePskProvider
|
||||
{
|
||||
public ValueTask<string> GetAsync(string siteId, CancellationToken ct)
|
||||
=> keys.TryGetValue(siteId, out var key)
|
||||
? new ValueTask<string>(key)
|
||||
: throw new InvalidOperationException($"no key for '{siteId}'");
|
||||
|
||||
public void Invalidate(string siteId) { }
|
||||
}
|
||||
|
||||
private static CentralControlAuthInterceptor CreateInterceptor()
|
||||
=> new(
|
||||
new MapPskProvider(new Dictionary<string, string> { [SiteA] = SiteAKey, [SiteB] = SiteBKey }),
|
||||
NullLogger<CentralControlAuthInterceptor>.Instance);
|
||||
|
||||
private static ServerCallContext CreateContext(
|
||||
string method, string? siteHeader, string? authorizationHeader)
|
||||
{
|
||||
var headers = new Metadata();
|
||||
if (siteHeader is not null)
|
||||
headers.Add(ControlPlaneCredentials.SiteHeader, siteHeader);
|
||||
if (authorizationHeader is not null)
|
||||
headers.Add(ControlPlaneCredentials.AuthorizationHeader, authorizationHeader);
|
||||
|
||||
return new FakeServerCallContext(method, headers);
|
||||
}
|
||||
|
||||
private sealed class FakeServerCallContext(string method, Metadata requestHeaders)
|
||||
: ServerCallContext
|
||||
{
|
||||
protected override string MethodCore => method;
|
||||
protected override string HostCore => "localhost";
|
||||
protected override string PeerCore => "ipv4:127.0.0.1:12345";
|
||||
protected override DateTime DeadlineCore => DateTime.UtcNow.AddMinutes(1);
|
||||
protected override Metadata RequestHeadersCore => requestHeaders;
|
||||
protected override CancellationToken CancellationTokenCore => CancellationToken.None;
|
||||
protected override Metadata ResponseTrailersCore { get; } = [];
|
||||
protected override Status StatusCore { get; set; }
|
||||
protected override WriteOptions? WriteOptionsCore { get; set; }
|
||||
protected override AuthContext AuthContextCore { get; } =
|
||||
new(null, new Dictionary<string, List<AuthProperty>>());
|
||||
|
||||
protected override ContextPropagationToken CreatePropagationTokenCore(
|
||||
ContextPropagationOptions? options)
|
||||
=> throw new NotSupportedException();
|
||||
|
||||
protected override Task WriteResponseHeadersAsyncCore(Metadata responseHeaders)
|
||||
=> Task.CompletedTask;
|
||||
}
|
||||
|
||||
private static Task<string> Invoke(
|
||||
CentralControlAuthInterceptor interceptor, ServerCallContext context)
|
||||
=> interceptor.UnaryServerHandler<string, string>(
|
||||
"request", context, (_, _) => Task.FromResult("ok"));
|
||||
|
||||
[Fact]
|
||||
public async Task NonGatedMethod_PassesThrough()
|
||||
{
|
||||
// Central's interceptor gates only CentralControlService; anything else on the listener
|
||||
// (there is nothing today) is not its concern.
|
||||
var interceptor = CreateInterceptor();
|
||||
var context = CreateContext(SiteStreamMethod, siteHeader: null, authorizationHeader: null);
|
||||
|
||||
Assert.Equal("ok", await Invoke(interceptor, context));
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task CorrectSiteAndKey_IsAccepted()
|
||||
{
|
||||
var interceptor = CreateInterceptor();
|
||||
var context = CreateContext(ControlMethod, SiteA, $"Bearer {SiteAKey}");
|
||||
|
||||
Assert.Equal("ok", await Invoke(interceptor, context));
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task MissingSiteHeader_IsDenied()
|
||||
{
|
||||
// Fail-closed: no header means no per-site key to verify against, so there is nothing to
|
||||
// pass through TO. A present bearer token does not rescue it.
|
||||
var interceptor = CreateInterceptor();
|
||||
var context = CreateContext(ControlMethod, siteHeader: null, $"Bearer {SiteAKey}");
|
||||
|
||||
var ex = await Assert.ThrowsAsync<RpcException>(() => Invoke(interceptor, context));
|
||||
Assert.Equal(StatusCode.PermissionDenied, ex.StatusCode);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task BlankSiteHeader_IsDenied()
|
||||
{
|
||||
var interceptor = CreateInterceptor();
|
||||
var context = CreateContext(ControlMethod, siteHeader: " ", $"Bearer {SiteAKey}");
|
||||
|
||||
var ex = await Assert.ThrowsAsync<RpcException>(() => Invoke(interceptor, context));
|
||||
Assert.Equal(StatusCode.PermissionDenied, ex.StatusCode);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task UnknownSite_IsDenied_NotPassedThrough()
|
||||
{
|
||||
// The provider throws for an unknown site; the interceptor must turn that into a denial,
|
||||
// never swallow it and let the call proceed unauthenticated.
|
||||
var interceptor = CreateInterceptor();
|
||||
var context = CreateContext(ControlMethod, "site-nonexistent", "Bearer whatever");
|
||||
|
||||
var ex = await Assert.ThrowsAsync<RpcException>(() => Invoke(interceptor, context));
|
||||
Assert.Equal(StatusCode.PermissionDenied, ex.StatusCode);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task NoBearerToken_IsDenied()
|
||||
{
|
||||
var interceptor = CreateInterceptor();
|
||||
var context = CreateContext(ControlMethod, SiteA, authorizationHeader: null);
|
||||
|
||||
var ex = await Assert.ThrowsAsync<RpcException>(() => Invoke(interceptor, context));
|
||||
Assert.Equal(StatusCode.PermissionDenied, ex.StatusCode);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task WrongKeyForTheSite_IsDenied()
|
||||
{
|
||||
// site-a presents site-b's key. Both keys are valid keys; the point is the token must
|
||||
// match the key for THIS site, not just be a key central knows.
|
||||
var interceptor = CreateInterceptor();
|
||||
var context = CreateContext(ControlMethod, SiteA, $"Bearer {SiteBKey}");
|
||||
|
||||
var ex = await Assert.ThrowsAsync<RpcException>(() => Invoke(interceptor, context));
|
||||
Assert.Equal(StatusCode.PermissionDenied, ex.StatusCode);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task EachSiteIsVerifiedAgainstItsOwnKey()
|
||||
{
|
||||
// site-b with site-b's key is accepted by the same interceptor instance that rejected
|
||||
// site-a-with-site-b's-key above — proving the per-site lookup, not a single shared key.
|
||||
var interceptor = CreateInterceptor();
|
||||
var context = CreateContext(ControlMethod, SiteB, $"Bearer {SiteBKey}");
|
||||
|
||||
Assert.Equal("ok", await Invoke(interceptor, context));
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void TheInterceptorHasExactlyOnePublicConstructor()
|
||||
{
|
||||
// Grpc.AspNetCore registers this interceptor BY TYPE; InterceptorRegistration.GetFactory()
|
||||
// throws "Multiple constructors accepting all given argument types have been found" the
|
||||
// moment a second public constructor is applicable, and the throw lands inside the
|
||||
// pipeline on every call — a gate that authorizes nothing while looking like a handler
|
||||
// bug. The explicit-prefix constructor is internal to keep it from recurring; this pins
|
||||
// that. (Same invariant as ControlPlaneAuthInterceptorTests.)
|
||||
var publicCtors = typeof(CentralControlAuthInterceptor).GetConstructors();
|
||||
|
||||
Assert.Single(publicCtors);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void DefaultGatedPrefixes_MatchTheRealCentralControlServicePath()
|
||||
{
|
||||
// A typo here disables the whole gate silently: every call would pass through. Pin it
|
||||
// against the generated service descriptor, not the proto text.
|
||||
var method = CentralControlService.Descriptor.FullName;
|
||||
Assert.Contains(
|
||||
CentralControlAuthInterceptor.DefaultGatedPrefixes,
|
||||
p => p == $"/{method}/");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,256 @@
|
||||
using Akka.Actor;
|
||||
using Google.Protobuf.WellKnownTypes;
|
||||
using Grpc.Core;
|
||||
using Grpc.Net.Client;
|
||||
using Microsoft.AspNetCore.Builder;
|
||||
using Microsoft.AspNetCore.Hosting;
|
||||
using Microsoft.AspNetCore.TestHost;
|
||||
using Microsoft.Extensions.DependencyInjection;
|
||||
using Microsoft.Extensions.Hosting;
|
||||
using Microsoft.Extensions.Logging.Abstractions;
|
||||
using Microsoft.Extensions.Options;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Audit;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Notification;
|
||||
using ZB.MOM.WW.ScadaBridge.Communication;
|
||||
using ZB.MOM.WW.ScadaBridge.Communication.Grpc;
|
||||
|
||||
namespace ZB.MOM.WW.ScadaBridge.Host.Tests;
|
||||
|
||||
/// <summary>
|
||||
/// End-to-end proof of the central-hosted gRPC control plane (T1A.2): the real
|
||||
/// <see cref="CentralControlAuthInterceptor"/> AND the real
|
||||
/// <see cref="CentralControlGrpcService"/> over a real gRPC stack, the service Asking a real
|
||||
/// (stub) <c>CentralCommunicationActor</c> stand-in.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// <para>
|
||||
/// The interceptor is registered <b>BY TYPE on <c>AddGrpc</c></b>, exactly as <c>Program.cs</c>
|
||||
/// does, and is deliberately NOT placed in DI as a singleton — pre-registering it lets DI hand
|
||||
/// the instance back and bypasses <c>Grpc.AspNetCore</c>'s own activation path, which is how the
|
||||
/// two-public-constructor defect escaped a green suite once before. This harness copies the
|
||||
/// shape of <see cref="ControlPlaneAuthEndToEndTests"/> for the same reason.
|
||||
/// </para>
|
||||
/// <para>
|
||||
/// Runs in-process over <see cref="TestServer"/>: no ports, no containers. A tiny Akka actor
|
||||
/// stands in for <c>CentralCommunicationActor</c> so the test never touches MSSQL or a cluster;
|
||||
/// the method paths, message types and mapper are the real generated/production ones.
|
||||
/// </para>
|
||||
/// </remarks>
|
||||
public class CentralControlEndToEndTests : IAsyncLifetime
|
||||
{
|
||||
private const string SiteA = "site-a";
|
||||
private const string SiteAKey = "site-a-preshared-key";
|
||||
private const string SiteB = "site-b";
|
||||
private const string SiteBKey = "site-b-preshared-key";
|
||||
|
||||
// The deterministic id the stub actor "accepts" for every ingest batch, so the ingest
|
||||
// bridge can be asserted without extracting ids out of a decoded AuditEvent.
|
||||
private static readonly Guid AcceptedId = Guid.Parse("11111111-1111-1111-1111-111111111111");
|
||||
|
||||
private IHost _host = null!;
|
||||
private TestServer _server = null!;
|
||||
private ActorSystem _actorSystem = null!;
|
||||
private CentralControlGrpcService _service = null!;
|
||||
|
||||
/// <inheritdoc />
|
||||
public async Task InitializeAsync()
|
||||
{
|
||||
_actorSystem = ActorSystem.Create("centralcontrol-e2e-test");
|
||||
var stub = _actorSystem.ActorOf(Props.Create(() => new StubCentralActor(AcceptedId)), "central-stub");
|
||||
|
||||
_service = new CentralControlGrpcService(
|
||||
NullLogger<CentralControlGrpcService>.Instance,
|
||||
Options.Create(new CommunicationOptions()));
|
||||
_service.SetReady(stub);
|
||||
|
||||
var pskProvider = new MapPskProvider(new Dictionary<string, string>
|
||||
{
|
||||
[SiteA] = SiteAKey,
|
||||
[SiteB] = SiteBKey,
|
||||
});
|
||||
|
||||
_host = await new HostBuilder()
|
||||
.ConfigureWebHost(web => web
|
||||
.UseTestServer()
|
||||
.ConfigureServices(services =>
|
||||
{
|
||||
// BY TYPE, and NOT also in DI — see the class remarks.
|
||||
services.AddGrpc(o => o.Interceptors.Add<CentralControlAuthInterceptor>());
|
||||
services.AddSingleton<ISitePskProvider>(pskProvider);
|
||||
services.AddSingleton(_service);
|
||||
})
|
||||
.Configure(app =>
|
||||
{
|
||||
app.UseRouting();
|
||||
app.UseEndpoints(e => e.MapGrpcService<CentralControlGrpcService>());
|
||||
}))
|
||||
.StartAsync();
|
||||
|
||||
_server = _host.GetTestServer();
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public async Task DisposeAsync()
|
||||
{
|
||||
await _host.StopAsync();
|
||||
_host.Dispose();
|
||||
await _actorSystem.Terminate();
|
||||
}
|
||||
|
||||
/// <summary>Builds a channel credentialed for <paramref name="siteId"/> with <paramref name="key"/>.</summary>
|
||||
private GrpcChannel Channel(string? key, string siteId)
|
||||
{
|
||||
var options = new GrpcChannelOptions { HttpHandler = _server.CreateHandler() };
|
||||
if (key is not null)
|
||||
{
|
||||
options.WithSiteCredentials(new FixedPskProvider(key), siteId);
|
||||
}
|
||||
return GrpcChannel.ForAddress(_server.BaseAddress, options);
|
||||
}
|
||||
|
||||
private CentralControlService.CentralControlServiceClient Client(string? key, string siteId)
|
||||
=> new(Channel(key, siteId));
|
||||
|
||||
// ---- Auth positives / negatives, all through the real pipeline ----
|
||||
|
||||
[Fact]
|
||||
public async Task CorrectSiteAndKey_ReachesTheService_OnAUnaryCall()
|
||||
{
|
||||
var client = Client(SiteAKey, SiteA);
|
||||
|
||||
var ack = await client.SubmitNotificationAsync(NewNotificationDto());
|
||||
|
||||
Assert.True(ack.Accepted);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task NoCredentialsAtAll_IsRejected_WithPermissionDenied()
|
||||
{
|
||||
var client = Client(key: null, SiteA);
|
||||
|
||||
var ex = await Assert.ThrowsAsync<RpcException>(
|
||||
async () => await client.SubmitNotificationAsync(new NotificationSubmitDto()));
|
||||
|
||||
Assert.Equal(StatusCode.PermissionDenied, ex.StatusCode);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task WrongKeyForTheSite_IsRejected_WithPermissionDenied()
|
||||
{
|
||||
// site-a presents site-b's (valid, but wrong-for-this-site) key.
|
||||
var client = Client(SiteBKey, SiteA);
|
||||
|
||||
var ex = await Assert.ThrowsAsync<RpcException>(
|
||||
async () => await client.SubmitNotificationAsync(new NotificationSubmitDto()));
|
||||
|
||||
Assert.Equal(StatusCode.PermissionDenied, ex.StatusCode);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task UnknownSite_IsRejected_WithPermissionDenied()
|
||||
{
|
||||
var client = Client("any-key", "site-nonexistent");
|
||||
|
||||
var ex = await Assert.ThrowsAsync<RpcException>(
|
||||
async () => await client.SubmitNotificationAsync(new NotificationSubmitDto()));
|
||||
|
||||
Assert.Equal(StatusCode.PermissionDenied, ex.StatusCode);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task TheAcceptAndRejectPathsAreDistinguishable()
|
||||
{
|
||||
// A gate whose accept and reject paths produce the same observable result is not a gate.
|
||||
// Correct key → the service answers (Accepted:true); wrong key → PermissionDenied,
|
||||
// never reaching the service. These are two different outcomes, which is the whole point.
|
||||
var ok = await Client(SiteAKey, SiteA).SubmitNotificationAsync(NewNotificationDto());
|
||||
Assert.True(ok.Accepted);
|
||||
|
||||
var ex = await Assert.ThrowsAsync<RpcException>(
|
||||
async () => await Client("wrong", SiteA).SubmitNotificationAsync(new NotificationSubmitDto()));
|
||||
Assert.Equal(StatusCode.PermissionDenied, ex.StatusCode);
|
||||
}
|
||||
|
||||
// ---- One RPC per shape: unary (above) + the ingest bridge ----
|
||||
|
||||
[Fact]
|
||||
public async Task IngestAuditEvents_DecodesTheBatch_AsksTheActor_EncodesTheAck()
|
||||
{
|
||||
var client = Client(SiteAKey, SiteA);
|
||||
|
||||
var batch = new AuditEventBatch();
|
||||
batch.Events.Add(NewAuditDto());
|
||||
batch.Events.Add(NewAuditDto());
|
||||
|
||||
var ack = await client.IngestAuditEventsAsync(batch);
|
||||
|
||||
// The stub actor accepts one deterministic id per non-empty batch; its presence proves
|
||||
// the full DTO→command→Ask→reply→ack bridge ran, gated call and all.
|
||||
Assert.Contains(AcceptedId.ToString(), ack.AcceptedEventIds);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task IngestAuditEvents_EmptyBatch_ShortCircuits_WithoutAskingTheActor()
|
||||
{
|
||||
// Even the empty-batch fast path is behind the gate — it still needs a valid key.
|
||||
var client = Client(SiteAKey, SiteA);
|
||||
|
||||
var ack = await client.IngestAuditEventsAsync(new AuditEventBatch());
|
||||
|
||||
Assert.Empty(ack.AcceptedEventIds);
|
||||
}
|
||||
|
||||
// The mapper reads SiteEnqueuedAt unconditionally; only DTOs that clear the gate reach it,
|
||||
// so the accept-path tests carry a timestamp while the negative tests can pass a bare DTO.
|
||||
private static NotificationSubmitDto NewNotificationDto() => new()
|
||||
{
|
||||
NotificationId = Guid.NewGuid().ToString(),
|
||||
ListName = "ops",
|
||||
SiteEnqueuedAt = Timestamp.FromDateTime(
|
||||
DateTime.SpecifyKind(new DateTime(2026, 5, 20, 10, 0, 0), DateTimeKind.Utc)),
|
||||
};
|
||||
|
||||
private static AuditEventDto NewAuditDto() => new()
|
||||
{
|
||||
EventId = Guid.NewGuid().ToString(),
|
||||
OccurredAtUtc = Timestamp.FromDateTime(
|
||||
DateTime.SpecifyKind(new DateTime(2026, 5, 20, 10, 0, 0), DateTimeKind.Utc)),
|
||||
Channel = "ApiOutbound",
|
||||
Kind = "ApiCall",
|
||||
Status = "Delivered",
|
||||
SourceSiteId = SiteA,
|
||||
};
|
||||
|
||||
private sealed class FixedPskProvider(string key) : ISitePskProvider
|
||||
{
|
||||
public ValueTask<string> GetAsync(string siteId, CancellationToken ct) => new(key);
|
||||
public void Invalidate(string siteId) { }
|
||||
}
|
||||
|
||||
private sealed class MapPskProvider(IReadOnlyDictionary<string, string> keys) : ISitePskProvider
|
||||
{
|
||||
public ValueTask<string> GetAsync(string siteId, CancellationToken ct)
|
||||
=> keys.TryGetValue(siteId, out var key)
|
||||
? new ValueTask<string>(key)
|
||||
: throw new InvalidOperationException($"no key for '{siteId}'");
|
||||
|
||||
public void Invalidate(string siteId) { }
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Minimal stand-in for <c>CentralCommunicationActor</c>: answers the two RPC shapes this
|
||||
/// test exercises. Replies straight to the Ask's temp sender, exactly as the real actor's
|
||||
/// Forward/PipeTo paths do.
|
||||
/// </summary>
|
||||
private sealed class StubCentralActor : ReceiveActor
|
||||
{
|
||||
public StubCentralActor(Guid acceptedId)
|
||||
{
|
||||
Receive<NotificationSubmit>(msg =>
|
||||
Sender.Tell(new NotificationSubmitAck(msg.NotificationId, Accepted: true, Error: null)));
|
||||
|
||||
Receive<IngestAuditEventsCommand>(_ =>
|
||||
Sender.Tell(new IngestAuditEventsReply(new[] { acceptedId })));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,56 @@
|
||||
using Xunit;
|
||||
|
||||
namespace ZB.MOM.WW.ScadaBridge.Host.Tests;
|
||||
|
||||
/// <summary>
|
||||
/// Pins <see cref="Program.ParseHttpBindPorts"/>, which re-declares central's HTTP surface
|
||||
/// after the gRPC listener switches Kestrel into explicit-endpoints mode.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// This exists because the regression it guards is invisible to every other test: a central
|
||||
/// node that binds only the gRPC port boots clean, joins the cluster and serves gRPC, while
|
||||
/// the Central UI, Management/Inbound API and <c>/health/*</c> are silently dead. TestServer
|
||||
/// never binds real Kestrel, so the parser is the one seam that can be asserted in-process.
|
||||
/// The rig caught the original defect; this keeps it caught.
|
||||
/// </remarks>
|
||||
public class CentralHttpBindPortsTests
|
||||
{
|
||||
[Theory]
|
||||
[InlineData("http://+:5000", new[] { 5000 })]
|
||||
[InlineData("http://0.0.0.0:5000", new[] { 5000 })]
|
||||
[InlineData("http://[::]:5000", new[] { 5000 })]
|
||||
[InlineData("http://localhost:5000", new[] { 5000 })]
|
||||
[InlineData("http://*:8085", new[] { 8085 })]
|
||||
[InlineData("http://+:5000;http://+:5001", new[] { 5000, 5001 })]
|
||||
[InlineData("http://+:5000 ; http://+:5001", new[] { 5000, 5001 })]
|
||||
[InlineData("http://+:5000;http://+:5000", new[] { 5000 })] // de-duped
|
||||
public void ParsesPortsFromServerUrls(string urls, int[] expected)
|
||||
{
|
||||
Assert.Equal(expected, Program.ParseHttpBindPorts(urls));
|
||||
}
|
||||
|
||||
[Theory]
|
||||
[InlineData("https://+:443", 443)] // scheme default when no explicit port
|
||||
[InlineData("http://+:80", 80)]
|
||||
public void HandlesSchemeDefaultAndExplicitPort(string url, int expected)
|
||||
{
|
||||
Assert.Equal(new[] { expected }, Program.ParseHttpBindPorts(url));
|
||||
}
|
||||
|
||||
[Theory]
|
||||
[InlineData(null)]
|
||||
[InlineData("")]
|
||||
[InlineData(" ")]
|
||||
public void EmptyOrNull_YieldsNoPorts(string? urls)
|
||||
{
|
||||
Assert.Empty(Program.ParseHttpBindPorts(urls));
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void UnparseableEntry_IsSkipped_NotThrown()
|
||||
{
|
||||
// A malformed URL must not take the node down at boot; the good one still binds.
|
||||
var ports = Program.ParseHttpBindPorts("not-a-url;http://+:5000");
|
||||
Assert.Equal(new[] { 5000 }, ports);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,373 @@
|
||||
using System.Collections.Concurrent;
|
||||
using Akka.Actor;
|
||||
using Microsoft.AspNetCore.Builder;
|
||||
using Microsoft.AspNetCore.Hosting;
|
||||
using Microsoft.AspNetCore.TestHost;
|
||||
using Microsoft.Extensions.DependencyInjection;
|
||||
using Microsoft.Extensions.Hosting;
|
||||
using Microsoft.Extensions.Logging.Abstractions;
|
||||
using Microsoft.Extensions.Options;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Messages.Notification;
|
||||
using ZB.MOM.WW.ScadaBridge.Communication;
|
||||
using ZB.MOM.WW.ScadaBridge.Communication.Grpc;
|
||||
|
||||
namespace ZB.MOM.WW.ScadaBridge.Host.Tests;
|
||||
|
||||
/// <summary>
|
||||
/// T1A.3: <see cref="GrpcCentralTransport"/> + <see cref="CentralChannelProvider"/> over a real
|
||||
/// gRPC stack (two in-process <see cref="TestServer"/> central nodes, the real
|
||||
/// <see cref="CentralControlGrpcService"/> and <see cref="CentralControlAuthInterceptor"/>). Proves
|
||||
/// the sticky failover/failback policy, the PSK + site-header attachment, the per-call deadline,
|
||||
/// and — the hard rule — no cross-node retry on <c>DeadlineExceeded</c>.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// A "down" node is modelled by a <see cref="ToggleHandler"/> that throws before reaching the
|
||||
/// TestServer, so BOTH the unary call and the failback <c>Heartbeat</c> probe see it as
|
||||
/// <c>Unavailable</c> — the honest shape of a refused connection, and the only class the transport
|
||||
/// fails over on. Readiness is always set, so a node that is "up" answers everything.
|
||||
/// </remarks>
|
||||
public class GrpcCentralTransportTests : IAsyncLifetime
|
||||
{
|
||||
private const string SiteA = "site-a";
|
||||
private const string SiteAKey = "site-a-preshared-key";
|
||||
private const string EndpointA = "http://central-a/";
|
||||
private const string EndpointB = "http://central-b/";
|
||||
|
||||
private ActorSystem _system = null!;
|
||||
private CentralNode _nodeA = null!;
|
||||
private CentralNode _nodeB = null!;
|
||||
|
||||
/// <inheritdoc />
|
||||
public async Task InitializeAsync()
|
||||
{
|
||||
_system = ActorSystem.Create("grpc-central-transport-test");
|
||||
_nodeA = await CentralNode.StartAsync(_system, "A", SiteA, SiteAKey, repliesToSubmit: true);
|
||||
_nodeB = await CentralNode.StartAsync(_system, "B", SiteA, SiteAKey, repliesToSubmit: true);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public async Task DisposeAsync()
|
||||
{
|
||||
await _nodeA.DisposeAsync();
|
||||
await _nodeB.DisposeAsync();
|
||||
await _system.Terminate();
|
||||
}
|
||||
|
||||
private CentralChannelProvider NewProvider(string? pskKey = SiteAKey) => new(
|
||||
new[] { EndpointA, EndpointB },
|
||||
new FixedPskProvider(pskKey),
|
||||
SiteA,
|
||||
new CommunicationOptions(),
|
||||
NullLogger<CentralChannelProvider>.Instance,
|
||||
handlerFactory: HandlerFor,
|
||||
probeDeadline: TimeSpan.FromSeconds(2),
|
||||
backoffBase: TimeSpan.FromMilliseconds(50),
|
||||
backoffCap: TimeSpan.FromMilliseconds(200));
|
||||
|
||||
private HttpMessageHandler HandlerFor(string endpoint) => endpoint == EndpointA
|
||||
? new ToggleHandler(_nodeA.Server.CreateHandler(), () => _nodeA.IsUp)
|
||||
: new ToggleHandler(_nodeB.Server.CreateHandler(), () => _nodeB.IsUp);
|
||||
|
||||
private GrpcCentralTransport NewTransport(CentralChannelProvider provider, CommunicationOptions? options = null)
|
||||
=> new(provider, options ?? new CommunicationOptions(), NullLogger<GrpcCentralTransport>.Instance);
|
||||
|
||||
[Fact]
|
||||
public async Task HappyPath_ReachesThePreferredNode_AndRoutesTheAckBack()
|
||||
{
|
||||
using var provider = NewProvider();
|
||||
var transport = NewTransport(provider);
|
||||
var inbox = new Capture(_system);
|
||||
|
||||
transport.SubmitNotification(NewSubmit("n1"), inbox.Ref);
|
||||
|
||||
var ack = Assert.IsType<NotificationSubmitAck>(inbox.Receive(TimeSpan.FromSeconds(5)));
|
||||
Assert.True(ack.Accepted);
|
||||
Assert.Equal("n1", ack.NotificationId);
|
||||
Assert.Equal(0, provider.CurrentIndex); // stayed on preferred
|
||||
Assert.Equal(1, _nodeA.SubmitCount);
|
||||
Assert.Equal(0, _nodeB.SubmitCount);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task Sticky_StaysOnThePreferredNode_WhileHealthy()
|
||||
{
|
||||
using var provider = NewProvider();
|
||||
var transport = NewTransport(provider);
|
||||
|
||||
for (var i = 0; i < 4; i++)
|
||||
{
|
||||
var inbox = new Capture(_system);
|
||||
transport.SubmitNotification(NewSubmit($"n{i}"), inbox.Ref);
|
||||
Assert.IsType<NotificationSubmitAck>(inbox.Receive(TimeSpan.FromSeconds(5)));
|
||||
}
|
||||
|
||||
Assert.Equal(0, provider.CurrentIndex);
|
||||
Assert.Equal(4, _nodeA.SubmitCount);
|
||||
Assert.Equal(0, _nodeB.SubmitCount);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task Failover_FlipsToThePeer_WhenThePreferredIsUnavailable()
|
||||
{
|
||||
using var provider = NewProvider();
|
||||
var transport = NewTransport(provider);
|
||||
_nodeA.IsUp = false; // preferred refuses connections
|
||||
|
||||
var inbox = new Capture(_system);
|
||||
transport.SubmitNotification(NewSubmit("n1"), inbox.Ref);
|
||||
|
||||
var ack = Assert.IsType<NotificationSubmitAck>(inbox.Receive(TimeSpan.FromSeconds(5)));
|
||||
Assert.True(ack.Accepted);
|
||||
Assert.Equal(1, provider.CurrentIndex); // flipped to the peer
|
||||
Assert.Equal(0, _nodeA.SubmitCount);
|
||||
Assert.Equal(1, _nodeB.SubmitCount);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task Failback_ReturnsToThePreferred_OnceItIsReachableAgain()
|
||||
{
|
||||
using var provider = NewProvider();
|
||||
var transport = NewTransport(provider);
|
||||
|
||||
// Take the preferred down and drive one call so we flip to the peer + arm the failback probe.
|
||||
_nodeA.IsUp = false;
|
||||
var inbox = new Capture(_system);
|
||||
transport.SubmitNotification(NewSubmit("n1"), inbox.Ref);
|
||||
Assert.IsType<NotificationSubmitAck>(inbox.Receive(TimeSpan.FromSeconds(5)));
|
||||
Assert.Equal(1, provider.CurrentIndex);
|
||||
|
||||
// Bring the preferred back; the background probe should fail us back within a few backoffs.
|
||||
_nodeA.IsUp = true;
|
||||
await WaitUntil(() => provider.CurrentIndex == 0, TimeSpan.FromSeconds(5));
|
||||
Assert.Equal(0, provider.CurrentIndex);
|
||||
|
||||
// New calls resume on the preferred node.
|
||||
var inbox2 = new Capture(_system);
|
||||
transport.SubmitNotification(NewSubmit("n2"), inbox2.Ref);
|
||||
Assert.IsType<NotificationSubmitAck>(inbox2.Receive(TimeSpan.FromSeconds(5)));
|
||||
Assert.True(_nodeA.SubmitCount >= 1);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task PskAndSiteHeader_AreAttached_SoTheGatedCallReachesTheService()
|
||||
{
|
||||
// The service is gated by CentralControlAuthInterceptor; a call that reaches it (and gets
|
||||
// Accepted) proves both the bearer PSK and the x-scadabridge-site header were attached.
|
||||
using var provider = NewProvider(pskKey: SiteAKey);
|
||||
var transport = NewTransport(provider);
|
||||
var inbox = new Capture(_system);
|
||||
|
||||
transport.SubmitNotification(NewSubmit("n1"), inbox.Ref);
|
||||
|
||||
var ack = Assert.IsType<NotificationSubmitAck>(inbox.Receive(TimeSpan.FromSeconds(5)));
|
||||
Assert.True(ack.Accepted);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task WrongPsk_IsRejected_AndNotRetriedOnThePeer()
|
||||
{
|
||||
// PermissionDenied is not a connect failure — the transport surfaces it as a transient
|
||||
// Status.Failure without flipping to the peer.
|
||||
using var provider = NewProvider(pskKey: "the-wrong-key");
|
||||
var transport = NewTransport(provider);
|
||||
var inbox = new Capture(_system);
|
||||
|
||||
transport.SubmitNotification(NewSubmit("n1"), inbox.Ref);
|
||||
|
||||
Assert.IsType<Status.Failure>(inbox.Receive(TimeSpan.FromSeconds(5)));
|
||||
Assert.Equal(0, provider.CurrentIndex); // no flip
|
||||
Assert.Equal(0, _nodeB.SubmitCount); // peer never tried
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task DeadlineExceeded_IsNotRetriedOnThePeer()
|
||||
{
|
||||
// THE hard rule. Node A is UP but never replies, so the call deadlines. The transport must
|
||||
// surface Status.Failure and must NOT try node B (the call may already have executed).
|
||||
_nodeA.SetBlackHole();
|
||||
var shortDeadline = new CommunicationOptions { NotificationForwardTimeout = TimeSpan.FromMilliseconds(300) };
|
||||
|
||||
using var provider = NewProvider();
|
||||
var transport = NewTransport(provider, shortDeadline);
|
||||
var inbox = new Capture(_system);
|
||||
|
||||
transport.SubmitNotification(NewSubmit("n1"), inbox.Ref);
|
||||
|
||||
// A per-call deadline is applied (the call returns fast instead of hanging on the black hole).
|
||||
Assert.IsType<Status.Failure>(inbox.Receive(TimeSpan.FromSeconds(5)));
|
||||
Assert.Equal(0, provider.CurrentIndex); // no failover on a deadline
|
||||
Assert.Equal(0, _nodeB.SubmitCount); // peer never tried
|
||||
}
|
||||
|
||||
private static NotificationSubmit NewSubmit(string id) => new(
|
||||
NotificationId: id,
|
||||
ListName: "ops",
|
||||
Subject: "s",
|
||||
Body: "b",
|
||||
SourceSiteId: SiteA,
|
||||
SourceInstanceId: null,
|
||||
SourceScript: null,
|
||||
SiteEnqueuedAt: DateTimeOffset.UtcNow);
|
||||
|
||||
private static async Task WaitUntil(Func<bool> condition, TimeSpan timeout)
|
||||
{
|
||||
var deadline = DateTime.UtcNow + timeout;
|
||||
while (DateTime.UtcNow < deadline)
|
||||
{
|
||||
if (condition())
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
await Task.Delay(25);
|
||||
}
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// A raw message sink used as the transport's <c>replyTo</c>. Unlike Akka's <c>Inbox</c>, which
|
||||
/// rethrows a <see cref="Status.Failure"/>'s cause on receive, this captures every message
|
||||
/// verbatim so a test can assert on the <see cref="Status.Failure"/> itself.
|
||||
/// </summary>
|
||||
private sealed class Capture
|
||||
{
|
||||
private readonly BlockingCollection<object> _messages = new();
|
||||
|
||||
public Capture(ActorSystem system)
|
||||
{
|
||||
Ref = system.ActorOf(Props.Create(() => new CaptureActor(_messages)));
|
||||
}
|
||||
|
||||
public IActorRef Ref { get; }
|
||||
|
||||
public object Receive(TimeSpan timeout)
|
||||
=> _messages.TryTake(out var message, timeout)
|
||||
? message
|
||||
: throw new TimeoutException("No message captured within the timeout.");
|
||||
|
||||
private sealed class CaptureActor : ReceiveActor
|
||||
{
|
||||
public CaptureActor(BlockingCollection<object> messages) => ReceiveAny(messages.Add);
|
||||
}
|
||||
}
|
||||
|
||||
/// <summary>A gRPC channel handler that throws (a refused connection) while its node is "down".</summary>
|
||||
private sealed class ToggleHandler : DelegatingHandler
|
||||
{
|
||||
private readonly Func<bool> _isUp;
|
||||
|
||||
public ToggleHandler(HttpMessageHandler inner, Func<bool> isUp)
|
||||
{
|
||||
InnerHandler = inner;
|
||||
_isUp = isUp;
|
||||
}
|
||||
|
||||
protected override async Task<HttpResponseMessage> SendAsync(HttpRequestMessage request, CancellationToken cancellationToken)
|
||||
{
|
||||
if (!_isUp())
|
||||
{
|
||||
throw new HttpRequestException("simulated central node down");
|
||||
}
|
||||
|
||||
return await base.SendAsync(request, cancellationToken).ConfigureAwait(false);
|
||||
}
|
||||
}
|
||||
|
||||
private sealed class FixedPskProvider(string? key) : ISitePskProvider
|
||||
{
|
||||
public ValueTask<string> GetAsync(string siteId, CancellationToken ct)
|
||||
=> key is null ? throw new InvalidOperationException("no key") : new ValueTask<string>(key);
|
||||
|
||||
public void Invalidate(string siteId) { }
|
||||
}
|
||||
|
||||
/// <summary>One in-process central node: TestServer + real service/interceptor + a stub actor.</summary>
|
||||
private sealed class CentralNode : IAsyncDisposable
|
||||
{
|
||||
private IHost _host = null!;
|
||||
private IActorRef _stub = null!;
|
||||
private readonly StubCounters _counters = new();
|
||||
|
||||
public TestServer Server { get; private set; } = null!;
|
||||
public volatile bool IsUp = true;
|
||||
public int SubmitCount => _counters.Submits;
|
||||
|
||||
public static async Task<CentralNode> StartAsync(
|
||||
ActorSystem system, string label, string site, string key, bool repliesToSubmit)
|
||||
{
|
||||
var node = new CentralNode();
|
||||
node._stub = system.ActorOf(
|
||||
Props.Create(() => new StubCentralActor(node._counters, repliesToSubmit)), $"stub-{label}");
|
||||
|
||||
var service = new CentralControlGrpcService(
|
||||
NullLogger<CentralControlGrpcService>.Instance,
|
||||
Options.Create(new CommunicationOptions()));
|
||||
service.SetReady(node._stub);
|
||||
|
||||
var psk = new MapPskProvider(new Dictionary<string, string> { [site] = key });
|
||||
|
||||
node._host = await new HostBuilder()
|
||||
.ConfigureWebHost(web => web
|
||||
.UseTestServer()
|
||||
.ConfigureServices(services =>
|
||||
{
|
||||
services.AddGrpc(o => o.Interceptors.Add<CentralControlAuthInterceptor>());
|
||||
services.AddSingleton<ISitePskProvider>(psk);
|
||||
services.AddSingleton(service);
|
||||
})
|
||||
.Configure(app =>
|
||||
{
|
||||
app.UseRouting();
|
||||
app.UseEndpoints(e => e.MapGrpcService<CentralControlGrpcService>());
|
||||
}))
|
||||
.StartAsync();
|
||||
|
||||
node.Server = node._host.GetTestServer();
|
||||
return node;
|
||||
}
|
||||
|
||||
/// <summary>Switches the node's actor to a black hole that counts but never replies.</summary>
|
||||
public void SetBlackHole() => _counters.BlackHole = true;
|
||||
|
||||
public async ValueTask DisposeAsync()
|
||||
{
|
||||
await _host.StopAsync();
|
||||
_host.Dispose();
|
||||
}
|
||||
|
||||
private sealed class StubCounters
|
||||
{
|
||||
private int _submits;
|
||||
public int Submits => Volatile.Read(ref _submits);
|
||||
public void IncrementSubmits() => Interlocked.Increment(ref _submits);
|
||||
public volatile bool BlackHole;
|
||||
}
|
||||
|
||||
private sealed class StubCentralActor : ReceiveActor
|
||||
{
|
||||
public StubCentralActor(StubCounters counters, bool repliesToSubmit)
|
||||
{
|
||||
Receive<NotificationSubmit>(msg =>
|
||||
{
|
||||
counters.IncrementSubmits();
|
||||
if (repliesToSubmit && !counters.BlackHole)
|
||||
{
|
||||
Sender.Tell(new NotificationSubmitAck(msg.NotificationId, Accepted: true, Error: null));
|
||||
}
|
||||
});
|
||||
|
||||
// Heartbeat lands here as a Tell (the failback probe); ignore it, no reply expected.
|
||||
ReceiveAny(_ => { });
|
||||
}
|
||||
}
|
||||
|
||||
private sealed class MapPskProvider(IReadOnlyDictionary<string, string> keys) : ISitePskProvider
|
||||
{
|
||||
public ValueTask<string> GetAsync(string siteId, CancellationToken ct)
|
||||
=> keys.TryGetValue(siteId, out var key)
|
||||
? new ValueTask<string>(key)
|
||||
: throw new InvalidOperationException($"no key for '{siteId}'");
|
||||
|
||||
public void Invalidate(string siteId) { }
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user