Files
ScadaBridge/src/ZB.MOM.WW.ScadaBridge.Communication/Protos/central_control.proto
T
Joseph Doherty 63c16d6912 refactor(comm): retire ClusterClient naming after the gRPC cutover
Phase 4 of the ClusterClient -> gRPC migration deleted the Akka transports
but left the naming behind: `ClusterClientSiteAuditClient` was transport-
agnostic and worked unchanged, so it survived the deletion under a name
that now describes a transport the repo no longer has. Same for a scatter
of doc-comments still framing gRPC as "the new transport" beside an Akka
one that is gone.

Renames it to `SiteCommunicationAuditClient` (and its test file) and
rewrites the stale comments to describe the single transport that exists.
Also tightens CLAUDE.md: drops the self-describing directory listing and
the 27-component enumeration in favour of the non-obvious parts only.

Behaviour-neutral: names and prose only. Recorded as the Phase 4
follow-up in docs/plans/2026-07-22-clusterclient-to-grpc-plan.md.
2026-07-27 15:40:01 -04:00

248 lines
10 KiB
Protocol Buffer

syntax = "proto3";
option csharp_namespace = "ZB.MOM.WW.ScadaBridge.Communication.Grpc";
package scadabridge.centralcontrol.v1;
import "google/protobuf/empty.proto";
import "google/protobuf/timestamp.proto";
import "google/protobuf/wrappers.proto";
// The two ingest RPCs deliberately REUSE the batch/ack messages already defined
// for the site-hosted SiteStreamService rather than redeclaring them. The site
// telemetry actor builds AuditEventBatch / CachedTelemetryBatch today and hands
// them to ISiteStreamAuditClient; duplicating the shapes here would fork one
// wire contract into two that must be kept in lockstep by hand.
import "Protos/sitestream.proto";
// Central-hosted control plane (Phase 1A of the ClusterClient→gRPC migration).
//
// Direction: SITE is the client, CENTRAL is the server — the inverse of
// SiteStreamService, where central dials the site. That asymmetry is deliberate
// and mirrors the direction the Akka ClusterClient traffic used to flow: these
// seven calls are exactly the seven messages SiteCommunicationActor sends to
// /user/central-communication.
//
// Every call is gated by ControlPlaneAuthInterceptor: `authorization: Bearer <psk>`
// plus the `x-scadabridge-site` metadata header naming which site's preshared key
// central must verify against.
service CentralControlService {
// Store-and-forward handoff of one notification for central delivery. The
// ack is idempotent on notification_id — a duplicate submit after a lost ack
// must not produce a second delivery.
rpc SubmitNotification(NotificationSubmitDto) returns (NotificationSubmitAckDto);
// Notify.Status(id) round-trip for a notification that has already left the
// site buffer. `found = false` sends the caller back to the site-local buffer
// to decide Forwarding vs Unknown.
rpc QueryNotificationStatus(NotificationStatusQueryDto) returns (NotificationStatusResponseDto);
// Audit Log (#23) push telemetry. Reuses the SiteStreamService messages: the
// batch a site drains from its SQLite hot path is byte-identical whichever
// transport carries it.
rpc IngestAuditEvents(sitestream.AuditEventBatch) returns (sitestream.IngestAck);
// Audit Log (#23) M3 combined cached-call telemetry (audit row + SiteCalls
// operational upsert, written in one central transaction).
rpc IngestCachedTelemetry(sitestream.CachedTelemetryBatch) returns (sitestream.IngestAck);
// Node-startup self-heal: the node's local deployed inventory in, fetch
// tokens for whatever it is missing or stale out.
rpc ReconcileSite(ReconcileSiteRequestDto) returns (ReconcileSiteResponseDto);
// Periodic site health report (30 s cadence). The ack makes delivery
// observable end-to-end so the sender can restore its per-interval counters
// when a report is lost.
rpc ReportSiteHealth(SiteHealthReportDto) returns (SiteHealthReportAckDto);
// Application heartbeat. Returns Empty because the message is
// fire-and-forget: nothing on the site consumes a reply, and a failure here
// must never surface as a fault on the heartbeat timer path.
rpc Heartbeat(HeartbeatDto) returns (google.protobuf.Empty);
}
// ---------------------------------------------------------------------------
// Notification Outbox (#21)
// ---------------------------------------------------------------------------
// Site -> Central: submit a buffered notification for central delivery.
// Mirrors Commons NotificationSubmit.
message NotificationSubmitDto {
string notification_id = 1; // GUID string, the idempotency key
string list_name = 2;
string subject = 3;
string body = 4;
string source_site_id = 5;
string source_instance_id = 6; // empty string represents null
string source_script = 7; // empty string represents null
google.protobuf.Timestamp site_enqueued_at = 8;
string origin_execution_id = 9; // GUID string; empty represents null
string origin_parent_execution_id = 10; // GUID string; empty represents null
string source_node = 11; // empty string represents null
}
// Central -> Site: ack sent after the Notifications row is persisted.
message NotificationSubmitAckDto {
string notification_id = 1;
bool accepted = 2;
string error = 3; // empty string represents null
}
// Site -> Central: Notify.Status(id) lookup against the central outbox.
message NotificationStatusQueryDto {
string correlation_id = 1;
string notification_id = 2;
}
// Central -> Site: current central delivery state for a queried notification.
message NotificationStatusResponseDto {
string correlation_id = 1;
bool found = 2;
string status = 3;
int32 retry_count = 4;
string last_error = 5; // empty string represents null
google.protobuf.Timestamp delivered_at = 6; // absent when null
}
// ---------------------------------------------------------------------------
// Startup reconciliation (Deployment Manager)
// ---------------------------------------------------------------------------
// Site -> Central: the node's local deployed inventory at startup.
message ReconcileSiteRequestDto {
string site_identifier = 1;
string node_id = 2;
// Instance unique name -> revision hash of the config the node currently holds.
map<string, string> local_name_to_revision_hash = 3;
}
// Central -> Site: the gap the node must (re)fetch, plus orphans to log.
message ReconcileSiteResponseDto {
repeated ReconcileGapItemDto gap = 1;
repeated string orphan_names = 2;
string central_fetch_base_url = 3;
}
// One instance the node must (re)fetch, with a freshly-minted short-TTL token.
message ReconcileGapItemDto {
string instance_unique_name = 1;
string deployment_id = 2;
string revision_hash = 3;
bool is_enabled = 4;
string fetch_token = 5;
}
// ---------------------------------------------------------------------------
// Health Monitoring (#11)
// ---------------------------------------------------------------------------
// Wire form of the Commons ConnectionHealth enum.
//
// CONNECTION_HEALTH_UNSPECIFIED exists only to keep the proto3 zero value from
// meaning something. Mapping Connected onto 0 would make an absent/garbled
// value decode as "healthy", which is precisely the wrong direction to fail;
// the mapper decodes UNSPECIFIED as Error instead and never emits it.
enum ConnectionHealthEnum {
CONNECTION_HEALTH_UNSPECIFIED = 0;
CONNECTION_HEALTH_CONNECTED = 1;
CONNECTION_HEALTH_DISCONNECTED = 2;
CONNECTION_HEALTH_CONNECTING = 3;
CONNECTION_HEALTH_ERROR = 4;
}
message TagResolutionStatusDto {
int32 total_subscribed = 1;
int32 successfully_resolved = 2;
}
message TagQualityCountsDto {
int32 good = 1;
int32 bad = 2;
int32 uncertain = 3;
}
message NodeStatusDto {
string hostname = 1;
bool is_online = 2;
string role = 3;
}
// Point-in-time snapshot of the site-local SQLite audit queue.
message SiteAuditBacklogSnapshotDto {
int32 pending_count = 1;
google.protobuf.Timestamp oldest_pending_utc = 2; // absent when the queue is empty
int64 on_disk_bytes = 3;
}
// The three collection wrappers below exist so null and empty stay
// distinguishable. proto3 cannot express presence on a `repeated` or `map`
// field — an unset one and an empty one are the same bytes — but the
// corresponding SiteHealthReport members are genuinely nullable
// (SiteHealthCollector emits `ClusterNodes: _clusterNodes?.ToList()`), and the
// central health surface reads null as "this producer doesn't report the
// signal" rather than "the signal is empty". Wrapping in a message restores
// message presence and makes the distinction survive the round-trip.
message ConnectionEndpointMapDto {
map<string, string> entries = 1;
}
message TagQualityMapDto {
map<string, TagQualityCountsDto> entries = 1;
}
message NodeStatusListDto {
repeated NodeStatusDto nodes = 1;
}
// Site -> Central: periodic site health report. Mirrors Commons SiteHealthReport.
// Additive-only evolution: field numbers are never reused.
message SiteHealthReportDto {
string site_id = 1;
int64 sequence_number = 2;
google.protobuf.Timestamp report_timestamp = 3;
map<string, ConnectionHealthEnum> data_connection_statuses = 4;
map<string, TagResolutionStatusDto> tag_resolution_counts = 5;
int32 script_error_count = 6;
int32 alarm_evaluation_error_count = 7;
map<string, int32> store_and_forward_buffer_depths = 8;
int32 dead_letter_count = 9;
int32 deployed_instance_count = 10;
int32 enabled_instance_count = 11;
int32 disabled_instance_count = 12;
string node_role = 13;
string node_hostname = 14;
ConnectionEndpointMapDto data_connection_endpoints = 15; // absent when null
TagQualityMapDto data_connection_tag_quality = 16; // absent when null
int32 parked_message_count = 17;
NodeStatusListDto cluster_nodes = 18; // absent when null
int32 site_audit_write_failures = 19;
int32 audit_redaction_failure = 20;
SiteAuditBacklogSnapshotDto site_audit_backlog = 21; // absent when no data yet
int64 site_event_log_write_failures = 22;
google.protobuf.DoubleValue oldest_parked_message_age_seconds = 23; // absent when nothing parked
int32 script_queue_depth = 24;
int32 script_busy_threads = 25;
google.protobuf.DoubleValue script_oldest_busy_age_seconds = 26; // absent when the pool is idle
// Nullable on purpose: absent means "replication not wired on this node",
// which is NOT the same as false ("wired but currently disconnected").
google.protobuf.BoolValue local_db_replication_connected = 27;
// Absent means UNKNOWN, never zero — a failed backlog read rendered as 0
// would report a broken replication pair as perfectly healthy.
google.protobuf.Int64Value local_db_oplog_backlog = 28;
}
// Central -> Site: health report ack, so a lost report is observable.
message SiteHealthReportAckDto {
string site_id = 1;
int64 sequence_number = 2;
bool accepted = 3;
string error = 4; // empty string represents null
}
// Site -> Central: application heartbeat (fire-and-forget; reply is Empty).
message HeartbeatDto {
string site_id = 1;
string node_hostname = 2;
bool is_active = 3;
google.protobuf.Timestamp timestamp = 4;
}