{ "ScadaBridge": { "Node": { "Role": "Site", "NodeName": "node-a", "NodeHostname": "scadabridge-site-a-a", "SiteId": "site-a", "RemotingPort": 8082, "GrpcPort": 8083, "MetricsPort": 8084 }, "Cluster": { "SeedNodes": [ "akka.tcp://scadabridge@scadabridge-site-a-a:8082", "akka.tcp://scadabridge@scadabridge-site-a-b:8082" ], "SplitBrainResolverStrategy": "auto-down", "StableAfter": "00:00:15", "HeartbeatInterval": "00:00:02", "FailureDetectionThreshold": "00:00:10", "MinNrOfMembers": 1 }, "Database": { // Migration-only as of LocalDb Phase 2. The site config tables now live in the // consolidated LocalDb database (LocalDb:Path). SiteDbPath is read once at boot to drain // a pre-Phase-2 scadabridge.db, and is unused after that - keep it until this node has // started once. "SiteDbPath": "/app/data/scadabridge.db" }, "DataConnection": { "ReconnectInterval": "00:00:05", "TagResolutionRetryInterval": "00:00:10", "WriteTimeout": "00:00:30", "SeedReadTimeout": "00:00:30" }, "StoreAndForward": { // Migration-only as of LocalDb Phase 2. The store-and-forward buffer now lives in the // consolidated LocalDb database (LocalDb:Path) as the replicated sf_messages table. // SqliteDbPath is read once at boot by SiteLocalDbLegacyMigrator to drain a pre-Phase-2 // file, and is unused after that - keep it until this node has started once. "SqliteDbPath": "/app/data/store-and-forward.db" }, "Communication": { // DEV-ONLY control-plane preshared key — NOT a real secret. Must be // IDENTICAL on both nodes of the pair and match the central-side entry in // ScadaBridge__Communication__SitePsks__ (docker-compose.yml). // Production supplies this as ${secret:SB-GRPC-PSK-}. Without it the // node fails StartupValidator: the gate is fail-closed, so an unset key would // refuse every SiteStream call while the node still looked healthy. "GrpcPsk": "dev-grpc-psk-docker-site-a", "CentralGrpcEndpoints": [ "http://scadabridge-central-a:8083", "http://scadabridge-central-b:8083" ], "DeploymentTimeout": "00:02:00", "LifecycleTimeout": "00:00:30", "QueryTimeout": "00:00:30", "TransportHeartbeatInterval": "00:00:05", "TransportFailureThreshold": "00:00:15" }, "HealthMonitoring": { "ReportInterval": "00:00:30", "OfflineTimeout": "00:01:00" }, "SiteEventLog": { "RetentionDays": 30, "MaxStorageMb": 1024, "PurgeScheduleCron": "0 2 * * *" }, "Notification": {}, "Logging": { "MinimumLevel": "Information" } }, // Consolidated site database (LocalDb Phase 1): OperationTracking + site_events. // On the mounted /app/data volume so it survives container recreate - unlike the // legacy site-tracking.db / site_events.db, which defaulted to CWD-relative paths // outside the volume and were lost on every recreate. // Replication is opt-in and configured separately; absent = local-only. "LocalDb": { "Path": "/app/data/site-localdb.db", // Site-a is the rig's REPLICATED pair; site-b and site-c deliberately stay // unreplicated so the default-OFF posture is proven side-by-side on one rig. // // This node is the initiator: it dials the peer. Replication is still // bidirectional - the passive node's writes flow back over the same stream - // so only one side needs PeerAddress. Port 8083 is the existing site gRPC // listener (h2c); the sync endpoint shares it, no new port. // // The ApiKey must be IDENTICAL on both nodes: the host's // LocalDbSyncAuthInterceptor is fail-closed, so a mismatch (or a missing key) // rejects every sync stream. A plain dev key here matches the rig's existing // posture; PRODUCTION uses a "${secret:...}" reference resolved by the // pre-host secret expander. "Replication": { "PeerAddress": "http://scadabridge-site-a-b:8083", "ApiKey": "dev-site-a-localdb-sync-key", // ---- Phase 2 sizing, from the Task 1 rig soak (not from the defaults) ---- // // MaxBatchSize (default 500) is a ROW count, not a byte budget, so the batch // size in bytes is set by the widest replicated column. That is // deployed_configurations.config_json: ~721 B on this rig, but up to ~60-70 KB // in production (measured, Task 1) - and 70 KB x 500 is ~35 MB against gRPC's // 4 MB default receive limit. 16 keeps a worst-case batch near 1.1 MB. "MaxBatchSize": 16, // Backlog caps bound the oplog while the peer is offline. Exceeding them is // NOT data loss: the oplog is pruned to the ceiling and needs_snapshot is set, // so the peer catches up by snapshot resync instead of incrementally. That // makes tighter-than-default correct here - it trades a rare full resync for a // bounded file. Sized from the soak's 0.80 sf_messages rows/sec (the only // non-zero writer measured): ~69k rows/day, so 2 days is ~138k. 250,000 leaves // room for burst without approaching the 1,000,000 default. "MaxOplogRows": 250000, "MaxOplogAge": "2.00:00:00" } } }