using Microsoft.Extensions.DependencyInjection.Extensions;
using Microsoft.Extensions.Options;
using ZB.MOM.WW.ScadaBridge.AuditLog;
using ZB.MOM.WW.ScadaBridge.ClusterInfrastructure;
using ZB.MOM.WW.ScadaBridge.Communication;
using ZB.MOM.WW.ScadaBridge.Commons.Interfaces.Services;
using ZB.MOM.WW.ScadaBridge.Commons.Observability;
using ZB.MOM.WW.ScadaBridge.DataConnectionLayer;
using ZB.MOM.WW.ScadaBridge.ExternalSystemGateway;
using ZB.MOM.WW.ScadaBridge.HealthMonitoring;
using ZB.MOM.WW.ScadaBridge.Host.Actors;
using ZB.MOM.WW.ScadaBridge.Host.Health;
using ZB.MOM.WW.ScadaBridge.NotificationService;
using ZB.MOM.WW.ScadaBridge.SiteEventLogging;
using ZB.MOM.WW.ScadaBridge.SiteRuntime;
using ZB.MOM.WW.ScadaBridge.StoreAndForward;
using ZB.MOM.WW.LocalDb;
using ZB.MOM.WW.LocalDb.Replication;
using ZB.MOM.WW.Secrets.DependencyInjection;
using ZB.MOM.WW.Telemetry;
namespace ZB.MOM.WW.ScadaBridge.Host;
///
/// Extracted site-role DI registrations so both Program.cs and tests
/// use the same composition root.
///
public static class SiteServiceRegistration
{
///
/// Every OpenTelemetry is told to observe.
///
///
/// ZbTelemetryOptions.Meters is an ALLOWLIST, and an unlisted meter's instruments
/// are simply never observed — no error, no warning, no missing-metric signal anywhere.
/// The LocalDb replication meter was silently absent from the rig's /metrics
/// scrape until the Phase 1 live gate went looking for it (the plan had assumed it
/// flowed through automatically). Hoisted to a named constant so the list is
/// assertable; adding a meter anywhere in the product means adding it here.
///
public static readonly string[] ObservedMeters =
[
ScadaBridgeTelemetry.MeterName,
// Emitted only on site nodes; harmless on central, which never records against it.
LocalDbMetrics.MeterName,
];
/// Registers all DI services required for the site role.
/// The service collection to register into.
/// Application configuration for options binding.
public static void Configure(IServiceCollection services, IConfiguration config)
{
// Shared components
services.AddClusterInfrastructure();
services.AddCommunication();
services.AddSiteHealthMonitoring();
services.AddExternalSystemGateway();
// AddNotificationService() is intentionally NOT registered on the site path.
// Sites no longer deliver notifications over SMTP — a buffered notification is
// forwarded to the central cluster (via NotificationForwarder / SiteCommunicationActor),
// and central owns SMTP delivery through the Notification Outbox. The SMTP machinery
// (OAuth2TokenService, ISmtpClientWrapper) has no consumer on a site node.
// Health report transport: sends SiteHealthReport to SiteCommunicationActor via Akka
services.AddSingleton();
services.AddSingleton();
// Site-only components — AddSiteRuntime registers SiteStorageService and the
// site-local repository implementations (IExternalSystemRepository,
// INotificationRepository). It takes no connection string any more:
// SiteStorageService persists to the consolidated LocalDb database registered
// just below (LocalDb:Path). ScadaBridge:Database:SiteDbPath survives only as
// the legacy migrator's source location.
services.AddSiteRuntime();
// Consolidated site database (LocalDb Phase 1). Holds OperationTracking and
// site_events as replicated tables so the pair stops losing them on failover.
//
// Storage is registered UNCONDITIONALLY and needs no flag: with no peer
// configured, LocalDb is simply a local SQLite file and the replication
// initiator idles.
//
// Design: scadaproj docs/plans/2026-07-19-scadabridge-localdb-design.md
//
// The parent directory of LocalDb:Path is created by the library as of 0.1.1.
// ScadaBridge carried a SiteLocalDbDirectory shim for it during Phase 2, because
// SQLite creates the database file on demand but not its directory and
// SqliteLocalDb opens the file eagerly — a missing directory was a hard boot
// failure, and the default site path is the relative "./data/site-localdb.db".
// The shim is gone; SiteLocalDbDirectoryTests still pins the outcome here, since
// what this host needs is the guarantee, not any particular owner of it.
services.AddZbLocalDb(config, db => SiteLocalDbSetup.OnReady(db, config));
// The replication engine, likewise unconditional but INERT by default: with no
// LocalDb:Replication:PeerAddress the initiating SyncBackgroundService starts and
// immediately idles, and nothing dials the passive endpoint. A site pair
// replicates only once an operator sets a peer on both nodes.
//
// Registered HERE rather than in Program.cs (where the plan first put it) so the
// composition-root tests, which build this graph and not Program.cs, actually
// cover it. The endpoint half — MapZbLocalDbSync — has to stay in Program.cs
// because it needs the WebApplication.
services.AddZbLocalDbReplication(config);
services.AddDataConnectionLayer();
// Local SQLite store by default; a local store replicating against a shared SQL-Server hub
// only when Secrets:Replication:Enabled is true AND a hub connection string is present.
// Hub mode matters most here: a site node keeps resolving secrets from its local store
// straight through a WAN outage to central. See SecretsRegistration.
services.AddScadaBridgeSecrets(config);
// Adapter that surfaces the site id to
// StoreAndForwardService through DI WITHOUT introducing a
// StoreAndForward → HealthMonitoring project-reference cycle. Must be
// registered BEFORE AddStoreAndForward so the S&F factory resolves a
// non-empty SiteId at construction time (otherwise the S&F service is
// a singleton and the empty-string value would be cached for the
// lifetime of the process).
services.AddSingleton();
services.AddStoreAndForward();
services.AddSiteEventLogging();
// Site Event Logging — bridge ISiteEventLogger.FailedWriteCount
// into the site health report as a point-in-time SiteEventLogWriteFailures field.
// Must come AFTER both AddSiteHealthMonitoring (registers ISiteHealthCollector) and
// AddSiteEventLogging (registers ISiteEventLogger). The outer Func
// is evaluated once at hosted-service resolution time (root IServiceProvider is available);
// the inner Func is called on every poll tick and reads FailedWriteCount from the
// already-resolved ISiteEventLogger singleton.
services.AddSiteEventLogHealthMetricsBridge(
sp => () => sp.GetRequiredService().FailedWriteCount);
// LocalDb replication — bridge ISyncStatus onto the site health report. Registered
// unconditionally alongside the engine: on a node with no peer this reports
// Connected=false with a real 0 backlog, which is the healthy default-OFF state.
// OplogBacklog is passed through NULLABLE on purpose — null means the poll failed
// or no provider is wired, and flattening it to 0 would render a replication pair
// that cannot read its own oplog as perfectly healthy.
services.AddLocalDbReplicationHealthBridge(
sp => () => sp.GetRequiredService().Connected,
sp => () => sp.GetRequiredService().OplogBacklog);
// Audit Log — site-side hot-path writer + telemetry collaborators.
// The SiteAuditTelemetryActor itself is registered by AkkaHostedService
// in the site-role block; this call wires every DI dependency it (and
// ScriptRuntimeContext) reaches for.
services.AddAuditLog(config);
// Bridge FallbackAuditWriter primary
// failures into the site health report payload as
// SiteAuditWriteFailures. Must come AFTER both AddSiteHealthMonitoring
// (registers ISiteHealthCollector) and AddAuditLog (registers the
// NoOp default this call replaces).
services.AddAuditLogHealthMetricsBridge();
// Akka.NET bootstrap via hosted service
services.AddSingleton();
services.AddHostedService(sp => sp.GetRequiredService());
// Bridge the AkkaHostedService-owned ActorSystem to DI as a SINGLETON via
// GetOrCreateActorSystem(). The shared ZB.MOM.WW.Health Akka checks resolve ActorSystem
// from DI, per probe, inside a child scope. ActorSystem is IDisposable, so a TRANSIENT
// (or scoped) bridge is captured-and-disposed by each probe's scope — disposing the live
// system mid-flight (CoordinatedShutdown/ActorSystemTerminateReason) and tearing down the
// node. A singleton is resolved from the root and never disposed by a child scope; routing
// through GetOrCreateActorSystem (instead of a plain singleton factory over .ActorSystem)
// means the first resolve CREATES the system rather than caching a null if a probe wins
// the startup race.
services.AddSingleton(sp =>
sp.GetRequiredService().GetOrCreateActorSystem());
// Cluster node status provider for health reports
services.AddSingleton(sp =>
{
var akkaService = sp.GetRequiredService();
var nodeOptions = sp.GetRequiredService>().Value;
var siteRole = $"site-{nodeOptions.SiteId}";
return new AkkaClusterNodeProvider(akkaService, siteRole);
});
// The EventLogPurgeService runs on every
// site host node but consults this optional gate each tick and early-exits on
// the standby. Register it to delegate to IClusterNodeProvider.SelfIsPrimary
// (the canonical "this node is the oldest Up member (singleton host)" check) so purge
// runs ONLY on the active node — no duplicated cluster logic. Non-clustered test hosts that
// never call SiteServiceRegistration leave it unregistered, so the purge defaults
// to always-run (the pre-fix behaviour, preserved).
services.AddSingleton(sp =>
{
var nodeProvider = sp.GetRequiredService();
return () => nodeProvider.SelfIsPrimary;
});
// Options binding
BindSharedOptions(services, config);
// Bind + eagerly validate the site-pipeline options. The validators live with their
// owner component projects (SiteRuntime / StoreAndForward / SiteEventLogging) per the
// "options classes owned by component projects" convention, but these three sections
// are bound HERE in the Host (not in a component SCE), so both the ValidateOnStart()
// chain and the TryAddEnumerable validator registration happen here. A bad section
// fails fast at host build with a clear, key-naming message instead of crashing an
// actor / hosted service later with an opaque ArgumentOutOfRangeException.
services.AddOptions()
.Bind(config.GetSection("ScadaBridge:SiteRuntime"))
.ValidateOnStart();
services.TryAddEnumerable(
ServiceDescriptor.Singleton, SiteRuntimeOptionsValidator>());
// OperationTrackingOptions for the site-local cached-call tracking store
// (registered by AddSiteRuntime). Bind + eagerly validate here — the store
// opens its SQLite connection string at construction, so a blank connection
// string should fail the host at boot with a key-naming message rather than
// throw opaquely on first resolve (arch-review 08r2 NF4/T8).
services.AddOptions()
.Bind(config.GetSection("ScadaBridge:OperationTracking"))
.ValidateOnStart();
services.TryAddEnumerable(
ServiceDescriptor.Singleton,
SiteRuntime.Tracking.OperationTrackingOptionsValidator>());
// NF8: DataConnectionOptions is now bound by AddDataConnectionLayer() itself
// (canonical ScadaBridge:DataConnection section) — the duplicate Host binding
// that used to mask the SCE's wrong section name is deleted.
services.AddOptions()
.Bind(config.GetSection("ScadaBridge:StoreAndForward"))
.ValidateOnStart();
services.TryAddEnumerable(
ServiceDescriptor.Singleton, StoreAndForwardOptionsValidator>());
services.AddOptions()
.Bind(config.GetSection("ScadaBridge:SiteEventLog"))
.ValidateOnStart();
services.TryAddEnumerable(
ServiceDescriptor.Singleton, SiteEventLogOptionsValidator>());
}
/// Binds shared options sections (Node, Cluster, Database, Communication, etc.) used by both site and central roles.
/// The service collection to bind options into.
/// Application configuration supplying the option values.
public static void BindSharedOptions(IServiceCollection services, IConfiguration config)
{
// Bind + eagerly validate: an empty NodeName would stamp the SourceNode audit
// column NULL, so fail the host at boot with a key-naming message instead of
// silently degrading the audit trail (arch-review 08r2 NF4).
services.AddOptions().Bind(config.GetSection("ScadaBridge:Node")).ValidateOnStart();
services.TryAddEnumerable(
ServiceDescriptor.Singleton, NodeOptionsValidator>());
// Bind + eagerly validate: ClusterOptionsValidator is registered (TryAddEnumerable)
// by the ClusterInfrastructure module, so chaining ValidateOnStart() here makes a bad
// ScadaBridge:Cluster section fail fast at host build instead of lazily on first resolve.
services.AddOptions().Bind(config.GetSection("ScadaBridge:Cluster")).ValidateOnStart();
// Bind + eagerly validate: a present-but-blank connection setting fails fast
// here rather than opaquely at first DB use (arch-review 08r2 NF4).
services.AddOptions().Bind(config.GetSection("ScadaBridge:Database")).ValidateOnStart();
services.TryAddEnumerable(
ServiceDescriptor.Singleton, DatabaseOptionsValidator>());
// NF8: CommunicationOptions is now bound by AddCommunication() itself (canonical
// ScadaBridge:Communication section) — the duplicate Host binding that used to mask
// the SCE's wrong section name is deleted.
// Bind + eagerly validate: HealthMonitoringOptionsValidator is registered (TryAddEnumerable)
// by the HealthMonitoring module, so chaining ValidateOnStart() here makes a bad
// ScadaBridge:HealthMonitoring section fail fast at host build instead of lazily on first resolve.
services.AddOptions().Bind(config.GetSection("ScadaBridge:HealthMonitoring")).ValidateOnStart();
// NF8: NotificationOptions is bound by AddNotificationService() (central-only,
// canonical ScadaBridge:Notification section). Sites don't deliver notifications
// and have no IOptions consumer, so the redundant Host
// duplicate that bound it on both paths is deleted.
// Bind + eagerly validate: an unrecognised MinimumLevel would silently fall back
// to Information; fail fast at boot so the operator's intended floor is honoured
// (arch-review 08r2 NF4).
services.AddOptions().Bind(config.GetSection("ScadaBridge:Logging")).ValidateOnStart();
services.TryAddEnumerable(
ServiceDescriptor.Singleton, LoggingOptionsValidator>());
// Audit Log — exposes ScadaBridge:Node:NodeName to downstream audit
// writers so they can stamp the SourceNode column. Registered here in
// shared bootstrap because every node (central + site) needs it.
services.AddSingleton();
// Observability — shared ZB.MOM.WW.Telemetry. Registered in shared bootstrap so
// BOTH the central and site composition roots wire the OTel Resource (the
// service.name/site.id/node.role identity triple) + standard instrumentation +
// the always-on Prometheus exporter. Mount the /metrics scrape endpoint per role
// with app.MapZbMetrics(). The same `?? "central"` SiteId default Program.cs uses
// is applied here so the Resource attribute matches the log-enricher value.
// The application meter is named so OTel observes its instruments; emit points are
// wired by follow-on tasks (the instruments are no-op until a listener attaches).
services.AddZbTelemetry(o =>
{
o.ServiceName = "scadabridge";
o.SiteId = config["ScadaBridge:Node:SiteId"] ?? "central";
o.NodeRole = config["ScadaBridge:Node:Role"];
o.Meters = ObservedMeters;
if (Enum.TryParse(config["ScadaBridge:Telemetry:Exporter"], ignoreCase: true, out var exporter))
o.Exporter = exporter;
var otlp = config["ScadaBridge:Telemetry:OtlpEndpoint"];
if (!string.IsNullOrWhiteSpace(otlp))
o.OtlpEndpoint = otlp;
});
}
}