diff --git a/CLAUDE.md b/CLAUDE.md
index a424866c..0b0d2588 100644
--- a/CLAUDE.md
+++ b/CLAUDE.md
@@ -219,6 +219,24 @@ The server supports configurable OPC UA transport security via the `OpcUa:Enable
The server supports non-transparent warm/hot redundancy via the `Redundancy` section in `appsettings.json`. Two instances share the same Galaxy DB and the same mxaccessgw (under distinct `MxAccess.ClientName` values) but have unique `ApplicationUri` values. Each exposes `RedundancySupport`, `ServerUriArray`, and a dynamic `ServiceLevel` based on role and runtime health. The primary advertises a higher ServiceLevel than the secondary. See `docs/Redundancy.md` for the full guide.
+## LocalDb pair-local config cache (Phase 1)
+
+Every **driver-role** node keeps a consolidated `ZB.MOM.WW.LocalDb` SQLite database (the retired
+LiteDB `LocalCache` subsystem was deleted as superseded). Phase-1 scope: it caches the
+**deployed-configuration artifact** (chunked, SHA-256-verified, newest-2 retention per cluster) so a
+driver node can **boot from cache when central SQL Server is unreachable** — `DriverHostActor` writes
+the cache after each successful apply (`IDeploymentArtifactCache` → `LocalDbDeploymentArtifactCache`,
+in `Runtime/Deployment/`) and reads it as a boot fallback, flagging running-from-cache. Wired by
+`AddOtOpcUaLocalDb` in the `hasDriver` branch only (`Host/Configuration/LocalDbRegistration.cs` +
+`LocalDbSetup.OnReady`, whose DDL→`RegisterReplicated` order is load-bearing); **admin-only nodes
+never register it.** The cache can optionally replicate to the node's redundant pair peer over the
+library's gRPC sync — **default-OFF, fail-closed bearer auth** (`LocalDbSyncAuthInterceptor`), gated
+on `LocalDb:SyncListenPort` (a dedicated Kestrel h2c listener) + `LocalDb:Replication:*`. On the
+docker-dev rig the **site-a pair** has replication on; central + site-b stay default-OFF (site-b is
+the pin). ApiKey via `${secret:}`/env, never committed (the rig's dev key is the committed-dev-secret
+exception). See `docs/Configuration.md` (`LocalDb` section), `docs/Redundancy.md` (pair-local config
+cache), and the runbook `docs/operations/2026-07-20-localdb-pair-replication.md`.
+
## LDAP Authentication
The server uses LDAP-based user authentication via the `Security:Ldap` section in `appsettings.json`. When enabled, credentials are validated by LDAP bind against a GLAuth server, and LDAP group membership maps to OPC UA permissions: `ReadOnly` (browse/read), `WriteOperate` (write FreeAccess/Operate attributes), `WriteTune` (write Tune attributes), `WriteConfigure` (write Configure attributes), `AlarmAck` (alarm acknowledgment). `LdapOpcUaUserAuthenticator` (`src/Server/ZB.MOM.WW.OtOpcUa.Host/OpcUa/LdapOpcUaUserAuthenticator.cs`) implements `IOpcUaUserAuthenticator`, delegating the LDAP bind + group lookup to `OtOpcUaLdapAuthService` (`src/Server/ZB.MOM.WW.OtOpcUa.Security/Ldap/OtOpcUaLdapAuthService.cs`, an `ILdapAuthService`). See `docs/security.md` for the full guide.
diff --git a/Directory.Packages.props b/Directory.Packages.props
index 155dd762..b62ff772 100644
--- a/Directory.Packages.props
+++ b/Directory.Packages.props
@@ -29,10 +29,10 @@
+
-
@@ -122,6 +122,13 @@
+
+
+
diff --git a/NuGet.config b/NuGet.config
index 48599f6d..9ae3c0d0 100644
--- a/NuGet.config
+++ b/NuGet.config
@@ -27,6 +27,8 @@
+
+
diff --git a/docker-dev/docker-compose.yml b/docker-dev/docker-compose.yml
index af4cd50e..2d2d712d 100644
--- a/docker-dev/docker-compose.yml
+++ b/docker-dev/docker-compose.yml
@@ -224,8 +224,15 @@ services:
Serilog__MinimumLevel__Override__Microsoft.EntityFrameworkCore: "Warning"
Serilog__MinimumLevel__Override__Microsoft.AspNetCore: "Warning"
GALAXY_MXGW_API_KEY: "${GALAXY_MXGW_API_KEY:-mxgw_otopcua2_GI7-tNozYE6cXGUSgEzL3AHDV7bYcYIHdMwKYgyHdX4}"
+ # Consolidated pair-local config cache (ZB.MOM.WW.LocalDb). Every driver-role node
+ # (central is admin,driver) gets one, on a per-node named volume so the cache survives
+ # container recreates. The central pair leaves replication OFF (default) — only the
+ # site-a pair replicates, as the enablement demo; site-b is the default-OFF pin.
+ LocalDb__Path: "/app/data/otopcua-localdb.db"
ports:
- "4840:4840"
+ volumes:
+ - otopcua-localdb-central-1:/app/data
central-2:
<<: *otopcua-host
@@ -283,8 +290,12 @@ services:
Serilog__MinimumLevel__Override__Microsoft.EntityFrameworkCore: "Warning"
Serilog__MinimumLevel__Override__Microsoft.AspNetCore: "Warning"
GALAXY_MXGW_API_KEY: "${GALAXY_MXGW_API_KEY:-mxgw_otopcua2_GI7-tNozYE6cXGUSgEzL3AHDV7bYcYIHdMwKYgyHdX4}"
+ # Pair-local config cache; central pair leaves replication OFF (see central-1).
+ LocalDb__Path: "/app/data/otopcua-localdb.db"
ports:
- "4841:4840"
+ volumes:
+ - otopcua-localdb-central-2:/app/data
# ── Site A cluster (2-node driver-only) ─────────────────────────────────────
# Driver-only members of the single mesh, scoped to SITE-A by ClusterId. No UI,
@@ -313,8 +324,24 @@ services:
# Resolved at runtime by GalaxyDriver.ResolveApiKey when a DriverInstance's
# Gateway.ApiKeySecretRef = "env:GALAXY_MXGW_API_KEY".
GALAXY_MXGW_API_KEY: "${GALAXY_MXGW_API_KEY:-mxgw_otopcua2_GI7-tNozYE6cXGUSgEzL3AHDV7bYcYIHdMwKYgyHdX4}"
+ # Pair-local config cache + REPLICATION ON — the site-a pair is the enablement demo.
+ # site-a-1 is the initiator: it binds the h2c sync listener on 9001 AND dials the peer.
+ # Setting SyncListenPort makes Program.cs add a dedicated Http2-only Kestrel listener and
+ # re-bind the primary HTTP port (an explicit Listen* otherwise discards ASPNETCORE_URLS).
+ LocalDb__Path: "/app/data/otopcua-localdb.db"
+ LocalDb__SyncListenPort: "9001"
+ LocalDb__Replication__PeerAddress: "http://site-a-2:9001"
+ # DEV-ONLY committed key (rig philosophy — like the SQL password + secrets KEK above). It
+ # MUST be byte-identical on both nodes: the interceptor is fail-closed, so any mismatch
+ # silently stops the pair converging. NEVER reuse this key outside this local dev rig.
+ LocalDb__Replication__ApiKey: "dev-site-a-localdb-sync-key"
+ # Row-count batching against gRPC's 4 MB message cap; artifact chunk rows are ≈171 KB, so
+ # 16 × 171 KB ≈ 2.7 MB stays under the cap with headroom.
+ LocalDb__Replication__MaxBatchSize: "16"
ports:
- "4842:4840"
+ volumes:
+ - otopcua-localdb-site-a-1:/app/data
site-a-2:
<<: *otopcua-host
@@ -334,8 +361,18 @@ services:
Serilog__MinimumLevel__Override__Microsoft.EntityFrameworkCore: "Warning"
Serilog__MinimumLevel__Override__Microsoft.AspNetCore: "Warning"
GALAXY_MXGW_API_KEY: "${GALAXY_MXGW_API_KEY:-mxgw_otopcua2_GI7-tNozYE6cXGUSgEzL3AHDV7bYcYIHdMwKYgyHdX4}"
+ # site-a-2 is the passive half of the replicating pair: it binds the sync listener on 9001
+ # but does NOT dial (no PeerAddress). The replication stream is bidirectional, so a-1's
+ # single dial carries both directions. Same key + MaxBatchSize as a-1 (byte-identical key
+ # is mandatory — the interceptor fail-closes on a mismatch).
+ LocalDb__Path: "/app/data/otopcua-localdb.db"
+ LocalDb__SyncListenPort: "9001"
+ LocalDb__Replication__ApiKey: "dev-site-a-localdb-sync-key"
+ LocalDb__Replication__MaxBatchSize: "16"
ports:
- "4843:4840"
+ volumes:
+ - otopcua-localdb-site-a-2:/app/data
# ── Site B cluster (2-node driver-only) ─────────────────────────────────────
@@ -357,8 +394,14 @@ services:
Serilog__MinimumLevel__Override__Microsoft.EntityFrameworkCore: "Warning"
Serilog__MinimumLevel__Override__Microsoft.AspNetCore: "Warning"
GALAXY_MXGW_API_KEY: "${GALAXY_MXGW_API_KEY:-mxgw_otopcua2_GI7-tNozYE6cXGUSgEzL3AHDV7bYcYIHdMwKYgyHdX4}"
+ # site-b is the default-OFF pin: it gets the pair-local cache but NO SyncListenPort and NO
+ # replication config, so no sync listener binds and ISyncStatus stays disconnected/Healthy.
+ # This is what the live gate's check 8 verifies — the cache works locally with replication off.
+ LocalDb__Path: "/app/data/otopcua-localdb.db"
ports:
- "4844:4840"
+ volumes:
+ - otopcua-localdb-site-b-1:/app/data
site-b-2:
<<: *otopcua-host
@@ -378,8 +421,12 @@ services:
Serilog__MinimumLevel__Override__Microsoft.EntityFrameworkCore: "Warning"
Serilog__MinimumLevel__Override__Microsoft.AspNetCore: "Warning"
GALAXY_MXGW_API_KEY: "${GALAXY_MXGW_API_KEY:-mxgw_otopcua2_GI7-tNozYE6cXGUSgEzL3AHDV7bYcYIHdMwKYgyHdX4}"
+ # site-b default-OFF pin (see site-b-1): cache on, replication off.
+ LocalDb__Path: "/app/data/otopcua-localdb.db"
ports:
- "4845:4840"
+ volumes:
+ - otopcua-localdb-site-b-2:/app/data
traefik:
image: traefik:v3.1
@@ -400,3 +447,13 @@ services:
volumes:
# SQL Server data dir — persists the OtOpcUa ConfigDb across container recreates.
otopcua-mssql-data:
+ # Per-node pair-local config-cache (ZB.MOM.WW.LocalDb) files. One volume per node — the
+ # cache is node-local, not shared; it persists the cached deployment artifact (and, on the
+ # site-a pair, the replicated oplog state) across container recreates so a node can boot
+ # from cache after a restart even with central SQL down.
+ otopcua-localdb-central-1:
+ otopcua-localdb-central-2:
+ otopcua-localdb-site-a-1:
+ otopcua-localdb-site-a-2:
+ otopcua-localdb-site-b-1:
+ otopcua-localdb-site-b-2:
diff --git a/docs/Configuration.md b/docs/Configuration.md
index fdcac78c..5f01bfdc 100644
--- a/docs/Configuration.md
+++ b/docs/Configuration.md
@@ -146,6 +146,28 @@ key must carry the scopes `historian:read`, `historian:write`, `historian:tags:w
[`docs/Historian.md`](Historian.md) for the full key reference, the migration note (old Wonderware keys →
gateway keys), and the deployment prerequisites.
+### `LocalDb` (pair-local config cache — driver role only)
+
+Driver-role nodes keep a consolidated **`ZB.MOM.WW.LocalDb`** SQLite database that caches the deployed
+configuration artifact (chunked) so the node can **boot from cache when central SQL is unreachable**. The
+section is bound by `AddOtOpcUaLocalDb` (`Host/Configuration/LocalDbRegistration.cs`) inside the `hasDriver`
+branch — **admin-only nodes never register it, and never require `LocalDb:Path`**. Storage is unconditional;
+replication stays inert until the sync port + peer are set.
+
+| Key | Default | Meaning |
+|---|---|---|
+| `LocalDb:Path` | `./data/otopcua-localdb.db` | SQLite file path. `ValidateOnStart`-required once `AddZbLocalDb` runs (driver nodes only). In containers put it on a durable volume. |
+| `LocalDb:SyncListenPort` | `0` | `0` = replication off (no sync listener). A non-zero port binds a dedicated **h2c** sync listener; set the **same** port on both pair nodes. Setting it makes the Host add an explicit Kestrel `Listen*` and re-bind the primary HTTP port (an explicit `Listen*` otherwise discards `ASPNETCORE_URLS`). |
+| `LocalDb:Replication:PeerAddress` | `""` | The peer this node **dials** (`http://:`). Set on the initiator only; leave empty on the passive node. The stream is bidirectional. |
+| `LocalDb:Replication:ApiKey` | `""` | Bearer token for the sync stream. **Fail-closed:** no key ⇒ the passive endpoint refuses everything; a mismatch ⇒ the pair silently stops converging. **Must be byte-identical on both nodes.** Supply via `${secret:...}` / env — never a cleartext literal in production. |
+| `LocalDb:Replication:MaxBatchSize` | (library default) | Rows per replication batch — **row-count-only** against gRPC's 4 MB cap. `16` for the artifact cache (chunk rows ≈171 KB, so 16 × 171 KB ≈ 2.7 MB). |
+
+Replication is **default-OFF** across the fleet; it is enabled per-pair as an opt-in. See
+[`docs/operations/2026-07-20-localdb-pair-replication.md`](operations/2026-07-20-localdb-pair-replication.md)
+for the enablement runbook (ApiKey handling, stop/start-together rule, tombstone-retention window, and the
+never-`sqlite3`-a-live-WAL-DB inspection rules) and the "pair-local config cache" section of
+[`docs/Redundancy.md`](Redundancy.md) for what boot-from-cache does and does not cover.
+
---
## Environment variables
diff --git a/docs/Redundancy.md b/docs/Redundancy.md
index 19fe48b7..6692bed9 100644
--- a/docs/Redundancy.md
+++ b/docs/Redundancy.md
@@ -226,6 +226,35 @@ Net effect: each alarm transition appears **once** on `/alerts` and would histor
See [ScriptedAlarms.md](ScriptedAlarms.md) and [AlarmTracking.md](AlarmTracking.md) for the scripted-alarm engine internals.
+## Pair-local config cache (LocalDb — Phase 1)
+
+Independently of the ServiceLevel machinery above, every **driver-role** node keeps a consolidated
+[`ZB.MOM.WW.LocalDb`](../CLAUDE.md) SQLite database that caches the **deployed-configuration
+artifact** (chunked). Its job is a single failure mode redundancy does not otherwise cover: a driver
+node restarting into a **central-SQL-Server outage** would come up with no configuration at all. With
+the cache, it **boots from the last artifact it applied** instead, and logs a running-from-cache
+signal.
+
+- **What replicates.** The cache can *optionally* replicate to the node's redundant pair peer over
+ the LocalDb library's gRPC sync (HLC-stamped, last-writer-wins). Two tables replicate:
+ `deployment_artifacts` and `deployment_pointer`. Replication is **default-OFF and fail-closed** —
+ it stays inert until `LocalDb:SyncListenPort` + a peer are configured, and a missing/mismatched
+ `LocalDb:Replication:ApiKey` refuses or silently halts sync. It is enabled per-pair as an opt-in.
+- **The replicated-cache payoff.** In a pair with replication on, *either* node can be the one that
+ applied the latest deploy; the peer holds a byte-identical copy. So a node that restarts during a
+ central outage can boot the current config **even if it never applied that deploy itself** — its
+ peer did, and the row replicated.
+- **What it does NOT cover.** The cache is a **boot fallback for central-SQL outages only**, not a
+ standby data path and not a replacement for central. A **new deployment still requires central
+ SQL**; the cache is read only when the central fetch fails at startup, and the node resumes
+ fresh-config behavior once central returns. It does not replicate live tag values, alarms, or
+ historian data — only the config artifact. It is orthogonal to the ServiceLevel/primary-gate
+ logic: a node running from cache still participates in redundancy normally.
+
+Full detail: the enablement runbook at
+[`docs/operations/2026-07-20-localdb-pair-replication.md`](operations/2026-07-20-localdb-pair-replication.md)
+and the `LocalDb` section of [`docs/Configuration.md`](Configuration.md).
+
## Client-side failover
The OtOpcUa Client CLI at `src/Client/ZB.MOM.WW.OtOpcUa.Client.CLI` supports `-F` / `--failover-urls` for automatic client-side failover; for long-running subscriptions the CLI monitors session KeepAlive and reconnects to the next available server, recreating the subscription on the new endpoint. See [`Client.CLI.md`](Client.CLI.md).
diff --git a/docs/operations/2026-07-20-localdb-pair-replication.md b/docs/operations/2026-07-20-localdb-pair-replication.md
new file mode 100644
index 00000000..5bc1dea5
--- /dev/null
+++ b/docs/operations/2026-07-20-localdb-pair-replication.md
@@ -0,0 +1,124 @@
+# LocalDb pair replication — operations runbook
+
+> **Phase 1 scope.** Every driver-role OtOpcUa node keeps a consolidated
+> [`ZB.MOM.WW.LocalDb`](../../CLAUDE.md) SQLite database. Today it caches exactly one thing: the
+> **deployed-configuration artifact**, chunked, so a node can **boot from cache when central SQL
+> Server is unreachable**. That cache can *optionally* replicate to the node's redundant pair peer
+> over the library's gRPC sync. **Replication is default-OFF and fail-closed.** This runbook covers
+> enabling it, the operational rules that keep a pair converging, and the DB-inspection safety
+> rules.
+
+## What replicates, and what it buys you
+
+- **Replicated tables:** `deployment_artifacts` (the artifact, split into base64 chunks) and
+ `deployment_pointer` (one current-deployment pointer per cluster). Registered — in this order,
+ after the DDL — by `LocalDbSetup.OnReady`.
+- **The payoff:** in a redundant driver pair, *either* node can be the one that applied the latest
+ deploy and wrote the cache. With replication on, the peer holds a byte-identical copy. So a node
+ that restarts into a central-SQL outage can boot the current config **even if it never applied
+ that deploy itself** — its peer did, and the row replicated.
+- **Convergence model:** last-writer-wins per primary key, HLC-stamped. Both nodes also store
+ locally after every apply, so the pair converges to identical content regardless of which node
+ deployed. Retention (newest-2 deployments per cluster) prunes on one node and the deletes
+ replicate as tombstones.
+
+## Enabling replication on a pair
+
+Replication has two knobs, both under `LocalDb`:
+
+| Key | Role | Notes |
+|---|---|---|
+| `LocalDb:SyncListenPort` | binds the dedicated **h2c** sync listener | `0` (default) = no listener, replication off. Set the **same** non-zero port on **both** nodes of the pair. |
+| `LocalDb:Replication:PeerAddress` | the address this node **dials** | Set on **one** node (the initiator); leave empty on the other (passive). The stream is bidirectional — one dial carries both directions. |
+| `LocalDb:Replication:ApiKey` | bearer token for the sync stream | **Must be byte-identical on both nodes.** See fail-closed rule below. Supply via `${secret:...}` / env — never a cleartext literal in production config. |
+| `LocalDb:Replication:MaxBatchSize` | rows per replication batch | `16` for the artifact cache. Batching is **row-count-only** against gRPC's 4 MB message cap; artifact chunk rows are ≈171 KB, so 16 × 171 KB ≈ 2.7 MB stays under the cap. Raising it risks tripping the cap. |
+
+Example (site-a-1 initiator, site-a-2 passive):
+
+```
+site-a-1: LocalDb__SyncListenPort=9001
+ LocalDb__Replication__PeerAddress=http://site-a-2:9001
+ LocalDb__Replication__ApiKey=${secret:site-a-localdb-sync-key}
+ LocalDb__Replication__MaxBatchSize=16
+site-a-2: LocalDb__SyncListenPort=9001
+ LocalDb__Replication__ApiKey=${secret:site-a-localdb-sync-key} # no PeerAddress
+ LocalDb__Replication__MaxBatchSize=16
+```
+
+> **Kestrel note.** Setting `SyncListenPort` makes the Host add an explicit `Listen*` for the h2c
+> sync listener. An explicit `Listen*` makes Kestrel **ignore `ASPNETCORE_URLS` entirely**, so the
+> Host also re-binds the primary HTTP port in the same block. If you change the primary port,
+> verify both `:` and `:` appear in the startup "Now listening on" lines.
+
+### The ApiKey rule (fail-closed)
+
+The replication library's passive endpoint verifies **no** authentication — the host
+`LocalDbSyncAuthInterceptor` is the only gate, and it is **fail-closed**:
+
+- **No key configured ⇒ every sync call is refused** (`PermissionDenied`). "No key" is never
+ "no auth required".
+- **A key mismatch ⇒ the pair silently stops converging.** The initiator's stream is rejected at
+ the peer; nothing errors loudly. A typo in one node's key looks exactly like "replication is
+ broken". If a pair is not converging, **check the keys match first.**
+
+## Operational rules
+
+- **Stop / start the pair together where you can.** Each node keeps working (and caching locally)
+ while its peer is down; the outage is not a data-loss event — the surviving node accumulates
+ writes and the peer catches up on rejoin. But a long-lived solo node drifts further from its
+ peer, so avoid leaving a pair split for extended periods.
+- **Tombstone-retention resurrection window.** Retention prunes to the newest 2 deployments and
+ replicates the prune as tombstones. Tombstones are themselves retained only for a bounded window.
+ If a node is offline **longer than the tombstone-retention window**, a delete that happened during
+ its outage may no longer be expressible as a tombstone on rejoin — a pruned deployment could
+ briefly reappear until the next deploy re-prunes it. Keep pair outages well inside that window.
+- **What boot-from-cache does NOT cover.** The cache is a *fallback for central-SQL outages at
+ boot*, not a replacement for central. A **new deployment still requires central SQL** — the cache
+ is only read when the central fetch fails at startup. When central returns, the node resumes
+ fresh-config behavior. Boot-from-cache logs a running-from-cache signal; treat a node that stays
+ on it as a central-connectivity incident, not steady state.
+
+## Inspecting the LocalDb file — safety rules
+
+The database runs in **WAL mode**. These rules exist because violating them corrupted a live DB in
+the 2026-07-20 ScadaBridge incident:
+
+- **Never run `sqlite3` on the live file** (host-side, against a bind-mounted or container path).
+ Opening a live WAL DB from a second process across virtiofs poisons the WAL. **Always copy the
+ triplet first** and query the copy:
+
+ ```bash
+ docker cp :/app/data/otopcua-localdb.db /tmp/localdb.db
+ docker cp :/app/data/otopcua-localdb.db-wal /tmp/localdb.db-wal
+ docker cp :/app/data/otopcua-localdb.db-shm /tmp/localdb.db-shm
+ sqlite3 /tmp/localdb.db 'SELECT cluster_id, deployment_id FROM deployment_pointer;'
+ ```
+
+- **Metrics** come from the container, not a host curl (`aspnet:10.0` has no `curl`):
+
+ ```bash
+ docker run --rm --network container: curlimages/curl:latest -s localhost:/metrics | grep localdb_
+ ```
+
+- **If an anomaly appears within seconds of your own measurement, suspect the measurement.**
+ Restart both nodes and re-observe untouched before blaming the library.
+
+## Useful queries (on a copied triplet)
+
+```sql
+-- What each node thinks the current deployment is
+SELECT cluster_id, deployment_id, revision_hash, applied_at_utc FROM deployment_pointer;
+
+-- Convergence check: the pointer's origin stamp should match on both nodes
+SELECT pk_json, hlc, node_id, is_tombstone
+FROM __localdb_row_version WHERE table_name = 'deployment_pointer' ORDER BY pk_json;
+
+-- Replication backlog (should drain to 0 when a pair is caught up)
+SELECT COUNT(*) FROM __localdb_oplog;
+```
+
+## See also
+
+- [`docs/Redundancy.md`](../Redundancy.md) — the pair-local config cache section.
+- [`docs/Configuration.md`](../Configuration.md) — the `LocalDb` appsettings section.
+- The two-node convergence harness: `tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDb/`.
diff --git a/docs/plans/2026-07-20-localdb-adoption-phase1.md b/docs/plans/2026-07-20-localdb-adoption-phase1.md
new file mode 100644
index 00000000..e70b2a10
--- /dev/null
+++ b/docs/plans/2026-07-20-localdb-adoption-phase1.md
@@ -0,0 +1,706 @@
+# OtOpcUa LocalDb Adoption — Phase 1 Implementation Plan
+
+> **For Claude:** REQUIRED SUB-SKILL: Use superpowers-extended-cc:executing-plans to implement this plan task-by-task.
+>
+> **Execution model:** This plan is optimized for **Claude Opus** agents (`claude --model opus`).
+> Dispatch implementer subagents on Opus for every task classified `high-risk`; `small`/`trivial`
+> tasks may use Sonnet. Work on branch **`feat/localdb-phase1`** in this repo (`~/Desktop/OtOpcUa`,
+> remote `lmxopcua`). Do NOT merge to `master` as part of this plan — stop at the DoD task and
+> report.
+>
+> **Design authority:** `~/Desktop/scadaproj/docs/plans/2026-07-20-otopcua-localdb-design.md`.
+> Read it before Task 0. The reference adoption is ScadaBridge (merged PR #23) — when in doubt,
+> mirror `~/Desktop/ScadaBridge` (files named per task below).
+
+**Goal:** Give every driver-role OtOpcUa node a consolidated `ZB.MOM.WW.LocalDb` database that caches
+the deployed-configuration artifact (chunked), boots from that cache when central SQL Server is
+unreachable, and optionally replicates it to the node's redundant pair peer over the library's gRPC
+sync (default-OFF, fail-closed bearer auth).
+
+**Architecture:** `AddZbLocalDb`/`AddZbLocalDbReplication` wired in a new `LocalDbRegistration`
+(mirroring `SecretsRegistration`), DDL + `RegisterReplicated` in `LocalDbSetup.OnReady`, a dedicated
+Kestrel h2c listener for `MapZbLocalDbSync` gated on `LocalDb:SyncListenPort`, a consumer-supplied
+fail-closed bearer interceptor, and an `IDeploymentArtifactCache` seam written by `DriverHostActor`
+after each successful apply and read as a boot fallback. The dormant LiteDB LocalCache subsystem is
+deleted as superseded.
+
+**Tech stack:** .NET 10, `ZB.MOM.WW.LocalDb`/`.Replication`/`.Contracts` **0.1.1** (Gitea feed),
+`Grpc.AspNetCore` 2.76.0, Akka.NET (existing), xunit.v3 (Host.IntegrationTests) + xunit2 TestKit
+(actor tests).
+
+**Hard rules (from the ScadaBridge adoption — violating any of these is a defect):**
+1. Order inside `OnReady` is load-bearing: **DDL → `RegisterReplicated` → writes**. Rows written
+ before registration are never captured or snapshotted — silently, forever.
+2. No autoincrement PKs, no BLOB columns in replicated tables.
+3. The library's passive sync endpoint verifies **no** ApiKey — the host interceptor is the only
+ auth, and it must be fail-closed (no key configured ⇒ deny everything).
+4. `ILocalDb.CreateConnection()` returns an **already-open**, pragma-configured connection with the
+ `zb_hlc_next()` UDF. Never call `Open()` on it. There is **no in-memory mode** — tests use temp
+ files + `SqliteConnection.ClearAllPools()` before delete.
+5. `Grpc.Core.Testing` does not exist on grpc-dotnet — hand-roll fake `ServerCallContext`s.
+6. Explicit Kestrel `Listen*` calls make Kestrel **ignore `ASPNETCORE_URLS`** — when adding the sync
+ listener you must re-bind the existing HTTP port in the same block, and verify it on the rig.
+7. Never run host `sqlite3` against a live bind-mounted WAL DB (copy the `db`/`-wal`/`-shm` triplet
+ with `cp` instead). `aspnet:10.0` containers have no `curl` — use a
+ `curlimages/curl` sidecar with `--network container:`.
+
+---
+
+### Task 0: Preflight + recon (produces `docs/plans/2026-07-20-localdb-phase1-recon.md`)
+
+**Classification:** standard
+**Estimated implement time:** ~5 min
+**Parallelizable with:** none (everything downstream consumes its findings)
+
+**Files:**
+- Create: `docs/plans/2026-07-20-localdb-phase1-recon.md`
+- Read-only: `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Drivers/DriverHostActor.cs`,
+ `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Drivers/DeploymentArtifact.cs`,
+ `src/Server/ZB.MOM.WW.OtOpcUa.Host/Program.cs`,
+ `src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/SecretsRegistration.cs`,
+ `Directory.Packages.props`, `docker-dev/docker-compose.yml`,
+ the observability registration (`AddOtOpcUaObservability` implementation),
+ `src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/*`
+
+**Step 1: Verify the feed packages restore.**
+
+Run:
+```bash
+cd ~/Desktop/OtOpcUa && dotnet package search ZB.MOM.WW.LocalDb --source dohertj2-gitea --format json | head -50
+```
+Expected: `ZB.MOM.WW.LocalDb`, `.Replication`, `.Contracts` at `0.1.1`. If absent → STOP, report.
+
+**Step 2: Map and record, with `file:line` citations, each of:**
+1. **Deploy fetch path:** where `DriverHostActor` loads `Deployment.ArtifactBlob` (the
+ `_dbFactory.CreateDbContext()` sites, ~L1472/1543), the artifact's runtime type
+ (`DeploymentArtifact` — how it's deserialized from the blob), the type/format of
+ `DeploymentId` and `RevisionHash`, and where a successful apply completes (the point after
+ which the cache write belongs).
+2. **Cold-boot path:** what the actor does at startup before any `DispatchDeployment` arrives —
+ does it query central SQL for the current deployment? Where does the failure path land
+ (the `Stale` state, `DbHealthProbeActor` interaction, L1345 dispatch-ignore)? Identify the
+ exact seam where "central fetch failed at boot" is known — that is where the cache fallback goes.
+3. **Cluster identity:** how a driver node knows its `ClusterId` (config key / `ClusterNode` row /
+ `IClusterRoleInfo`) — the cache is keyed by it.
+4. **DriverHostActor construction:** how its dependencies are injected
+ (`WithOtOpcUaRuntimeActors`, `Props` wiring) so `IDeploymentArtifactCache` can be threaded in.
+ ⚠ `Props.Create` builds an expression tree — adding a parameter silently rebinds
+ out-of-position named args at call sites; list every construction site.
+5. **Telemetry allowlist:** does `AddOtOpcUaObservability` restrict meters
+ (`ZbTelemetryOptions.Meters` or equivalent)? If yes, record where the list lives.
+6. **Kestrel/URLs:** confirm the Host binds via `ASPNETCORE_URLS=http://+:9000` with no
+ `ConfigureKestrel` calls; record any existing `builder.WebHost` usage.
+7. **LiteDB LocalCache:** confirm (grep) that nothing outside
+ `src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/` and its tests references
+ `ILocalConfigCache`/`GenerationSealedCache`/`ResilientConfigReader`/`LiteDB`.
+8. **docker-dev volumes:** whether nodes have a writable data volume; record what `LocalDb:Path`
+ should be per node and how env vars are passed (`Cluster__*` style double-underscore).
+
+**Step 3: STOP conditions — halt the plan and report instead of improvising if:**
+- the artifact apply path cannot be given a post-apply hook without restructuring the actor;
+- `DeploymentId`/cluster identity are not stable strings/GUIDs;
+- LiteDB LocalCache turns out to be referenced by live code.
+
+**Step 4: Commit.**
+```bash
+git add docs/plans/2026-07-20-localdb-phase1-recon.md
+git commit -m "docs(localdb): phase-1 recon findings"
+```
+
+---
+
+### Task 1: Package references and pins
+
+**Classification:** small
+**Estimated implement time:** ~3 min
+**Parallelizable with:** none (all code tasks build on it)
+
+**Files:**
+- Modify: `Directory.Packages.props`
+- Modify: `src/Server/ZB.MOM.WW.OtOpcUa.Host/ZB.MOM.WW.OtOpcUa.Host.csproj`
+- Modify: `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/ZB.MOM.WW.OtOpcUa.Runtime.csproj`
+
+**Step 1:** Add to `Directory.Packages.props` (versions exact):
+```xml
+
+
+
+```
+Check the existing `Grpc.Net.Client`/`Grpc.Core.Api` are already ≥ 2.76.0 and
+`Google.Protobuf` ≥ 3.34.1 (the LocalDb.Replication nuspec floors); raise if not. Confirm a
+`SQLitePCLRaw` pin ≥ 2.1.12 exists for the transitive `Microsoft.Data.Sqlite` chain (the family
+advisory GHSA-2m69-gcr7-jv3q); Core.AlarmHistorian already pins `bundle_e_sqlite3` 2.1.12 — add
+`SQLitePCLRaw.lib.e_sqlite3` 2.1.12 as an explicit reference in the Host if the audit flags it.
+
+**Step 2:** Host csproj: `ZB.MOM.WW.LocalDb`, `ZB.MOM.WW.LocalDb.Replication`, `Grpc.AspNetCore`.
+Runtime csproj: `ZB.MOM.WW.LocalDb` only (it needs `ILocalDb`, not the sync engine).
+
+**Step 3:** `dotnet build ZB.MOM.WW.OtOpcUa.slnx` → 0 warnings/errors (repo builds warnings-as-errors).
+
+**Step 4: Commit.** `git commit -m "build(localdb): reference ZB.MOM.WW.LocalDb 0.1.1 + Grpc.AspNetCore"`
+
+---
+
+### Task 2: Schema + `LocalDbSetup.OnReady` + registration tests
+
+**Classification:** standard
+**Estimated implement time:** ~5 min
+**Parallelizable with:** Task 3
+
+**Files:**
+- Create: `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/DeploymentCacheSchema.cs`
+- Create: `src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbSetup.cs`
+- Test: `tests/Server/ZB.MOM.WW.OtOpcUa.Host.Tests/LocalDbSetupTests.cs` (create the test project
+ reference layout to match neighboring Host unit tests; if Host has no unit-test project, place in
+ the closest existing one and note the deviation)
+
+**Step 1: Write the failing tests first** — using a real temp-file `ILocalDb` (build it via
+`new ServiceCollection().AddZbLocalDb(config, LocalDbSetup.OnReady)` with `LocalDb:Path` pointed at
+a temp file; dispose + `SqliteConnection.ClearAllPools()` in cleanup):
+- `OnReady_RegistersExactlyTheTwoDeploymentTables` — `db.ReplicatedTables.Keys` ordinal-sorted
+ equals `["deployment_artifacts", "deployment_pointer"]` (assert BOTH directions: no fewer, no more).
+- `DeploymentArtifacts_PkIsDeploymentIdPlusChunkIndex` and `DeploymentPointer_PkIsClusterId`
+ (pin `ReplicatedTable.PkColumns`).
+- `RowsWrittenAfterOnReady_EnterTheOplog` — insert a row, assert `SELECT COUNT(*) FROM __localdb_oplog` ≥ 1
+ (pins the DDL→register ordering; this is the assertion that catches a silently-broken CDC).
+
+**Step 2: Run tests, watch them fail** (types don't exist yet).
+
+**Step 3: Implement.** `DeploymentCacheSchema` depends only on `Microsoft.Data.Sqlite` (the
+ScadaBridge `*Schema.Apply` pattern — self-sufficient for direct construction in tests):
+
+```csharp
+namespace ZB.MOM.WW.OtOpcUa.Runtime.Deployment;
+
+public static class DeploymentCacheSchema
+{
+ public static void Apply(SqliteConnection connection)
+ {
+ using var cmd = connection.CreateCommand();
+ cmd.CommandText = """
+ CREATE TABLE IF NOT EXISTS deployment_artifacts (
+ deployment_id TEXT NOT NULL,
+ chunk_index INTEGER NOT NULL,
+ cluster_id TEXT NOT NULL,
+ revision_hash TEXT NOT NULL,
+ chunk_count INTEGER NOT NULL,
+ chunk_base64 TEXT NOT NULL,
+ cached_at_utc TEXT NOT NULL,
+ PRIMARY KEY (deployment_id, chunk_index)
+ );
+ CREATE TABLE IF NOT EXISTS deployment_pointer (
+ cluster_id TEXT NOT NULL PRIMARY KEY,
+ deployment_id TEXT NOT NULL,
+ revision_hash TEXT NOT NULL,
+ artifact_sha256 TEXT NOT NULL,
+ applied_at_utc TEXT NOT NULL
+ );
+ """;
+ cmd.ExecuteNonQuery();
+ }
+}
+```
+
+`LocalDbSetup` (Host):
+```csharp
+namespace ZB.MOM.WW.OtOpcUa.Host.Configuration;
+
+internal static class LocalDbSetup
+{
+ // ORDER IS LOAD-BEARING: DDL, then RegisterReplicated, then (nothing else in Phase 1).
+ // Rows written before RegisterReplicated are invisible to the peer forever, silently.
+ public static void OnReady(ILocalDb db)
+ {
+ using var connection = db.CreateConnection(); // already open — do NOT call Open()
+ DeploymentCacheSchema.Apply(connection);
+ db.RegisterReplicated("deployment_artifacts");
+ db.RegisterReplicated("deployment_pointer");
+ }
+}
+```
+
+**Step 4:** `dotnet test --filter "FullyQualifiedName~LocalDbSetupTests"` → PASS.
+
+**Step 5: Commit.** `git commit -m "feat(localdb): deployment-cache schema + OnReady registration"`
+
+---
+
+### Task 3: Fail-closed sync auth interceptor + tests
+
+**Classification:** standard
+**Estimated implement time:** ~4 min
+**Parallelizable with:** Task 2
+
+**Files:**
+- Create: `src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbSyncAuthInterceptor.cs`
+- Test: `tests/Server/ZB.MOM.WW.OtOpcUa.Host.Tests/LocalDbSyncAuthInterceptorTests.cs`
+
+Port ScadaBridge's `src/ZB.MOM.WW.ScadaBridge.Host/LocalDbSyncAuthInterceptor.cs` (read it first;
+keep behavior identical, adjust namespace only). Semantics to preserve exactly:
+- `ServicePrefix = "/localdb_sync.v1.LocalDbSync/"`; any other method path passes through untouched.
+- Expected token from `IOptions.Value.ApiKey`.
+- **Fail-closed:** null/empty configured key ⇒ `RpcException(new Status(StatusCode.PermissionDenied, ...))`
+ for every sync call. Missing/mismatched bearer ⇒ same.
+- `CryptographicOperations.FixedTimeEquals` over UTF-8 bytes; header `authorization`,
+ case-insensitive, `"Bearer "` prefix. Override all four server handler kinds.
+
+**Tests** (hand-rolled fake `ServerCallContext` — `Grpc.Core.Testing` does not exist on grpc-dotnet;
+copy ScadaBridge's `LocalDbSyncAuthInterceptorTests.cs` fake): non-sync method passes with no key;
+sync + no configured key → PermissionDenied; wrong bearer → PermissionDenied; correct bearer → passes.
+
+Write tests first (fail), implement, `dotnet test --filter "FullyQualifiedName~LocalDbSyncAuthInterceptor"`,
+commit `feat(localdb): fail-closed bearer interceptor for the sync endpoint`.
+
+---
+
+### Task 4: `LocalDbRegistration` + Program.cs DI wiring + config defaults
+
+**Classification:** high-risk (touches Program.cs / role gating)
+**Estimated implement time:** ~5 min
+**Parallelizable with:** none (Task 5 edits the same file)
+
+**Files:**
+- Create: `src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbRegistration.cs`
+- Modify: `src/Server/ZB.MOM.WW.OtOpcUa.Host/Program.cs` (inside the `hasDriver` branch)
+- Modify: `src/Server/ZB.MOM.WW.OtOpcUa.Host/appsettings.json`
+
+**Step 1:** `LocalDbRegistration`, mirroring `SecretsRegistration`'s shape:
+```csharp
+namespace ZB.MOM.WW.OtOpcUa.Host.Configuration;
+
+internal static class LocalDbRegistration
+{
+ /// Driver-role nodes only. Storage is unconditional; replication stays inert until
+ /// LocalDb:Replication:PeerAddress (initiator) / LocalDb:SyncListenPort (listener) are set.
+ public static IServiceCollection AddOtOpcUaLocalDb(
+ this IServiceCollection services, IConfiguration configuration)
+ {
+ services.AddZbLocalDb(configuration, LocalDbSetup.OnReady);
+ services.AddZbLocalDbReplication(configuration);
+ return services;
+ }
+
+ public static int SyncListenPort(IConfiguration configuration) =>
+ configuration.GetValue("LocalDb:SyncListenPort");
+}
+```
+
+**Step 2:** Program.cs, in the `hasDriver` block (near `AddAlarmHistorian`):
+`builder.Services.AddOtOpcUaLocalDb(builder.Configuration);` and
+`builder.Services.AddGrpc(o => o.Interceptors.Add());`
+(AddGrpc only under `hasDriver` — admin-only nodes expose no sync surface).
+
+**Step 3:** appsettings.json:
+```json
+"LocalDb": {
+ "Path": "./data/otopcua-localdb.db",
+ "SyncListenPort": 0,
+ "Replication": {}
+}
+```
+`LocalDb:Path` is `ValidateOnStart`-required once `AddZbLocalDb` runs — since registration is
+driver-gated, admin-only configs need nothing. Check the role-overlay files
+(`appsettings.driver.json` / `appsettings.admin-driver.json`) and any deploy templates for
+conflicting `LocalDb` keys.
+
+**Step 4:** Build + full Host unit tests. Verify an admin-only graph doesn't require the key: this
+is pinned properly in Task 10's integration tests, but do a quick
+`OTOPCUA_ROLES=admin dotnet run` smoke only if cheap; otherwise rely on Task 10.
+
+**Step 5: Commit.** `feat(localdb): wire AddOtOpcUaLocalDb into the driver role`
+
+---
+
+### Task 5: Dedicated h2c sync listener + endpoint mapping (THE Kestrel task)
+
+**Classification:** high-risk
+**Estimated implement time:** ~5 min
+**Parallelizable with:** none (Program.cs)
+
+**Files:**
+- Modify: `src/Server/ZB.MOM.WW.OtOpcUa.Host/Program.cs`
+
+**Why high-risk:** the Host today binds exclusively via `ASPNETCORE_URLS=http://+:9000`. Any
+explicit `ConfigureKestrel(... Listen*)` makes Kestrel **ignore URLs entirely** (it logs
+"Overriding address(es)"), which would silently kill the AdminUI/deploy API behind Traefik.
+
+**Step 1:** Add, before `builder.Build()`:
+```csharp
+var syncPort = LocalDbRegistration.SyncListenPort(builder.Configuration);
+if (hasDriver && syncPort > 0)
+{
+ // Explicit Listen* replaces ASPNETCORE_URLS wholesale, so re-bind the primary HTTP
+ // endpoint here too. Parse it from the configured URLs rather than hard-coding 9000.
+ var urls = builder.Configuration["ASPNETCORE_URLS"] ?? builder.Configuration["urls"] ?? "http://+:9000";
+ var httpPort = new Uri(urls.Split(';')[0].Replace("+", "localhost").Replace("*", "localhost")).Port;
+ builder.WebHost.ConfigureKestrel(k =>
+ {
+ k.ListenAnyIP(httpPort); // HTTP/1.1 (existing surface)
+ k.ListenAnyIP(syncPort, o => o.Protocols = HttpProtocols.Http2); // h2c, prior-knowledge gRPC
+ });
+}
+```
+When `syncPort == 0` (the default) nothing changes — URLs binding stays untouched. Cleartext
+`Http1AndHttp2` cannot serve prior-knowledge h2c, hence the dedicated `Http2`-only listener.
+
+**Step 2:** After the app pipeline's other `Map*` calls, driver-gated:
+```csharp
+if (hasDriver && syncPort > 0)
+{
+ app.MapZbLocalDbSync();
+}
+```
+(Mapping is harmless when unauthenticated — the interceptor fail-closes — but gating on the port
+keeps admin-only and default-OFF graphs entirely free of the endpoint.)
+
+**Step 3: Verify both surfaces.**
+```bash
+dotnet build ZB.MOM.WW.OtOpcUa.slnx
+```
+Then a local smoke with the port on:
+`ASPNETCORE_URLS=http://+:9000 OTOPCUA_ROLES=driver LocalDb__SyncListenPort=9001 dotnet run --project src/Server/ZB.MOM.WW.OtOpcUa.Host ...`
+— expect startup logs to show BOTH `:9000` and `:9001` bound, and `curl -s localhost:9000/healthz`
+(or the mapped health route) still answering. If the app needs SQL to boot, defer the smoke to the
+Task 12 rig check but say so explicitly in the task notes.
+
+**Step 4: Commit.** `feat(localdb): dedicated h2c listener + MapZbLocalDbSync gated on LocalDb:SyncListenPort`
+
+---
+
+### Task 6: `IDeploymentArtifactCache` + chunked LocalDb implementation + tests
+
+**Classification:** standard
+**Estimated implement time:** ~5 min
+**Parallelizable with:** Task 3, Task 5
+
+**Files:**
+- Create: `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/IDeploymentArtifactCache.cs`
+- Create: `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/LocalDbDeploymentArtifactCache.cs`
+- Test: `tests/Server/ZB.MOM.WW.OtOpcUa.Runtime.Tests/Deployment/LocalDbDeploymentArtifactCacheTests.cs`
+
+**Step 1: Failing tests** (real temp-file `ILocalDb` with `LocalDbSetup.OnReady` applied):
+- round-trip: store a 300 KiB random artifact → `GetCurrentAsync` returns byte-identical payload
+ (forces ≥ 3 chunks at 128 KiB);
+- retention: store 3 deployments for one cluster → only the newest 2 remain in
+ `deployment_artifacts`; pointer names the newest;
+- integrity: corrupt one chunk row via SQL → `GetCurrentAsync` returns null (miss), never a
+ truncated artifact;
+- missing pointer → null.
+
+**Step 2: Implement:**
+```csharp
+public interface IDeploymentArtifactCache
+{
+ Task StoreAsync(string clusterId, string deploymentId, string revisionHash,
+ byte[] artifact, CancellationToken ct = default);
+ Task GetCurrentAsync(string clusterId, CancellationToken ct = default);
+}
+
+public sealed record CachedDeploymentArtifact(
+ string DeploymentId, string RevisionHash, byte[] Artifact, DateTimeOffset AppliedAtUtc);
+```
+`LocalDbDeploymentArtifactCache(ILocalDb db)`:
+- `StoreAsync`: one `ILocalDbTransaction` — delete any existing chunks for this `deployment_id`
+ (idempotent re-store), insert chunks (raw 128 * 1024 bytes per chunk, `Convert.ToBase64String`),
+ upsert the pointer (`INSERT ... ON CONFLICT(cluster_id) DO UPDATE`), then prune: delete
+ `deployment_artifacts` rows whose `deployment_id` is not among the newest 2 `cached_at_utc` for
+ this `cluster_id`. SHA-256 over the raw artifact into the pointer. ISO-8601 UTC timestamps.
+- `GetCurrentAsync`: read pointer; read chunks `ORDER BY chunk_index`; verify count ==
+ `chunk_count` and SHA-256 matches; return null on any mismatch (log a warning naming which check
+ failed).
+- Use `db.ExecuteAsync`/`QueryAsync` with anonymous-object parameters (`@Name` markers). Remember
+ a `Dictionary` parameter **throws** — use anonymous objects.
+
+**Step 3:** run tests → PASS. **Step 4: Commit.**
+`feat(localdb): chunked deployment-artifact cache over ILocalDb`
+
+---
+
+### Task 7: Cache write path in `DriverHostActor`
+
+**Classification:** high-risk (actor model, `Props` expression-tree trap)
+**Estimated implement time:** ~5 min
+**Parallelizable with:** none
+
+**Files:** (exact edit sites come from the Task 0 recon doc — cite it in the commit)
+- Modify: `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Drivers/DriverHostActor.cs`
+- Modify: the actor-registration site (`WithOtOpcUaRuntimeActors` / DI extension) to provide
+ `IDeploymentArtifactCache`
+- Modify: `src/Server/ZB.MOM.WW.OtOpcUa.Host/Program.cs` or `LocalDbRegistration` — register
+ `IDeploymentArtifactCache` → `LocalDbDeploymentArtifactCache` singleton (driver branch)
+- Test: the **xunit2** TestKit project used for existing DriverHostActor tests (recon names it)
+
+**Steps:**
+1. **Failing test:** after a successful artifact apply, the cache received `StoreAsync` with the
+ applied deployment's id/hash/bytes (inject a recording fake `IDeploymentArtifactCache`).
+ Also: a cache that throws on `StoreAsync` does NOT fail the apply (apply result unchanged, error
+ logged).
+2. Thread the dependency through the actor's constructor. ⚠ `Props.Create` is an expression tree:
+ after adding the parameter, re-check **every** construction site for positional/named-arg
+ rebinding (the recon lists them) — do not rely on the compiler.
+3. Invoke `StoreAsync` fire-and-forget (`PipeTo`-style or a guarded `Task.Run` per the actor's
+ existing async conventions — match whatever pattern the actor already uses for side-effect IO)
+ at the post-apply point identified in recon. Cluster id from the recon-identified source.
+4. Run the actor test suite for this project. Commit:
+ `feat(localdb): DriverHostActor stores applied artifacts in the pair-local cache`
+
+---
+
+### Task 8: Boot-from-cache read path + running-from-cache signal
+
+**Classification:** high-risk
+**Estimated implement time:** ~5 min
+**Parallelizable with:** none
+
+**Files:**
+- Modify: `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Drivers/DriverHostActor.cs`
+- Test: same xunit2 TestKit project as Task 7
+
+**Steps:**
+1. **Failing tests:**
+ - central fetch fails at startup AND cache has an artifact → the actor applies the cached
+ artifact (assert via whatever observable the apply already exposes) and logs/flags
+ running-from-cache;
+ - central fetch fails AND cache empty → today's behavior exactly (Stale, no apply);
+ - central fetch **succeeds** → cache is NOT consulted (fresh config wins; assert the fake
+ cache's `GetCurrentAsync` was never called on the happy path).
+2. Implement at the recon-identified boot-failure seam. The signal: a log warning at minimum plus,
+ if the actor already publishes health/status (recon says how), a `RunningFromCache` marker on it.
+ Do NOT touch `DispatchDeployment` handling — a new deployment still requires central.
+3. Run tests; commit `feat(localdb): boot from the pair-local artifact cache when central SQL is unreachable`.
+
+---
+
+### Task 9: Delete the dormant LiteDB LocalCache subsystem
+
+**Classification:** small
+**Estimated implement time:** ~3 min
+**Parallelizable with:** Task 6, Task 10, Task 11
+
+**Files:**
+- Delete: `src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/` (entire directory:
+ `ILocalConfigCache`, `LiteDbConfigCache`, `GenerationSealedCache`, `ResilientConfigReader`,
+ `GenerationSnapshot`, `StaleConfigFlag`, exceptions)
+- Delete: `tests/Core/ZB.MOM.WW.OtOpcUa.Configuration.Tests/GenerationSealedCacheTests.cs`,
+ `ResilientConfigReaderTests.cs`
+- Modify: `Directory.Packages.props` + the Configuration csproj — remove the `LiteDB` package if
+ nothing else references it (grep first)
+
+**Steps:** Confirm the Task 0 recon's "nothing references it" finding still holds
+(`grep -rn "ILocalConfigCache\|GenerationSealedCache\|ResilientConfigReader\|LiteDB" src/ tests/`),
+delete, build the full solution, run the Configuration test project. DoD phrasing: **no references
+from code** — leave any explanatory prose/comments that point readers at LocalDb instead. Add one
+line to the recon doc noting LiteDB's removal. Commit:
+`refactor(localdb): delete the dormant LiteDB LocalCache (superseded by ZB.MOM.WW.LocalDb)`
+
+---
+
+### Task 10: Health check + telemetry meter allowlist
+
+**Classification:** small
+**Estimated implement time:** ~4 min
+**Parallelizable with:** Task 9, Task 11
+
+**Files:**
+- Create: `src/Server/ZB.MOM.WW.OtOpcUa.Host/Health/LocalDbReplicationHealthCheck.cs` (or the
+ directory `AddOtOpcUaHealth` uses — recon names it)
+- Modify: the `AddOtOpcUaHealth` registration (driver branch)
+- Modify: the observability meter list IF Task 0 found an allowlist
+
+**Steps:**
+1. Failing unit tests: replication unconfigured (no peer, no listener) → `Healthy` (default-OFF
+ must not degrade a plain node); peer configured + `ISyncStatus.Connected == false` → `Degraded`;
+ connected + backlog `null` → `Degraded` (unknown backlog is not healthy); connected + backlog
+ small → `Healthy`.
+2. Implement against `ISyncStatus` + `IOptions` (peer-configured = non-empty
+ `PeerAddress` OR `SyncListenPort > 0` — pass the latter in via options/config).
+3. **Meter allowlist:** if `AddOtOpcUaObservability` restricts meters, add
+ `"ZB.MOM.WW.LocalDb.Replication"` (use `LocalDbMetrics.MeterName`) — this is a silent allowlist;
+ the ScadaBridge live gate is the proof it bites. If there is no allowlist, record that in the
+ recon doc and skip.
+4. Commit `feat(localdb): replication health check + meter export`.
+
+---
+
+### Task 11: DI-pin integration tests over the real Program.cs
+
+**Classification:** standard
+**Estimated implement time:** ~5 min
+**Parallelizable with:** Task 9, Task 10
+
+**Files:**
+- Test: `tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbWiringTests.cs` (xunit.v3,
+ `WebApplicationFactory` — the project already builds the real Program.cs; set
+ `OTOPCUA_ROLES` per test case the way neighboring tests do)
+
+**Pins (each its own test):**
+1. Driver graph: `ILocalDb` resolves, is a singleton, `ReplicatedTables` ordinal-sorted ==
+ `["deployment_artifacts", "deployment_pointer"]` — exact set, both directions load-bearing.
+2. Driver graph: `IDeploymentArtifactCache` resolves to `LocalDbDeploymentArtifactCache`.
+3. Default-OFF pin: `ISyncStatus` resolves with `Connected == false`, `PeerNodeId == null`.
+4. Admin-only graph: `ILocalDb` is NOT registered (`GetService()` is null) and boot does
+ not demand `LocalDb:Path`.
+5. Health: the LocalDb health check is registered in the driver graph.
+
+Point `LocalDb:Path` at a per-test temp file via the factory's config overrides;
+`ClearAllPools` + delete in cleanup. These tests exist because DI extensions without a
+container-built test have shipped inert three times in this family (Secrets 0.2.0/0.2.2,
+ScadaBridge#22). Commit `test(localdb): DI pins over the real host graph`.
+
+---
+
+### Task 12: Two-node convergence harness + tests
+
+**Classification:** high-risk
+**Estimated implement time:** ~5 min (harness) + ~4 min (scenarios) — split the commit if needed
+**Parallelizable with:** none (depends on Tasks 2, 3, 5, 6)
+
+**Files:**
+- Create: `tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDb/LocalDbPairHarness.cs`
+- Test: `tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDb/LocalDbPairConvergenceTests.cs`
+
+**Harness:** copy the pattern from
+`~/Desktop/ScadaBridge/tests/ZB.MOM.WW.ScadaBridge.IntegrationTests/LocalDbSitePairHarness.cs` and
+the library's `ConvergenceFixture` (`~/Desktop/scadaproj/ZB.MOM.WW.LocalDb/tests/.../Convergence/ConvergenceFixture.cs`):
+- Two full stacks over a real loopback Kestrel h2c socket (port 0), **through the real
+ `LocalDbSyncAuthInterceptor`** with a shared test key; node A initiator, node B passive.
+- Both initialized via the **production `LocalDbSetup.OnReady`** — a hand-written schema proves
+ only that the test agrees with itself.
+- DBs owned by the harness, registered as pre-constructed singletons (so MS.DI won't dispose them
+ on host teardown); `KillTransportAsync`/`RestartTransportAsync`; tight `FlushInterval` (50 ms),
+ bounded backoff (2 s); temp files, `ClearAllPools` cleanup; a collection definition to serialize
+ these tests.
+
+**Scenarios:**
+1. `ArtifactStoredOnA_ConvergesToB_ByteIdentical` — store a multi-chunk artifact via
+ `LocalDbDeploymentArtifactCache` on A; B's cache returns byte-identical bytes; both nodes'
+ `__localdb_row_version` rows for the pointer carry the **same HLC and origin node id** (B holds
+ A's row, not a re-derived one).
+2. `RetentionPruneOnA_TombstonesReachB` — third deployment on A prunes the first; B's chunk rows
+ for the pruned deployment disappear and `__localdb_row_version WHERE is_tombstone=1` rows exist
+ on B.
+3. `WritesWhileTransportDown_SurviveRejoin` — kill transport, store on A, restart, converge.
+4. `WrongApiKey_NeverConverges` — harness with mismatched keys: assert **no** convergence after a
+ bounded wait AND (positive control) that the same scenario with matching keys converges — an
+ absence assertion without a positive control passed vacuously in ScadaBridge.
+
+**DoD within this task:** temporarily comment out the two `RegisterReplicated` calls and confirm
+scenarios 1–3 go red (run locally, do not commit the red state); restore. Record the red/green
+evidence in the task notes. Commit `test(localdb): 2-node convergence harness + scenarios`.
+
+---
+
+### Task 13: docker-dev rig configuration
+
+**Classification:** small
+**Estimated implement time:** ~4 min
+**Parallelizable with:** Task 14
+
+**Files:**
+- Modify: `docker-dev/docker-compose.yml`
+
+**Steps** (adjust to the recon's findings on volumes/env style):
+1. Every driver node: a named volume (or existing data mount) backing `/app/data`, and
+ `LocalDb__Path=/app/data/otopcua-localdb.db`.
+2. **site-a pair only** (mirror the ScadaBridge default-OFF posture — site-b stays unreplicated as
+ the pin):
+ - site-a-1: `LocalDb__SyncListenPort=9001`, `LocalDb__Replication__PeerAddress=http://site-a-2:9001`,
+ `LocalDb__Replication__ApiKey=dev-site-a-localdb-sync-key`, `LocalDb__Replication__MaxBatchSize=16`
+ - site-a-2: `LocalDb__SyncListenPort=9001`, same `ApiKey`, same `MaxBatchSize`, **no PeerAddress**
+ (passive; the stream is bidirectional).
+ The key must be byte-identical on both — the interceptor fail-closes on any mismatch and the
+ pair silently stops converging.
+3. `MaxBatchSize=16` because batching is row-count-only against gRPC's 4 MB cap and artifact chunk
+ rows are ≈ 171 KB (16 × 171 KB ≈ 2.7 MB worst case).
+4. `docker compose config` to validate; do NOT bring the rig up in this task (the live gate task
+ owns rig runs).
+5. Commit `chore(localdb): rig config — consolidated path everywhere, replication on the site-a pair`.
+
+---
+
+### Task 14: Documentation
+
+**Classification:** small
+**Estimated implement time:** ~4 min
+**Parallelizable with:** Task 13
+
+**Files:**
+- Create: `docs/operations/2026-07-20-localdb-pair-replication.md` (runbook: enable/disable,
+ ApiKey handling via `${secret:}`, stop/start-together rule, tombstone-retention resurrection
+ window, MaxBatchSize row-count-vs-4MB, never host-`sqlite3` a live WAL DB / cp-triplet recipe)
+- Modify: `docs/Redundancy.md` (a short "pair-local config cache" section: what replicates, what
+ boot-from-cache does and does not cover)
+- Modify: `CLAUDE.md` (this repo): LocalDb adoption row/paragraph — state Phase 1 scope, default-OFF,
+ rig-only enablement
+- Modify: `docs/Configuration.md` if it documents config keys (add the `LocalDb` section)
+
+Commit `docs(localdb): phase-1 runbook + redundancy/config docs`.
+
+---
+
+### Task 15: DoD sweep (offline)
+
+**Classification:** standard
+**Estimated implement time:** ~5 min
+**Parallelizable with:** none (final offline task)
+
+**Steps:**
+1. `dotnet build ZB.MOM.WW.OtOpcUa.slnx` → **0 warnings** (warnings-as-errors).
+2. `dotnet test ZB.MOM.WW.OtOpcUa.slnx` → full suite green; record counts. Baseline any
+ pre-existing failures BEFORE this branch (`git stash` → run → unstash) rather than assuming;
+ report deltas only.
+3. Greps (phrase as "no references from code"; explanatory comments may remain):
+ `grep -rn "LiteDB\|ILocalConfigCache\|GenerationSealedCache" src/ tests/` → no code references.
+4. Confirm the positive-control evidence from Task 12 is recorded.
+5. Update `docs/plans/2026-07-20-localdb-adoption-phase1.md.tasks.json` statuses.
+6. Commit `chore(localdb): phase-1 DoD sweep` and STOP. Report: branch name, commit list, test
+ counts, and that the live gate (Task 16) needs the rig + operator go-ahead.
+
+---
+
+### Task 16: Live gate on the docker-dev rig (run only with explicit user go-ahead)
+
+**Classification:** high-risk (rig)
+**Estimated implement time:** ~30 min wall-clock (mostly waiting)
+**Parallelizable with:** none
+
+**Preamble:** All DB inspection via `cp` of the `db`/`-wal`/`-shm` triplet out of the container
+(`docker cp`) and querying the copy — never host `sqlite3` on the live file (it poisons the WAL
+across virtiofs; root cause of the 2026-07-20 ScadaBridge incident). Metrics via
+`docker run --rm --network container: curlimages/curl:latest -s localhost:/metrics`.
+If an anomaly appears within seconds of one of your own measurements, suspect the measurement
+first — restart both nodes and re-run untouched before blaming the library.
+
+**Checks (all must PASS; record evidence in `docs/plans/2026-07-20-localdb-phase1-live-gate.md`):**
+1. Rig up, site-a pair healthy, `/metrics` on both shows `localdb_` series (proves the meter
+ allowlist work).
+2. Deploy a config through the normal deploy API → both site-a nodes' `deployment_pointer` +
+ `deployment_artifacts` rows are byte-identical with **identical HLC + origin node id** (one
+ node's row replicated, not two independent derivations — both nodes also locally store after
+ apply, so expect LWW-converged identical content either way; assert convergence, and note which
+ origin won).
+3. **Boot-from-cache:** stop the SQL container; restart site-a-2; it comes up serving the cached
+ config with the running-from-cache signal in its log; start SQL again; node recovers fresh
+ behavior.
+4. **Replicated-cache payoff:** wipe site-a-2's local DB file (container stopped), restart it with
+ SQL **up**, let replication repopulate the cache from site-a-1, then repeat check 3's SQL-down
+ restart — it boots from a cache it never wrote itself.
+5. Transport kill (pause site-a-2 container): store/deploy on a-1, oplog depth rises; unpause;
+ drains to 0; converged.
+6. Retention: deploy a 3rd config; pruned deployment's chunks gone on BOTH nodes with tombstone
+ rows present on both.
+7. Both-nodes-together restart: clean rejoin, identical counts, zero
+ `disk I/O error`/`SQLITE_IOERR`/`corrupt` in either log.
+8. site-b pair (default-OFF pin): LocalDb file exists and works locally, `ISyncStatus` health
+ Healthy, no sync connections, no listener on 9001.
+
+Record PASS/FAIL per check with the actual evidence (row counts, md5s, log lines). Commit the gate
+doc. Do not merge — report.
+
+---
+
+## Task persistence
+
+Tasks file: `docs/plans/2026-07-20-localdb-adoption-phase1.md.tasks.json` (same directory).
+Update statuses as tasks complete; any deviation from this plan gets a `deviation` note on the task
+(the ScadaBridge tasks.json convention — the record of *why* is what makes the plan auditable).
diff --git a/docs/plans/2026-07-20-localdb-adoption-phase1.md.tasks.json b/docs/plans/2026-07-20-localdb-adoption-phase1.md.tasks.json
new file mode 100644
index 00000000..8136c608
--- /dev/null
+++ b/docs/plans/2026-07-20-localdb-adoption-phase1.md.tasks.json
@@ -0,0 +1,153 @@
+{
+ "planPath": "docs/plans/2026-07-20-localdb-adoption-phase1.md",
+ "tasks": [
+ {
+ "id": 0,
+ "subject": "Task 0: Preflight + recon (findings doc, STOP conditions)",
+ "status": "completed",
+ "notes": "Feed verified: LocalDb/.Contracts/.Replication all 0.1.1. All 3 STOP conditions clear. Findings: docs/plans/2026-07-20-localdb-phase1-recon.md. 8 deviations recorded (D-1..D-8); D-1/D-3/D-5/D-6 change downstream work materially.",
+ "deviation": "D-1 cache read cannot be keyed by ClusterId at the boot seam (unkeyed newest-pointer read instead); D-3 a third LiteDB test file must be deleted; D-5 WebApplicationFactory is deliberately unused in this repo (use TwoNodeClusterHarness); D-6 driver-only nodes have no ASPNETCORE_URLS so the Kestrel re-bind fallback is wrong. See recon doc \u00a79."
+ },
+ {
+ "id": 1,
+ "subject": "Task 1: Package references and pins (LocalDb 0.1.1, Grpc.AspNetCore, SQLitePCLRaw)",
+ "status": "completed",
+ "blockedBy": [
+ 0
+ ]
+ },
+ {
+ "id": 2,
+ "subject": "Task 2: Schema + LocalDbSetup.OnReady + registration tests",
+ "status": "completed",
+ "blockedBy": [
+ 1
+ ]
+ },
+ {
+ "id": 3,
+ "subject": "Task 3: Fail-closed sync auth interceptor + tests",
+ "status": "completed",
+ "blockedBy": [
+ 1
+ ]
+ },
+ {
+ "id": 4,
+ "subject": "Task 4: LocalDbRegistration + Program.cs DI wiring + config defaults",
+ "status": "completed",
+ "blockedBy": [
+ 2,
+ 3
+ ]
+ },
+ {
+ "id": 5,
+ "subject": "Task 5: Dedicated h2c sync listener + MapZbLocalDbSync (Kestrel URLs-override risk)",
+ "status": "completed",
+ "blockedBy": [
+ 4
+ ]
+ },
+ {
+ "id": 6,
+ "subject": "Task 6: IDeploymentArtifactCache + chunked LocalDb implementation + tests",
+ "status": "completed",
+ "blockedBy": [
+ 2
+ ]
+ },
+ {
+ "id": 7,
+ "subject": "Task 7: Cache write path in DriverHostActor (Props expression-tree trap)",
+ "status": "completed",
+ "blockedBy": [
+ 0,
+ 6
+ ]
+ },
+ {
+ "id": 8,
+ "subject": "Task 8: Boot-from-cache read path + running-from-cache signal",
+ "status": "completed",
+ "blockedBy": [
+ 7
+ ]
+ },
+ {
+ "id": 9,
+ "subject": "Task 9: Delete the dormant LiteDB LocalCache subsystem",
+ "status": "completed",
+ "blockedBy": [
+ 0
+ ]
+ },
+ {
+ "id": 10,
+ "subject": "Task 10: Health check + telemetry meter allowlist",
+ "status": "completed",
+ "blockedBy": [
+ 4
+ ]
+ },
+ {
+ "id": 11,
+ "subject": "Task 11: DI-pin integration tests over the real Program.cs",
+ "status": "completed",
+ "blockedBy": [
+ 5,
+ 10
+ ]
+ },
+ {
+ "id": 12,
+ "subject": "Task 12: Two-node convergence harness + scenarios (+ positive control)",
+ "status": "completed",
+ "blockedBy": [
+ 5,
+ 6
+ ]
+ },
+ {
+ "id": 13,
+ "subject": "Task 13: docker-dev rig configuration (site-a pair on, site-b pin off)",
+ "status": "completed",
+ "blockedBy": [
+ 5
+ ]
+ },
+ {
+ "id": 14,
+ "subject": "Task 14: Documentation (runbook, Redundancy.md, CLAUDE.md)",
+ "status": "completed",
+ "blockedBy": [
+ 8
+ ]
+ },
+ {
+ "id": 15,
+ "subject": "Task 15: DoD sweep (offline) \u2014 STOP and report after this",
+ "status": "completed",
+ "blockedBy": [
+ 7,
+ 8,
+ 9,
+ 11,
+ 12,
+ 13,
+ 14
+ ],
+ "notes": "Offline DoD sweep: full-solution build 0 errors (824 pre-existing OTOPCUA0001 analyzer warnings in driver *test* projects; none reference any LocalDb/DeploymentCache file). Grep: no LiteDB/ILocalConfigCache/GenerationSealedCache code refs (2 explanatory doc-comment lines in ILdapGroupRoleMappingService remain, allowed). Runtime.Tests 407/0/31 twice (the previously-flagged intermittent did NOT reproduce). Host.IntegrationTests LocalDb subset 40/0/0. Full-solution `dotnet test` with stash-baseline NOT run: many suites are infra-gated (driver fixtures, full Akka mesh, LDAP real-bind, shared-SQL DB tests) and cannot run offline on macOS \u2014 deferred to Task 16 / CI. Task 12 positive-control evidence recorded in commit afa5be71. Did NOT merge to master \u2014 stopped here per plan."
+ },
+ {
+ "id": 16,
+ "subject": "Task 16: Live gate on the docker-dev rig (needs explicit user go-ahead)",
+ "status": "completed",
+ "blockedBy": [
+ 15
+ ],
+ "notes": "Live gate RAN on the docker-dev rig with explicit user go-ahead. 8/8 checks pass (check 4 with a documented limitation). Caught + fixed FOUR real defects offline tests missed (4b2f0e6e NU1101 packageSourceMapping, ce9fa07f ASPNETCORE_HTTP_PORTS re-bind, 9137cb41 empty-address-space-on-cache-boot, c6a9f93a cache-not-repopulated-on-RestoreApplied), each with a regression test. Two documented limitations (follow-ups): no replication back-fill of a fully-wiped node; oplog growth on default-OFF nodes. Post-fix: full build 0 errors, Runtime.Tests 409/0, all LocalDb Host.IntegrationTests green; only consistent failure is the infra-gated AbCip_Green_AgainstSim (sim not up). Evidence: 2026-07-20-localdb-phase1-live-gate.md. NOT merged."
+ }
+ ],
+ "lastUpdated": "2026-07-20T00:00:00Z"
+}
\ No newline at end of file
diff --git a/docs/plans/2026-07-20-localdb-adoption-phase2.md b/docs/plans/2026-07-20-localdb-adoption-phase2.md
new file mode 100644
index 00000000..943bcc48
--- /dev/null
+++ b/docs/plans/2026-07-20-localdb-adoption-phase2.md
@@ -0,0 +1,293 @@
+# OtOpcUa LocalDb Adoption — Phase 2 Implementation Plan (alarm-historian store-and-forward)
+
+> **For Claude:** REQUIRED SUB-SKILL: Use superpowers-extended-cc:executing-plans to implement this plan task-by-task.
+>
+> **Execution model:** Optimized for **Claude Opus** agents (`claude --model opus`); dispatch
+> `high-risk` tasks on Opus. Branch **`feat/localdb-phase2`** in `~/Desktop/OtOpcUa` (remote
+> `lmxopcua`). **Prerequisite: Phase 1 (`docs/plans/2026-07-20-localdb-adoption-phase1.md`) is
+> merged (or this branch is stacked on it) and its live gate passed.** Do not merge as part of this
+> plan; stop at the DoD task.
+>
+> **Design authority:** `~/Desktop/scadaproj/docs/plans/2026-07-20-otopcua-localdb-design.md` §3.4,
+> D8, D9. Reference implementation for "replace a bespoke store" phasing: ScadaBridge Phase 2
+> (`~/Desktop/ScadaBridge/docs/plans/2026-07-19-localdb-adoption-phase2.md` + its live gate doc).
+
+**Goal:** Move the alarm-historian store-and-forward buffer (today the standalone
+`alarm-historian.db` owned by `SqliteStoreAndForwardSink`) into the consolidated LocalDb file as a
+replicated table with a primary-gated drain, so a redundant pair no longer loses buffered alarm
+history when a node dies — and delete the sink's bespoke file/connection management outright.
+
+**Architecture:** `alarm_sf_events` (TEXT GUID PK) registered in `LocalDbSetup.OnReady`; the sink
+rewired onto `ILocalDb` behind its unchanged public seam; drain gated on the delivered-snapshot
+Primary role via `PrimaryGatePolicy` (at-least-once across failover, accepted and documented); a
+one-time idempotent migrator from the legacy file, running **after** registration.
+
+**Risk framing (from ScadaBridge):** Phase 2 replaces a *working* mechanism — a harder risk class
+than Phase 1's "add where none existed." Cutover (delete + rewire) lands in **one commit**; there is
+no dual-mechanism period, and the cutover is the test.
+
+**Hard rules:** identical to Phase 1's list (OnReady ordering, no autoincrement/BLOB, fail-closed
+auth already in place, no in-memory SQLite in tests, cp-triplet-only rig inspection), plus:
+- **Legacy-copy column lists must INTERSECT** with what the legacy file actually has
+ (`pragma_table_info` probe) — a missing column throws and readers silently discard every row.
+- Absence assertions need a positive control.
+- DoD greps phrased as "no references from code" (explanatory comments may survive).
+
+---
+
+### Task 0: Recon (produces `docs/plans/2026-07-20-localdb-phase2-recon.md`)
+
+**Classification:** standard
+**Estimated implement time:** ~5 min
+**Parallelizable with:** none
+
+**Files:**
+- Create: `docs/plans/2026-07-20-localdb-phase2-recon.md`
+- Read-only: `src/Core/ZB.MOM.WW.OtOpcUa.Core.AlarmHistorian/SqliteStoreAndForwardSink.cs` (and the
+ whole `Core.AlarmHistorian` project), `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Historian/AlarmHistorianOptions.cs`,
+ the drain worker (whatever forwards batches to `GatewayHistorian`/`SendEvent`),
+ `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Drivers/PrimaryGatePolicy.cs`,
+ `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Drivers/DriverHostActor.cs` (how `_localRole` +
+ driver-member-count reach the gate today)
+
+**Record, with `file:line` citations:**
+1. The sink's exact table schema: name, columns, PK type (**autoincrement?** — decides whether the
+ migrator needs deterministic `mig-{node}-{legacyId}` ids), indices, and any BLOB columns
+ (**STOP condition:** a BLOB payload column cannot be registered — it must become base64 TEXT in
+ the new schema, and the recon must size the largest realistic payload against the 171 KB-ish
+ chunk guidance; alarm events are small JSON, so expect this to be fine, but verify).
+2. The public seam: the interface the drain worker and producers use (e.g. `IAlarmHistorianSink` /
+ enqueue+dequeue+markDelivered+deadLetter methods), so the rewire can keep it byte-compatible.
+3. Semantics to preserve: `Capacity` (1,000,000) enforcement, `MaxAttempts` (10), dead-letter
+ retention (30 d), `BatchSize` (100), `DrainIntervalSeconds` (5) — where each lives.
+4. The drain worker's lifecycle: hosted service or actor? Where a Primary-role check can be
+ injected, and how the delivered-snapshot role (`RedundancyStateChanged` cache) is accessible
+ from it (via `DriverHostActor`, a shared status service, or a message). If the role is only
+ available inside `DriverHostActor`, note the cleanest bridge (e.g. an `IRedundancyRoleView`
+ singleton the actor updates) — that becomes Task 4's shape.
+5. Whether the sink is constructed per-node config path (`AlarmHistorian:DatabasePath`) anywhere
+ else (tests, tooling).
+6. How `AlarmHistorian:Enabled=false` short-circuits (NullAlarmHistorianSink) — the rewire must
+ keep the disabled path allocating no LocalDb tables? No: tables are created unconditionally in
+ `OnReady` (cheap, empty); only the sink/drain stay Null. Note this in the doc.
+
+Commit: `docs(localdb): phase-2 recon findings`.
+
+---
+
+### Task 1: `alarm_sf_events` schema + registration (+ tests)
+
+**Classification:** standard
+**Estimated implement time:** ~4 min
+**Parallelizable with:** none (Task 2 depends on it)
+
+**Files:**
+- Create: `src/Core/ZB.MOM.WW.OtOpcUa.Core.AlarmHistorian/AlarmSfSchema.cs` (depends only on
+ `Microsoft.Data.Sqlite`)
+- Modify: `src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbSetup.cs`
+- Test: extend `LocalDbSetupTests`
+
+Schema shape (adjust column names to the recon's findings — preserve today's semantics):
+```sql
+CREATE TABLE IF NOT EXISTS alarm_sf_events (
+ id TEXT NOT NULL PRIMARY KEY, -- app-minted GUID (never autoincrement)
+ payload_json TEXT NOT NULL,
+ enqueued_at_utc TEXT NOT NULL,
+ attempts INTEGER NOT NULL DEFAULT 0,
+ status TEXT NOT NULL DEFAULT 'pending', -- pending | delivered | dead
+ last_attempt_utc TEXT NULL,
+ dead_at_utc TEXT NULL
+);
+CREATE INDEX IF NOT EXISTS ix_alarm_sf_events_status ON alarm_sf_events(status, enqueued_at_utc);
+```
+`OnReady` order becomes: Phase-1 DDL → `AlarmSfSchema.Apply` → the two Phase-1
+`RegisterReplicated` calls → `RegisterReplicated("alarm_sf_events")` → **migrator (Task 5) last**.
+(All DDL may run before all registrations; the invariant is registration-before-writes.)
+
+TDD: failing test first — exact replicated set becomes
+`["alarm_sf_events", "deployment_artifacts", "deployment_pointer"]` (ordinal-sorted; update the
+Phase-1 exact-set pins in the same commit — they are *supposed* to go red here, that's them
+working). Oplog-capture test for an `alarm_sf_events` insert. Commit
+`feat(localdb): alarm_sf_events replicated table`.
+
+---
+
+### Task 2: Rewire the sink onto `ILocalDb` + delete bespoke file management (the cutover commit, part 1 of 2 — see Task 3)
+
+**Classification:** high-risk
+**Estimated implement time:** ~5 min
+**Parallelizable with:** none
+
+**Files:**
+- Modify: `src/Core/ZB.MOM.WW.OtOpcUa.Core.AlarmHistorian/SqliteStoreAndForwardSink.cs` (or replace
+ with `LocalDbStoreAndForwardSink.cs` — keep the public seam identical either way)
+- Modify: its registration (`AddAlarmHistorian`) to inject `ILocalDb`
+- Modify: `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Historian/AlarmHistorianOptions.cs` — remove
+ `DatabasePath` (a breaking config key removal: note it in the runbook/CHANGELOG task)
+- Test: the sink's existing unit tests, rewired to a temp-file `ILocalDb` via `TestLocalDb`-style
+ helper (create `tests/.../TestSupport` helper if none exists — real DB, never a stub: a stubbed
+ bare `SqliteConnection` lacks `zb_hlc_next()` and fails closed on registered tables)
+
+**Steps:**
+1. Write/port failing tests for the seam's semantics: enqueue, drain batch of `BatchSize`,
+ `MaxAttempts` → dead-letter, capacity enforcement, dead-letter retention purge.
+2. Implement over `ILocalDb.ExecuteAsync/QueryAsync` (anonymous-object params). Delete the private
+ connection/pragma/file-open code and any `PRAGMA journal_mode` calls (LocalDb owns pragmas).
+ GUIDs minted at enqueue (`Guid.NewGuid().ToString("N")`).
+3. `Capacity` enforcement: count-based insert guard (preserve today's overflow behavior per recon).
+4. Core.AlarmHistorian gains a package ref on core `ZB.MOM.WW.LocalDb` (interface only).
+5. Build + project tests green. **Commit together with Task 3** if the drain gate can't compile
+ separately (the ScadaBridge tasks-14/15/16 circular-dependency landmine — check before assuming
+ they're independent commits).
+
+---
+
+### Task 3: Primary-gated drain
+
+**Classification:** high-risk
+**Estimated implement time:** ~5 min
+**Parallelizable with:** none
+
+**Files:** (exact shape from recon item 4)
+- Modify: the drain worker
+- Create (if recon says so): `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Redundancy/IRedundancyRoleView.cs`
+ — a singleton snapshot (`RedundancyRole? LocalRole`, `int DriverMemberCount`) updated by
+ `DriverHostActor` where it already caches `_localRole`
+- Test: drain-worker tests + an actor test pinning that `DriverHostActor` publishes role changes to
+ the view
+
+**Semantics:**
+- Drain runs only when `PrimaryGatePolicy.ShouldServiceAsPrimary(localRole, driverMemberCount)` is
+ true — same policy, same boot-window posture (unknown role drains only when the node is alone).
+- Delivered/dead-letter marks are row UPDATEs → they replicate, so the standby's copy tracks drain
+ progress and does not re-deliver already-marked rows after failover.
+- **At-least-once across failover is accepted:** rows delivered on the old primary whose
+ `delivered` mark hadn't replicated yet will be re-sent by the new primary. Document in the
+ runbook (Task 7); do NOT build dedup.
+- When replication is OFF (default), the gate still applies but `driverMemberCount` for a solo
+ node keeps today's behavior — verify with a test: single-node, role unknown → drains (no
+ regression for unpaired deployments).
+
+TDD: failing tests — secondary role does not drain; primary drains; unknown+alone drains;
+unknown+paired does not. Commit (with Task 2 if coupled):
+`feat(localdb): alarm S&F on LocalDb with primary-gated drain (cutover)`.
+
+---
+
+### Task 4: One-time legacy migrator (`alarm-historian.db` → consolidated)
+
+**Classification:** standard
+**Estimated implement time:** ~5 min
+**Parallelizable with:** Task 6
+
+**Files:**
+- Create: `src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/AlarmSfLegacyMigrator.cs`
+- Modify: `LocalDbSetup.OnReady` — call it **last**, after all `RegisterReplicated` calls
+- Test: `tests/Server/ZB.MOM.WW.OtOpcUa.Host.Tests/AlarmSfLegacyMigratorTests.cs`
+
+Mirror `~/Desktop/ScadaBridge/src/ZB.MOM.WW.ScadaBridge.Host/SiteLocalDbLegacyMigrator.cs`:
+- Source path from the pre-removal `AlarmHistorian:DatabasePath` default (`alarm-historian.db`) —
+ read the raw config key even though the option property is gone.
+- `ShouldMigrate`: skip if file missing or `.migrated` exists. `:memory:`/`file:` sources → no-op.
+- **If the legacy PK is autoincrement (recon):** deterministic ids `mig-{NodeName}-{legacyId}`
+ (node name from the recon-identified config key) — rerunnable without duplicates under
+ `INSERT OR IGNORE`, and no cross-node collision. If already GUIDs, copy as-is.
+- One transaction for the whole copy; `File.Move(path, path + ".migrated")` only after commit;
+ failure throws out of `OnReady` → boot fails, legacy untouched.
+- **Column intersection** via `pragma_table_info` on the legacy table; required-PK guard.
+- Both pair nodes migrate independently; their rows have distinct ids (node-prefixed), so the
+ merged buffer is the union — expected; note that the new primary will drain the standby's
+ migrated rows too.
+
+TDD: failing tests — happy path row counts; idempotent re-run; failure leaves legacy file
+untouched; migrated rows **enter the oplog** (assert `__localdb_oplog` — the registration-order
+pin); older legacy file missing a column still migrates (intersection). Commit
+`feat(localdb): one-time alarm-historian.db migrator`.
+
+---
+
+### Task 5: Convergence + failover scenarios in the pair harness
+
+**Classification:** high-risk
+**Estimated implement time:** ~5 min
+**Parallelizable with:** Task 4
+
+**Files:**
+- Test: `tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDb/AlarmSfConvergenceTests.cs`
+ (reuses the Phase-1 `LocalDbPairHarness`)
+
+**Scenarios:**
+1. `AlarmBurstOnA_ConvergesToB_AndOplogDrains` — enqueue N events on A; identical rowset on B;
+ oplog → 0.
+2. `DeliveredMarksReplicate` — mark rows delivered on A; B's copies show delivered (the
+ no-redeliver-after-failover property, asserted at the data layer).
+3. `WritesWhileTransportDown_SurviveRejoin`.
+4. Positive control: with `RegisterReplicated("alarm_sf_events")` commented out, scenarios 1–3 go
+ red (run locally, record, restore — do not commit red).
+
+Commit `test(localdb): alarm S&F convergence scenarios`.
+
+---
+
+### Task 6: Rig config + docs
+
+**Classification:** small
+**Estimated implement time:** ~4 min
+**Parallelizable with:** Task 4
+
+**Files:**
+- Modify: `docker-dev/docker-compose.yml` — enable `AlarmHistorian__Enabled=true` on the site-a
+ pair if not already (the gate needs a live buffer); remove any `AlarmHistorian__DatabasePath`
+ env vars (key deleted).
+- Modify: `docs/operations/2026-07-20-localdb-pair-replication.md` — add: alarm S&F replication
+ semantics, at-least-once-across-failover statement, migrator behavior (`.migrated` sidecar),
+ `DatabasePath` key removal.
+- Modify: `docs/AlarmHistorian.md` + `CLAUDE.md` — sink now lives in the consolidated LocalDb;
+ drain is primary-gated.
+
+Commit `docs+chore(localdb): phase-2 rig config + docs`.
+
+---
+
+### Task 7: DoD sweep (offline)
+
+**Classification:** standard
+**Estimated implement time:** ~5 min
+**Parallelizable with:** none
+
+1. Full solution build → 0 warnings; full test suite green (deltas vs pre-branch baseline only).
+2. Greps, phrased as "no references from **code**": the old bespoke connection management
+ (`AlarmHistorian:DatabasePath`, direct `new SqliteConnection` inside Core.AlarmHistorian except
+ via schema helpers/tests) — explanatory comments may remain.
+3. Positive-control evidence from Task 5 recorded.
+4. Exact-set replicated-tables pin = 3 tables, both directions.
+5. Update `…phase2.md.tasks.json`; commit `chore(localdb): phase-2 DoD sweep`; STOP and report.
+
+---
+
+### Task 8: Live gate on the docker-dev rig (run only with explicit user go-ahead)
+
+**Classification:** high-risk (rig)
+**Estimated implement time:** ~30 min wall-clock
+**Parallelizable with:** none
+
+Same inspection rules as Phase 1's gate (cp-triplet, curl sidecar, observer-suspicion rule).
+Record in `docs/plans/2026-07-20-localdb-phase2-live-gate.md`:
+1. Migration ran: `.migrated` sidecar present on both site-a nodes; row counts match legacy.
+2. Alarm burst (drive a real driver alarm or the historian-gateway-unreachable path) converges:
+ identical rowsets, oplog drains to 0, dead letters 0.
+3. Only the primary drains: stop the historian gateway egress, buffer builds on BOTH nodes'
+ tables (replicated), but only the primary's drain worker logs attempts.
+4. Failover: stop the primary; the standby (new primary) resumes draining the shared buffer;
+ count of double-delivered events observed and recorded (at-least-once evidence, not a failure).
+5. Both-nodes-together restart clean; zero `disk I/O error`/`SQLITE_IOERR` in logs.
+6. site-b (default-OFF pin): sink works locally, no sync traffic.
+
+Commit the gate doc; report; do not merge.
+
+---
+
+## Task persistence
+
+Tasks file: `docs/plans/2026-07-20-localdb-adoption-phase2.md.tasks.json`. Record deviations per
+task — especially if Tasks 2/3 had to land as one commit (the expected outcome).
diff --git a/docs/plans/2026-07-20-localdb-adoption-phase2.md.tasks.json b/docs/plans/2026-07-20-localdb-adoption-phase2.md.tasks.json
new file mode 100644
index 00000000..7cbfdc9a
--- /dev/null
+++ b/docs/plans/2026-07-20-localdb-adoption-phase2.md.tasks.json
@@ -0,0 +1,15 @@
+{
+ "planPath": "docs/plans/2026-07-20-localdb-adoption-phase2.md",
+ "tasks": [
+ { "id": 0, "subject": "Task 0: Recon — sink schema, seam, drain lifecycle, role-view bridge (STOP conditions)", "status": "pending" },
+ { "id": 1, "subject": "Task 1: alarm_sf_events schema + registration (+ exact-set pin update)", "status": "pending", "blockedBy": [0] },
+ { "id": 2, "subject": "Task 2: Rewire sink onto ILocalDb + delete bespoke file management (cutover 1/2)", "status": "pending", "blockedBy": [1] },
+ { "id": 3, "subject": "Task 3: Primary-gated drain via PrimaryGatePolicy (cutover 2/2 — may co-commit with Task 2)", "status": "pending", "blockedBy": [2] },
+ { "id": 4, "subject": "Task 4: One-time alarm-historian.db legacy migrator", "status": "pending", "blockedBy": [3] },
+ { "id": 5, "subject": "Task 5: Convergence + failover scenarios in the pair harness (+ positive control)", "status": "pending", "blockedBy": [3] },
+ { "id": 6, "subject": "Task 6: Rig config + docs", "status": "pending", "blockedBy": [3] },
+ { "id": 7, "subject": "Task 7: DoD sweep (offline) — STOP and report after this", "status": "pending", "blockedBy": [4, 5, 6] },
+ { "id": 8, "subject": "Task 8: Live gate on the docker-dev rig (needs explicit user go-ahead)", "status": "pending", "blockedBy": [7] }
+ ],
+ "lastUpdated": "2026-07-20T00:00:00Z"
+}
diff --git a/docs/plans/2026-07-20-localdb-phase1-live-gate.md b/docs/plans/2026-07-20-localdb-phase1-live-gate.md
new file mode 100644
index 00000000..379ca4be
--- /dev/null
+++ b/docs/plans/2026-07-20-localdb-phase1-live-gate.md
@@ -0,0 +1,68 @@
+# LocalDb Phase 1 — live gate (docker-dev rig)
+
+> Task 16 of `2026-07-20-localdb-adoption-phase1.md`. Evidence log for the 8 live-rig checks.
+> Rig: local `docker-dev/docker-compose.yml` (6 host nodes, one shared SQL). site-a pair replicates;
+> central + site-b are default-OFF.
+>
+> **Safety rules followed:** all DB inspection via `docker cp` of the `db`/`-wal`/`-shm` triplet out
+> of the container, querying the copy — never host `sqlite3` on the live WAL file. Metrics via a
+> `curlimages/curl` sidecar with `--network container:` (`aspnet:10.0` has no curl).
+
+## Rig build/up
+
+Two real defects surfaced by the rebuild (both fixed on-branch, re-verified):
+
+1. **`NU1101` — packageSourceMapping gap** (`4b2f0e6e`). `ZB.MOM.WW.LocalDb*` was not mapped to the
+ `dohertj2-gitea` feed, so any *clean* restore (Docker image build, CI, fresh clone) failed. Local
+ dev builds masked it via the warm global NuGet cache. Fixed by adding the two patterns; verified
+ with a temp-packages-dir restore.
+2. **`ASPNETCORE_HTTP_PORTS` re-bind gap** (`ce9fa07f`). The Kestrel re-bind read only
+ `urls`/`ASPNETCORE_URLS`; the aspnet:8.0+ base image sets `ASPNETCORE_HTTP_PORTS=8080` (all
+ interfaces) as the container default instead. A driver node that never sets `URLS` fell through to
+ `localhost:5000`, silently moving its health/metrics surface to loopback. Central nodes were fine
+ (they set `ASPNETCORE_URLS=9000` explicitly). Fixed to fall through URLS → HTTP_PORTS/HTTPS_PORTS →
+ 5000; 4 new unit tests. After the fix, driver nodes bind `[::]:8080` + `[::]:9001` (site-a) /
+ `[::]:8080` only (site-b).
+
+Rig: 6 nodes rebuilt from branch `feat/localdb-phase1` @ `ce9fa07f`, all up.
+
+## Checks
+
+| # | Check | Result | Evidence |
+|---|---|---|---|
+| 1 | Rig up, site-a pair healthy, `/metrics` shows `localdb_` series | ✅ PASS | Both site-a nodes `/healthz` 200 Healthy; `/metrics` shows `localdb_oplog_depth` + `localdb_sync_reconnects_total` on meter `ZB.MOM.WW.LocalDb.Replication` (allowlist works); site-a-2 served `/localdb_sync.v1.LocalDbSync/Sync` with routing `match_status=success` (interceptor passed, session established); both oplogs depth 0 |
+| 2 | Deploy → both site-a nodes' pointer + artifacts byte-identical, same HLC + origin | ✅ PASS | Deploy `6e687451…` (rev `efb04c79…`) via `POST :9200/api/deployments` → both site-a-1 and site-a-2 `deployment_pointer` = `SITE-A / 6e687451… / efb04c79… / sha EFB04C79…` (identical); artifact 1 chunk each; **pointer `__localdb_row_version` identical on both: hlc `116955620225646592`, origin node_id `a0576df6-…`, tombstone 0** — one row replicated, not two derivations; both oplogs depth 0 |
+| 3 | Boot-from-cache: SQL down, restart site-a-2, serves cached config w/ signal | ✅ PASS (found + fixed a defect) | With SQL stopped, site-a-2 restart logged `RUNNING FROM CACHE — … booted deployment 6e687451…`; **address space materialised from the cached blob**: `AddressSpaceApplier: applied plan (added=18)` / `OpcUaPublish: applied rebuild (added=18)`. Browse confirms **16 nodes on site-a-2 = 16 on site-a-1**. **Defect the gate caught (`9137cb41`):** pre-fix the rebuild re-read the artifact from the down ConfigDb and no-op'd, so the cache-booted node served **1** node vs the healthy peer's 16 — it logged success but browsed empty. Fixed by passing the in-hand cached blob to `RebuildAddressSpace`; regression test added |
+| 4 | Replicated-cache payoff: wipe a-2 DB, repopulate, then SQL-down boot | ⚠️ PARTIAL — 1 limitation documented, 1 defect fixed | Wiped a-2's LocalDb volume, restarted with **SQL down**: sync session re-established (a-2 served a `Sync` call) but a-2's cache stayed **0/0/0** and a-1's oplog was **0** — the deploy's rows had been acked by the old a-2 and pruned, so there was no delta to send and **no snapshot-resync fired**. ⇒ **Limitation: pure replication does not back-fill a fully-wiped, already-converged node** (library delta-replication; snapshot-resync is gated on the oplog cap, which the running engine never hits). Separately found a **defect (`c6a9f93a`)**: recovery via `RestoreApplied` never wrote the cache (only fresh `ApplyAndAck` did), so a wiped node stayed cache-less. Fixed: `RestoreApplied` now re-caches. Verified — after the fix, wiped a-2 restarted with SQL up: `restored served state … on bootstrap` → cache repopulated `SITE-A / 6e687451`, chunks=1. **Net: a wiped node self-heals its cache from central on next boot (regaining boot-from-cache for future outages); it is not back-filled by its peer during the same outage.** |
+| 5 | Transport kill: pause a-2, oplog rises; unpause, drains to 0 | ✅ PASS | `docker pause` on a-2 severed the sync transport; during the window a-2 accumulated **oplog=2** unacked local writes (its re-cache) while a-1 held 0. `docker unpause` → a-2's oplog **drained to 0 within 5s**, both converged on the same pointer. Note: the rig's redeploys carry an identical revision, so a fresh cache write on a-1 short-circuits (no-op) — the store-during-partition variant is covered by the automated harness scenario `WritesWhileTransportDown_SurviveRejoin` (proven red without `RegisterReplicated`) |
+| 6 | Retention: 3rd deploy prunes oldest; chunks gone + tombstones on BOTH nodes | ✅ PASS | Cached 3 distinct deployment-ids (via deploy + restart re-cache). Both nodes retain exactly the newest 2 (`475df87a`, `962962dc`); the oldest `6e687451`'s chunks are **0 on both** (pruned); **1 artifact tombstone on both** — the prune replicated as a tombstone, not left as a live chunk |
+| 7 | Both-nodes-together restart: clean rejoin, identical counts, no I/O errors | ✅ PASS | Restarted site-a-1 + site-a-2 together: **0** `disk I/O error`/`SQLITE_IOERR`/`corrupt` lines in either log; both re-cached and converged to **identical** counts (ptr `962962dc…`, chunks=2, rowver=3). Byte-identical convergence preserved across a simultaneous restart |
+| 8 | site-b default-OFF pin: cache works locally, Healthy, no sync listener on 9001 | ✅ PASS (+1 finding) | site-b-1 cache works locally (`SITE-B / 6e687451`, chunks=1, from its own apply); listeners **`[::]:8080` only — no 9001**; port 9001 **not bound** (`/proc/net/tcp`); `/healthz` **200**; no sync/reconnect metric series. **Finding (follow-up, not a blocker):** `localdb_oplog_depth=5` — because `OnReady` registers replication unconditionally, a default-OFF node captures cache writes into its oplog but has no peer to ack/prune them, so the oplog grows slowly (≈2 rows/deploy, amplified by the re-cache-on-restore fix). Bounded by the library `MaxOplogRows` cap (1M); worth a periodic pruner or skip-if-unchanged in a follow-up |
+
+## Summary
+
+**8/8 checks pass** (check 4 with a documented limitation). The gate exercised the real image built
+from the branch, the real interceptor over a real loopback+cross-container h2c transport, and real
+`docker cp`-triplet DB inspection — and caught **four real defects that every offline test had
+missed**, each now fixed on-branch with a regression test:
+
+1. **`NU1101` packageSourceMapping gap** (`4b2f0e6e`) — `ZB.MOM.WW.LocalDb*` unmapped to the gitea
+ feed; every clean restore (Docker, CI, fresh clone) failed. Masked locally by the warm NuGet cache.
+2. **`ASPNETCORE_HTTP_PORTS` re-bind gap** (`ce9fa07f`) — the sync-listener re-bind ignored the
+ aspnet:8.0+ container-default port var, silently moving driver health/metrics to loopback:5000.
+3. **Empty address space on boot-from-cache** (`9137cb41`) — the rebuild re-read the artifact from the
+ down ConfigDb and no-op'd, so a cache-booted node logged success but served clients **1 node vs the
+ healthy peer's 16**. The single most important find: boot-from-cache was half-working.
+4. **Cache not repopulated on `RestoreApplied`** (`c6a9f93a`) — a wiped/fresh-volume node recovered its
+ served state from central but stayed cache-less, unable to boot-from-cache on the next outage.
+
+**Two documented limitations (follow-ups, not blockers):**
+
+- **No replication back-fill of a fully-wiped node** (check 4): a peer's already-acked cache rows are
+ pruned from its oplog and snapshot-resync is gated on the (never-hit) oplog cap, so a wiped node is
+ not healed by its peer — it self-heals from central on next boot instead.
+- **Oplog growth on default-OFF nodes** (check 8): `OnReady` registers replication unconditionally, so
+ a node with no peer accumulates unacked/unpruned oplog rows (~2/deploy). Slow, bounded by the 1M cap;
+ a periodic pruner or a skip-if-unchanged cache write would close it.
+
+Not merged, per the plan. The two limitations are recorded as Phase 1 follow-ups.
diff --git a/docs/plans/2026-07-20-localdb-phase1-progress.md b/docs/plans/2026-07-20-localdb-phase1-progress.md
new file mode 100644
index 00000000..cafc7621
--- /dev/null
+++ b/docs/plans/2026-07-20-localdb-phase1-progress.md
@@ -0,0 +1,72 @@
+# LocalDb Phase 1 — Execution Progress (resume state)
+
+**Branch:** `feat/localdb-phase1` (off `master`). Working tree CLEAN, all work committed, nothing pushed.
+**Plan:** `docs/plans/2026-07-20-localdb-adoption-phase1.md`
+**Recon:** `docs/plans/2026-07-20-localdb-phase1-recon.md` (READ THIS — it has the 8 deviations D-1..D-8 that override the plan text).
+**Skill in use:** `superpowers-extended-cc:executing-plans` — batch, verify, commit per task, report between batches.
+**Do NOT merge to master** — plan stops at Task 15 (DoD) and reports; Task 16 is a rig live-gate needing explicit user go-ahead.
+
+## Status: Tasks 0–9 DONE (10 commits). Tasks 10–16 REMAIN.
+
+| # | Task | State | Commit |
+|---|---|---|---|
+| 0 | Preflight + recon | ✅ | `42f85507` |
+| 1 | Package refs (LocalDb 0.1.1, Grpc.AspNetCore 2.76.0) | ✅ | `44255fcb` |
+| 2 | Schema + LocalDbSetup.OnReady + tests | ✅ | `3b65c24b` |
+| 3 | Fail-closed sync auth interceptor + tests | ✅ | `3d29be83` |
+| 4 | LocalDbRegistration + Program.cs DI + appsettings | ✅ | `6aff9a83` |
+| 5 | h2c sync listener + MapZbLocalDbSync (Kestrel) | ✅ | `b9ddf20e` |
+| (fix) | rename Runtime.Deployment → Runtime.DeploymentCache | ✅ | `e771a11a` |
+| 6 | IDeploymentArtifactCache + chunked impl + tests | ✅ | `a38a52b8` |
+| 9 | Delete LiteDB LocalCache | ✅ | `2bae1b4c` |
+| 7 | Cache write path in DriverHostActor | ✅ | `1becf591` |
+| 8 | Boot-from-cache + RunningFromCache signal | ✅ | `a27eff32` |
+| 10 | Health check + telemetry meter allowlist | ⬜ TODO | |
+| 11 | DI-pin integration tests over real host graph | ⬜ TODO | |
+| 12 | 2-node convergence harness + scenarios | ⬜ TODO | |
+| 13 | docker-dev rig config | ⬜ TODO | |
+| 14 | Documentation | ⬜ TODO | |
+| 15 | DoD sweep (offline) — STOP + report | ⬜ TODO | |
+| 16 | Live gate on docker-dev rig — NEEDS USER GO-AHEAD | ⬜ TODO | |
+
+Native task list (TaskCreate ids 1–17 map to plan tasks 0–16) tracks the same; tasks.json in plan dir mirrors it.
+
+## What exists now (key files)
+
+- `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/DeploymentCacheSchema.cs` — DDL, namespace `Runtime.DeploymentCache`.
+- `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/IDeploymentArtifactCache.cs` — `StoreAsync` / `GetCurrentAsync` / **`GetCurrentUnkeyedAsync`** (D-1 addition) + `CachedDeploymentArtifact` record.
+- `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/LocalDbDeploymentArtifactCache.cs` — chunked impl. NOTE the prune guard `AND deployment_id <> @DeploymentId` (bug I found: pointer-target could be pruned under same-tick GUIDs → permanent silent miss).
+- `src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbSetup.cs` — `OnReady` (DDL→RegisterReplicated). PUBLIC (Host has no InternalsVisibleTo).
+- `src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbRegistration.cs` — `AddOtOpcUaLocalDb`, `SyncListenPort(config)`. Registers `IDeploymentArtifactCache → LocalDbDeploymentArtifactCache` singleton.
+- `src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbSyncAuthInterceptor.cs` — fail-closed bearer.
+- `src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/KestrelHttpBinding.cs` — URL parse/re-bind helper (D-6).
+- `src/Server/ZB.MOM.WW.OtOpcUa.Host/Program.cs` — LocalDb wired in `hasDriver` branch (~L176 after AddAlarmHistorian: `AddOtOpcUaLocalDb` + `AddGrpc(interceptor)`); Kestrel re-bind block before `builder.Build()`; `MapZbLocalDbSync()` gated `hasDriver && syncListenPort>0` before `MapOtOpcUaHealth()`.
+- `src/Server/ZB.MOM.WW.OtOpcUa.Host/appsettings.json` — `LocalDb` section (Path, SyncListenPort:0, Replication{}), between AlarmHistorian and Deployment.
+- `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Drivers/DriverHostActor.cs` — cache dep threaded LAST in Props/ctor/forwarding; `CacheAppliedArtifact` (write, after PushDesiredSubscriptions, empty-blob skip, own try/catch); `TryBootFromCache` + `ApplyCachedArtifact` (in Bootstrap catch); `PushDesiredSubscriptionsFromArtifact` split out; `_isRunningFromCache` field; `SingleClusterCacheKey="__single"`; `ReconcileDrivers` now returns `byte[]?`.
+- `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/ServiceCollectionExtensions.cs` — resolves `IDeploymentArtifactCache` (optional, nullable) and passes into DriverHostActor.Props.
+- `src/Core/ZB.MOM.WW.OtOpcUa.Commons/Interfaces/NodeDiagnosticsSnapshot.cs` — added `bool RunningFromCache = false`.
+- Tests (all green): `Host.IntegrationTests/LocalDbSetupTests.cs`, `LocalDbSyncAuthInterceptorTests.cs`, `LocalDbSyncListenerTests.cs`; `Runtime.Tests/Deployment/LocalDbDeploymentArtifactCacheTests.cs`, `Runtime.Tests/Drivers/DriverHostActorArtifactCacheTests.cs`, `DriverHostActorBootFromCacheTests.cs`.
+
+## Load-bearing facts for the remaining tasks
+
+- **Namespace trap (recurs!):** any `Deployment` child namespace under `Runtime` OR `Runtime.Tests` shadows the `Deployment` EF entity → CS0118 across the assembly. Use `...DeploymentCache`. Only a FULL-SOLUTION build catches it (per-project builds pass).
+- **xUnit1051 is an error** in these test projects: every `ILocalDb`/DB call in a test must pass `TestContext.Current.CancellationToken` (xunit.v3 Host.IntegrationTests) — but `Runtime.Tests` is **xunit v2** (no `TestContext.Current`).
+- **Test project reality (D-5):** NO `Host.Tests` project; NO `WebApplicationFactory` (deliberately — `TwoNodeClusterHarness.cs:48` explains why). Host unit tests live in `Host.IntegrationTests` (xunit.v3). Actor tests in `Runtime.Tests` (xunit v2 + Akka.TestKit.Xunit2).
+- **Task 10 (D-7):** `AddOtOpcUaHealth()` takes NO args, called unconditionally at `Program.cs:365`. DON'T change its signature — register the LocalDb check unconditionally and return `Healthy` when LocalDb absent (ActiveNodeHealthCheck precedent). Health check classes: there are ZERO in-repo today; `Host/Health/` has only `HealthEndpoints.cs`. Meter allowlist is REAL and REQUIRED: `Host/Observability/ObservabilityExtensions.cs:30` `o.Meters = [OtOpcUaTelemetry.MeterName]` — add `LocalDbMetrics.MeterName` (from `ZB.MOM.WW.LocalDb.Replication`). Health check reads `ISyncStatus` + `IOptions`.
+- **Task 11:** copy the `TwoNodeClusterHarness` direct-host-build pattern, NOT WebApplicationFactory. Pins: `ILocalDb` singleton + `ReplicatedTables` == `["deployment_artifacts","deployment_pointer"]`; `IDeploymentArtifactCache`→concrete; `ISyncStatus` default-OFF (`Connected==false`); admin-only graph has NO `ILocalDb` + doesn't demand `LocalDb:Path`; health check registered. Set `OTOPCUA_ROLES` per case; `LocalDb:Path`→temp file, `ClearAllPools`+delete triplet in cleanup.
+- **Task 12:** harness copied from `~/Desktop/ScadaBridge/tests/.../LocalDbSitePairHarness.cs` + `~/Desktop/scadaproj/ZB.MOM.WW.LocalDb/tests/.../Convergence/ConvergenceFixture.cs`. Both nodes init via PRODUCTION `LocalDbSetup.OnReady`. Real loopback Kestrel h2c through the REAL interceptor. DBs as pre-constructed singletons. Positive control: comment out both `RegisterReplicated`, confirm scenarios 1–3 red, restore (don't commit red).
+- **Task 13 (D-8):** NO host node has a writable volume today — must ADD named volumes. `docker-dev/docker-compose.yml` services: `central-1/2` (admin,driver, MAIN), `site-a-1/2` (driver, SITE-A), `site-b-1/2` (driver, SITE-B). Env style `Section__Key`. `9001` free container-internal. Enable replication on **site-a pair only** (a-1 initiator w/ PeerAddress, a-2 passive, byte-identical ApiKey, MaxBatchSize=16). site-b stays OFF = default-OFF pin. Validate `docker compose config`; DON'T bring rig up.
+- **Task 15:** baseline pre-existing test failures FIRST (`git stash` → run full `dotnet test` → unstash), report deltas only. **KNOWN intermittent:** one `Runtime.Tests` failure seen ~3× across runs (mine + subagent's), never reproduced under trx logger, NOT one of the new tests — investigate/baseline here.
+- **Warnings-as-errors** across src + most test projects. Pre-existing CS0618/xUnit1051 warnings exist ONLY in `Client.Shared*` (which don't set TreatWarningsAsErrors) — not mine, ignore.
+- **SQLitePCLRaw:** no pin needed — LocalDb 0.1.1 nuspec floors `lib.e_sqlite3` at 2.1.12 (CVE-fixed). Confirmed.
+
+## Follow-ups logged (NOT in scope, don't action without asking)
+
+- `Polly.Core` in `Configuration.csproj` now orphaned (ResilientConfigReader was its only real consumer). Left in place — recorded in recon §7.3b.
+- Pre-existing: `TryRecoverFromStale` marks Applied w/o reconciling drivers (recon §3.5); `Install-Services.ps1` gives driver-only nodes no `ASPNETCORE_URLS` (recon §6.4, worked around by D-6).
+
+## Verify-anything-quick
+
+Full build: `dotnet build ZB.MOM.WW.OtOpcUa.slnx` (expect 0 errors). LocalDb tests:
+`dotnet test tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests --filter "FullyQualifiedName~LocalDb"` and
+`dotnet test tests/Server/ZB.MOM.WW.OtOpcUa.Runtime.Tests --filter "FullyQualifiedName~ArtifactCache|FullyQualifiedName~BootFromCache"`.
diff --git a/docs/plans/2026-07-20-localdb-phase1-recon.md b/docs/plans/2026-07-20-localdb-phase1-recon.md
new file mode 100644
index 00000000..b2f90df4
--- /dev/null
+++ b/docs/plans/2026-07-20-localdb-phase1-recon.md
@@ -0,0 +1,617 @@
+# LocalDb Phase 1 — Task 0 Recon Findings
+
+**Date:** 2026-07-20
+**Branch:** `feat/localdb-phase1` (based on `master`)
+**Plan:** `docs/plans/2026-07-20-localdb-adoption-phase1.md`
+**Design authority:** `~/Desktop/scadaproj/docs/plans/2026-07-20-otopcua-localdb-design.md`
+
+All `file:line` citations are against the branch base commit.
+
+---
+
+## Step 1 — Feed verification: PASS
+
+```
+$ dotnet package search ZB.MOM.WW.LocalDb --source dohertj2-gitea --format json
+ZB.MOM.WW.LocalDb 0.1.1
+ZB.MOM.WW.LocalDb.Contracts 0.1.1
+ZB.MOM.WW.LocalDb.Replication 0.1.1
+```
+
+All three present at **0.1.1**. No STOP. (Note: the local NuGet cache holds both `0.1.0` and `0.1.1`
+— pin exactly `0.1.1` so a stale cache cannot win.)
+
+---
+
+## Step 3 — STOP conditions: NONE TRIGGERED
+
+| STOP condition | Verdict |
+|---|---|
+| Artifact apply path cannot take a post-apply hook without restructuring the actor | **Clear** — `ApplyAndAck` has a clean insertion point (§2.2) |
+| `DeploymentId` / cluster identity are not stable strings/GUIDs | **Clear** — both are stable strings, but cluster identity is **not resolvable at boot**; see D-1 |
+| LiteDB LocalCache is referenced by live code | **Clear** — zero live-code references (§7) |
+
+Three STOP conditions clear. **Eight deviations** from the plan's stated assumptions are recorded in
+§9 — four of them (D-1, D-3, D-5, D-6) change the work materially.
+
+---
+
+## 1. Deploy fetch path
+
+### 1.1 Blob-load sites — there are **five** `CreateDbContext()` calls, not two
+
+| Line | Method | Purpose |
+|---|---|---|
+| `DriverHostActor.cs:513` | `Bootstrap()` | `NodeDeploymentStates` + `Deployments` at PreStart |
+| `DriverHostActor.cs:1472` | `ReconcileDrivers(DeploymentId)` | **loads `ArtifactBlob`** |
+| `DriverHostActor.cs:1543` | `PushDesiredSubscriptions(DeploymentId)` | **loads `ArtifactBlob` again** |
+| `DriverHostActor.cs:2000` | `TryRecoverFromStale()` | latest `Sealed` deployment (id + hash only, no blob) |
+| `DriverHostActor.cs:2029` | `UpsertNodeDeploymentState(...)` | writes `NodeDeploymentState` |
+
+`DriverHostActor.cs:1472-1476`:
+```csharp
+using var db = _dbFactory.CreateDbContext();
+blob = db.Deployments.AsNoTracking()
+ .Where(d => d.DeploymentId == deploymentId.Value)
+ .Select(d => d.ArtifactBlob)
+ .FirstOrDefault() ?? Array.Empty();
+```
+`:1543-1547` is byte-identical apart from the local. **The same blob is read twice per apply.**
+
+### 1.2 Artifact runtime type — there is no instance type
+
+`DeploymentArtifact` (`DeploymentArtifact.cs:48`) is a **static parser class** over
+`ReadOnlySpan`. The blob is never materialised into an object graph.
+
+- Storage: `byte[]` — `Deployment.cs:29` `public byte[] ArtifactBlob { get; init; } = Array.Empty();`
+- Format: **UTF-8 JSON**, `JsonDocument.Parse` (`DeploymentArtifact.cs:66`, `:490`). Not MessagePack.
+- Leniency contract: empty/malformed blob → empty result, **never throws** (`:62`, `:537-541`).
+
+**Consequence for the cache:** we cache raw `byte[]`. No serializer coupling. Good.
+
+### 1.3 Identity types
+
+| Type | Definition | Shape |
+|---|---|---|
+| `DeploymentId` | `Commons/Types/DeploymentId.cs:3` — `readonly record struct DeploymentId(Guid Value)` | `ToString()` → `Value.ToString("N")`, 32 hex chars, **no hyphens** (`:10`) |
+| `RevisionHash` | `Commons/Types/RevisionHash.cs:8` — `readonly record struct RevisionHash(string Value)` | SHA-256 hex, **lowercase 64 chars**, no `0x`; empty invalid (`:3-6`) |
+
+Both are stable, filesystem/SQL-safe strings. Use `DeploymentId.ToString()` (the "N" form) as the
+`deployment_id` TEXT column value.
+
+---
+
+## 2. Post-apply point
+
+### 2.1 `ApplyAndAck` control flow — `DriverHostActor.cs:1413-1458`
+
+```
+1415 _applyingDeploymentId = deploymentId;
+1416 Become(Applying);
+1425 UpsertNodeDeploymentState(..., Applying, null)
+1427 try {
+1429 ReconcileDrivers(deploymentId); // blob fetch #1 — SWALLOWS DB errors
+1430 _currentRevision = revision;
+1431 UpsertNodeDeploymentState(..., Applied, null);
+1432 SendAck(..., ApplyAckOutcome.Applied, null, correlation);
+1436 _opcUaPublishActor?.Tell(RebuildAddressSpace(...));
+1439 PushDesiredSubscriptions(deploymentId); // blob fetch #2 — SWALLOWS DB errors
+1440 OtOpcUaTelemetry.DeploymentApplied.Add(1, "outcome"="ack");
+1441 _log.Info("applied deployment ...");
+1443 }
+1444 catch (Exception ex) { ...Failed ACK... }
+1452 finally { _applyingDeploymentId = null; Become(Steady); }
+```
+
+### 2.2 Where the cache write belongs
+
+**Immediately after `DriverHostActor.cs:1439`**, before `:1440` — but **wrapped in its own
+try/catch**. Reason: `:1432` already sent an `Applied` ACK to the coordinator. An unguarded cache
+write that throws would fall into the `catch` at `:1444` and send a *second*, contradictory
+`Failed` ACK for a deployment the fleet already believes is applied.
+
+### 2.3 Paths that must NOT write to the cache
+
+1. **`catch` at `:1444`** — any apply failure.
+2. **Idempotent short-circuit at `:1402-1409`** — `HandleDispatchFromSteady` returns early with an
+ `Applied` ACK when `_currentRevision == msg.RevisionHash`. `ApplyAndAck` is never entered and no
+ blob is read.
+3. **⚠ THE SILENT-DEGRADATION TRAP.** `ReconcileDrivers:1478-1483` and
+ `PushDesiredSubscriptions:1549-1553` catch DB failures, log a **warning**, and `return` **without
+ rethrowing**. `ApplyAndAck` therefore reaches `:1440` and declares success having applied **zero
+ drivers** when the blob load failed. Caching "whatever we read" would **persist an empty artifact
+ as a successful apply** — poisoning the boot-from-cache path with a config that starts no drivers.
+
+ **Mitigation (load-bearing):** the cache write must be gated on the blob having actually loaded
+ non-empty. Have `ReconcileDrivers` surface the loaded blob upward (a field or an out-param)
+ rather than re-reading it a third time, and skip the cache write when it is
+ `Array.Empty()`. This also removes the duplicate read noted in §1.1.
+
+4. **`RestoreApplied` (`:1514-1528`)** — the bootstrap replay path. Restoring from central; caching
+ here is harmless and arguably desirable, but it is a distinct seam. **Phase 1 scope: skip it.**
+5. **`TryRecoverFromStale` (`:1996-2022`)** — never reads the artifact at all (see §3.4).
+
+---
+
+## 3. Cold-boot path
+
+### 3.1 PreStart — `DriverHostActor.cs:415-436`
+
+Ctor already did `Become(Steady)` at `:411`. PreStart subscribes to topics, spawns the virtual-tag
+and scripted-alarm hosts, then calls `Bootstrap()` at `:435`.
+
+### 3.2 `Bootstrap()` — `DriverHostActor.cs:506-565` — **yes, it queries central SQL at boot**
+
+Two queries inside one `try`: latest `NodeDeploymentState` for self (`:514-518`), then the matching
+`Deployment` row (`:527-532`). Four outcomes: no prior deployments → Steady; `Applied` →
+`RestoreApplied` (`:544`); `Applying` orphan → `ApplyAndAck` replay (`:549`); `Failed` → Steady.
+
+### 3.3 **THE SEAM** — `DriverHostActor.cs:560-564`
+
+```csharp
+catch (Exception ex)
+{
+ _log.Warning(ex, "DriverHost {Node}: ConfigDb unreachable on bootstrap; entering Stale", _localNode);
+ Become(Stale);
+}
+```
+
+**This is the exact point where "central fetch failed at boot" is known.** The cache fallback goes
+between `:562` and `:563`: attempt a local-cache load, and only `Become(Stale)` if the cache is also
+a miss.
+
+⚠ The catch also fires if the *second* query throws after the first succeeded, so `latest` is not in
+scope-valid state from the catch. **The cache must be self-describing** — it must carry its own
+`DeploymentId` + `RevisionHash`, because at this seam the actor has no idea what it should be
+restoring. The planned `deployment_pointer` shape already satisfies this.
+
+### 3.4 The `Stale` state — `DriverHostActor.cs:1343-1379`
+
+`ReconnectInterval = 30s` (`:55`), retry timer started at `:1378`.
+
+The dispatch-ignore at **`:1345-1348`** discards `DispatchDeployment` entirely — no ACK, no
+buffering (contrast `Applying:599` which does `Self.Forward(msg)`). A Stale node looks like a
+timeout to the coordinator. Phase 1 does **not** change this (per design §3.3: a *new* deployment
+still requires central).
+
+`RouteNodeWrite` fast-fails at `:1359-1360` with `"driver host stale (config DB unreachable)"`.
+
+### 3.5 Pre-existing gap worth recording (not Phase 1 scope)
+
+`TryRecoverFromStale` (`:1996-2022`) sets `_currentRevision` and marks `Applied` **without calling
+`ReconcileDrivers` or `PushDesiredSubscriptions`** — documented at `:1692-1695` and `:1706`. A
+Stale-recovered node reports Applied while running **zero drivers**, and
+`HandleDispatchFromSteady:1402` then short-circuits on revision match. Boot-from-cache largely
+neutralises this in practice, but the bookkeeping remains optimistic. **Logged as a follow-up, not
+fixed here.**
+
+### 3.6 `DbHealthProbeActor`
+
+**No interaction with `DriverHostActor` whatsoever.** Spawned as a sibling
+(`ServiceCollectionExtensions.cs:243-246`), consumed only by `OpcUaPublishActor`
+(`OpcUaPublishActor.cs:105`, `:256`, `:528`, `:543-544`). `DriverHostActor` runs its own independent
+30s `retry-db` timer. If running-from-cache should later influence OPC UA ServiceLevel,
+`OpcUaPublishActor`'s DB-health seam is the precedent to mirror.
+
+---
+
+## 4. Cluster identity — **the one real design problem (see D-1)**
+
+**There is no `ClusterId` field on `DriverHostActor` and no accessor for it.** The actor knows only
+`_localNode` (`DriverHostActor.cs:61`, type `Commons.Types.NodeId`, aliased at `:28`).
+
+`ClusterId` is resolved **per-artifact**, by `DeploymentArtifact.ResolveClusterScope(blob, nodeId)`
+(`DeploymentArtifact.cs:485-516`) — it scans the artifact's `Nodes[]` array for a matching `NodeId`
+and reads that element's `ClusterId`. Types: `ClusterScope` (`:46`), `ClusterFilterMode` (`:28`).
+
+`IClusterRoleInfo` (`Commons/Interfaces/IClusterRoleInfo.cs:12`) exposes `LocalNode`, `LocalRoles`,
+`HasRole`, `MembersWithRole`, `RoleLeader`, `RoleLeaderChanged` — **no `ClusterId`**.
+`docker-dev/docker-compose.yml` has `Cluster__Hostname`/`Port`/`Roles`/`SeedNodes` — **no
+`Cluster__ClusterId`**. The `ClusterNode` entity
+(`Configuration/Entities/ClusterNode.cs:7,10`) carries both, but lives in central SQL — exactly
+what is unreachable at the seam that needs it.
+
+> **The bind:** the design keys the cache by `ClusterId` so a replicated pair shares one cache
+> entry. But `ClusterId` is only discoverable *from a blob you already have* or *from central SQL*.
+> At the §3.3 boot-failure seam we have neither.
+
+**Resolution adopted (D-1):** keep `cluster_id` as the `deployment_pointer` PK — pair sharing is the
+whole point of replication and must not be given up. Resolve it at each end as follows:
+
+- **Write path** (`ApplyAndAck`, §2.2): call `DeploymentArtifact.ResolveClusterScope(blob, nodeId)`
+ on the blob just applied. `ClusterFilterMode.ScopeTo` → use its `ClusterId`. `None` (single-cluster
+ artifact) or `Suppress` → fall back to the literal `"__single"` sentinel, matching the actor's own
+ existing degenerate-case convention at `:1827-1830`.
+- **Read path** (boot seam, §3.3): we cannot compute the key, so **do not key the read**. Select the
+ single `deployment_pointer` row, ordered by `applied_at_utc DESC`, `LIMIT 1`. A node belongs to
+ exactly one cluster and its pair peer replicates the same cluster's row, so the table holds one
+ row in every real topology. If more than one row is present (a node was re-homed between
+ clusters), take the newest and **log a warning naming both cluster ids** — an operator-visible
+ signal rather than a silent wrong-config boot.
+- Optional escape hatch: honour a `Cluster:ClusterId` config key when present. **Deferred** — not
+ needed for the rig, and adding a required key contradicts the design's "storage ships
+ unconditionally" posture.
+
+This preserves the replication payoff (design §3.3) and the schema exactly as specified, and
+confines the change to the read query's `WHERE` clause.
+
+---
+
+## 5. `DriverHostActor` construction — the `Props` expression tree
+
+### 5.1 The factory — `DriverHostActor.cs:326-346`
+
+`static Props(...)` takes **16 parameters**, 14 optional, and forwards them **fully positionally**
+into `Akka.Actor.Props.Create(() => new DriverHostActor(...))` at **`:343-346`**. Ctor at
+`:375-412`, assignments `:393-408`.
+
+> ⚠ Six `IActorRef?` parameters and three interface-typed parameters mean many mis-bindings are
+> **type-compatible and therefore compile clean**. Callers use *named* arguments (safe); the
+> internal forwarding at `:343-346` is *positional* (unsafe).
+>
+> **Rule for Task 7: append the new parameter LAST in the `Props` signature, LAST in the ctor
+> signature, and LAST in the positional forwarding list at `:346`. Change nothing else.**
+
+### 5.2 Construction sites — **27 total** (1 production + 26 test)
+
+**Production (1):** `Runtime/ServiceCollectionExtensions.cs:332`, inside `WithOtOpcUaRuntimeActors`
+(`:193`); registered as `DriverHostActorKey` (`:344`). All args named except the first three. Does
+not pass `virtualTagHostOverride`, `scriptedAlarmHostOverride`, or `driverMemberCountProvider`.
+
+`WithOtOpcUaRuntimeActors` callers: `Host/Program.cs:331`,
+`Host.IntegrationTests/TwoNodeClusterHarness.cs:398`, `ServiceCollectionExtensionsTests.cs:42,144`.
+
+**Tests (26)**, all in `tests/Server/ZB.MOM.WW.OtOpcUa.Runtime.Tests/`:
+
+| File | Lines |
+|---|---|
+| `Drivers/DriverHostActorTests.cs` | 28, 60, 81, 118, 143, 189 |
+| `Drivers/DriverHostActorReconcileTests.cs` | 37, 59, 90, 119, 156, 193 |
+| `Drivers/DriverHostActorDiscoveryTests.cs` | 385, 671, 748, 1023 |
+| `Drivers/DriverHostActorWriteRoutingTests.cs` | 148, 208 |
+| `Drivers/DriverHostActorPrimaryGateTests.cs` | 228, 247 |
+| `Drivers/DriverHostActorVirtualTagTests.cs` | 44, 76 |
+| `Drivers/DriverHostActorNativeAlarmTests.cs` | 350 |
+| `Drivers/DriverHostActorNativeAlarmAckRoutingTests.cs` | 126 |
+| `Drivers/DriverHostActorProbeResultDropTests.cs` | 66 |
+| `Drivers/DriverHostActorHistoryWriterTests.cs` | 59 |
+| `Drivers/DiscoveryInjectionEndToEndTests.cs` | 219 |
+| `VirtualTags/DependencyMuxActorTests.cs` | 139 |
+
+No site calls `Props.Create()` or `new DriverHostActor(...)` directly — the only
+`new DriverHostActor` in the tree is inside the factory at `:343`. Because the new parameter is
+optional and appended last, **no test site should need editing.**
+
+### 5.3 Status object for the running-from-cache signal
+
+`NodeDiagnosticsSnapshot` (`Commons/Interfaces/NodeDiagnosticsSnapshot.cs:17-22`) — built in
+`HandleGetDiagnostics` (`DriverHostActor.cs:1381-1398`), a positional record of
+`(NodeId, RevisionHash? CurrentRevision, IReadOnlyList Drivers, DateTime AsOfUtc)`.
+`GetDiagnostics` is handled in **all three states** (`Steady:570`, `Applying:601`, `Stale:1349`), so
+a flag here is observable regardless of state. **This is the natural home for `RunningFromCache`**
+(append last to the record). Consumer: `IFleetDiagnosticsClient` → AdminUI.
+
+`IDriverHealthPublisher` (`:73`, defaulted `:402`) is **not** a fit — the host never calls it; it
+only passes it down to `DriverInstanceActor` children (`:1845`, `:1859`). Per-driver, not per-node.
+
+### 5.4 Async / side-effect IO convention
+
+The actor is **overwhelmingly synchronous-blocking on DB IO** — all five `CreateDbContext()` sites
+use `using var db = ...` with synchronous LINQ on the actor thread. No `async`/`await` in the file,
+no `Task.Run`.
+
+The single async pattern is `ContinueWith(..., TaskScheduler.Default).PipeTo(replyTo)` in
+`HandleRouteNodeWrite` (`:1188-1200`), with an explicit `Sender` capture before the continuation.
+
+**Recommendation for Task 7:** a **synchronous** cache write at the §2.2 point is *consistent with
+the file's existing style* and is correct there — the actor is mid-apply, holds `Applying`, and
+every other DB call on that path already blocks. A fire-and-forget `PipeTo(Self)` would need a
+handler registered in `Steady`, `Applying` **and** `Stale`, or it dead-letters after the
+`Become(Steady)` at `:1456`. **Prefer synchronous + try/catch.** (This deviates from the plan's
+"fire-and-forget"; see D-4.)
+
+`IWithTimers` is already implemented (`:50`, `:282`); `"retry-db"` key in use at `:1378`/`:2009`.
+
+---
+
+## 6. Host composition root
+
+### 6.1 Role flag — `Program.cs:47-50`
+
+```csharp
+var roles = RoleParser.Parse(Environment.GetEnvironmentVariable("OTOPCUA_ROLES"));
+var hasAdmin = roles.Contains("admin");
+var hasDriver = roles.Contains("driver");
+```
+Allowed roles: `admin`, `driver`, `dev` (`Cluster/RoleParser.cs:5-8`); unknown throws (`:25-27`).
+
+### 6.2 `hasDriver` branch — `Program.cs:131-310`
+
+Anchor for the new call: **`Program.cs:173-175`** `builder.Services.AddAlarmHistorian(...)`, inside
+the config-gated cluster at `:154-195`. Must land **before** `AddAkka` at `:314`.
+
+### 6.3 Pipeline — `Program.cs:368-404`
+
+`:383` `MapStaticAssets` · `:385-396` `if (hasAdmin) { ... }` · **`:398` `MapOtOpcUaHealth()`** ·
+`:399` `MapOtOpcUaMetrics()` · `:404` `RunAsync()`.
+
+⚠ **There is no `hasDriver` block in the pipeline half** — only `hasAdmin` is branched on. A
+driver-gated `app.Map*` is a new construct; it goes between `:396` and `:398`.
+
+`Program.cs:406-410` re-exports `public partial class Program`.
+
+### 6.4 Kestrel / URLs — **CONFIRMED: no Kestrel configuration exists**
+
+- `Program.cs:57` `builder.WebHost.UseStaticWebAssets();` — the **only** `builder.WebHost` call.
+- **Zero** `ConfigureKestrel` / `UseUrls` / `ListenAnyIP` / `UseKestrel` anywhere in `src/`.
+- Binding is exclusively via `ASPNETCORE_URLS`: `docker-compose.yml:173` (central-1) and `:240`
+ (central-2) set `http://+:9000`; `launchSettings.json:7` sets `http://localhost:9000`.
+
+Two traps for Task 5:
+
+1. ⚠ **`scripts/install/Install-Services.ps1:186`** (`$hostEnv += "ASPNETCORE_URLS=..."`) is inside
+ `if ($hasAdmin)` at `:185`. **A driver-only Windows-service node gets no `ASPNETCORE_URLS` at
+ all** and binds the ASP.NET default. The plan's parse-URLs-then-rebind snippet would re-bind
+ `http://+:9000` there — a port that node never listened on. The `?? "http://+:9000"` fallback in
+ the plan's snippet is therefore **wrong for driver-only nodes**; see D-6.
+2. ⚠ **`Host.IntegrationTests/TwoNodeClusterHarness.cs:331`** already calls
+ `builder.WebHost.UseKestrel(o => o.Listen(IPAddress.Parse(LoopbackHost), 0));`. Adding a
+ production `ConfigureKestrel` gives two competing explicit configurations in integration tests —
+ last-one-wins. The `syncPort > 0` gate keeps this dormant by default (harness sets no sync port),
+ but any harness that *does* set one will fight the port-0 binding.
+
+**Health route for smoke tests:** `/healthz` — liveness, runs no checks, always 200 while the
+process is up, `AllowAnonymous`. (`Health/HealthEndpoints.cs:48` → `MapZbHealth()`; routes from the
+`ZB.MOM.WW.Health` package.) Also `/health/ready`, `/health/active`, and `/metrics` (`Program.cs:399`).
+
+### 6.5 `SecretsRegistration` template — `Host/Configuration/SecretsRegistration.cs`
+
+Shape for `LocalDbRegistration` to mirror: `public static class` in
+`ZB.MOM.WW.OtOpcUa.Host.Configuration`; `public const string` section/key path constants; a
+`public static bool IsXEnabled(IConfiguration)` predicate using
+`GetValue(key, defaultValue: false)` (**default-deny**) that `Program.cs` can also call;
+`AddOtOpcUaX(this IServiceCollection, IConfiguration)` returning `IServiceCollection`, both args
+`ArgumentNullException.ThrowIfNull`-guarded; XML docs with `` explaining *why* the
+gate is default-deny.
+
+Call-site precedent: `Program.cs:363` `AddOtOpcUaSecrets(...)`; predicate reused inside the `AddAkka`
+lambda at `Program.cs:323`.
+
+### 6.6 Telemetry meter allowlist — **EXISTS. Task 10 step 3 is REQUIRED, not conditional.**
+
+`Host/Observability/ObservabilityExtensions.cs:25-38`:
+```csharp
+return services.AddZbTelemetry(o =>
+{
+ o.ServiceName = "otopcua";
+ o.Meters = [OtOpcUaTelemetry.MeterName]; // <-- ObservabilityExtensions.cs:30
+ o.ActivitySources = [OtOpcUaTelemetry.ActivitySourceName];
+```
+
+`o.Meters` is a **strict allowlist** — `ZbTelemetryExtensions.cs:51-54` iterates it calling
+`metrics.AddMeter(name)`; **no wildcard support**. Anything unlisted is silently not exported.
+Single element today: `OtOpcUaTelemetry.MeterName` = `"ZB.MOM.WW.OtOpcUa"`
+(`Commons/Observability/OtOpcUaTelemetry.cs:19`).
+
+**Action:** add `LocalDbMetrics.MeterName` at `ObservabilityExtensions.cs:30`. This is exactly the
+omission the ScadaBridge live gate caught.
+
+### 6.7 Health registration — `Host/Health/HealthEndpoints.cs:19-41`
+
+`AddOtOpcUaHealth()` **takes no parameters** and is called **unconditionally** at `Program.cs:365`,
+outside both role blocks. Three `.AddTypeActivatedCheck<>` calls (`configdb`, `akka`,
+`admin-leader`); **no registration-time role gating exists** — `ActiveNodeHealthCheck` does runtime
+role scoping instead (returns `Healthy` when the node lacks the role).
+
+⚠ **There are zero `IHealthCheck` implementations in this repo** — all three come from the
+`ZB.MOM.WW.Health.*` packages (`Directory.Packages.props:124-126`, v0.1.0). A LocalDb check would be
+the **first health-check class in the tree**; `Host/Health/` holds only `HealthEndpoints.cs` today.
+
+**Consequence for Task 10:** driver-gating the new check requires changing
+`AddOtOpcUaHealth()`'s signature (add `bool hasDriver` or `IConfiguration`) and its `Program.cs:365`
+call site. **Preferred alternative:** follow the `ActiveNodeHealthCheck` precedent — register
+unconditionally and have the check return `Healthy` when LocalDb is not registered. That keeps the
+signature and matches the "default-OFF must not degrade a plain node" requirement for free. See D-7.
+
+### 6.8 Config files
+
+| File | Top-level sections |
+|---|---|
+| `appsettings.json` | `Serilog`, `Security`, `Secrets`, `ServerHistorian`, `ContinuousHistorization`, `AlarmHistorian`, `Deployment` |
+| `appsettings.driver.json` / `.admin.json` / `.admin-driver.json` / `.Development.json` | `Serilog`, `Security` only |
+
+Overlay selection: `Program.cs:67` sorts roles ordinally and joins with `-` → `admin,driver`
+becomes `appsettings.admin-driver.json`, added `optional: true` at `:70`; env vars (`:77`) and CLI
+args (`:78`) are re-appended after, so deployment overrides outrank the overlay.
+
+**`LocalDb` key search: CONFIRMED ABSENT.** Case-insensitive `localdb` across all `*.json`,
+`*.yml`, `*.ps1`, `*.csproj`, `*.props` matches only the two Phase-1/Phase-2 planning artifacts. No
+conflicting keys in any overlay or in `Install-Services.ps1`.
+
+**Placement convention:** each feature section opens with a `"_comment"` string then `"Enabled":
+false`; per-field caveats use sibling `"_Comment"` keys (`appsettings.json:34`, `:45`). Place
+`LocalDb` between `AlarmHistorian` (ends `:61`) and `Deployment` (`:62`).
+
+---
+
+## 7. LiteDB LocalCache — dormant, safe to delete
+
+### 7.1 Files (6) under `src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/`
+
+`ILocalConfigCache.cs`, `GenerationSnapshot.cs`, `StaleConfigFlag.cs`, `LiteDbConfigCache.cs`
+(also declares `LocalConfigCacheCorruptException` at `:128`), `GenerationSealedCache.cs`,
+`ResilientConfigReader.cs`. Single namespace `...Configuration.LocalCache`. The only two
+`using LiteDB;` statements in `src/` are `LiteDbConfigCache.cs:1` and `GenerationSealedCache.cs:1`.
+
+### 7.2 Reference classification — **no live-code references. NO STOP.**
+
+- **(c) live code elsewhere: ZERO.** Also confirms full dormancy: `SealedBootstrap` (referenced by
+ `docs/v2/v2-release-readiness.md:46`) has **zero hits** in `src/` and `tests/` — it no longer exists.
+- Two benign near-misses:
+ - `Configuration/Services/ILdapGroupRoleMappingService.cs:14` — the string `ResilientConfigReader`
+ inside an **XML doc comment**. Prose only. Needs a comment edit or the doc goes stale.
+ - `Configuration/ZB.MOM.WW.OtOpcUa.Configuration.csproj:29` — ``,
+ the only one in the repo.
+- Dozens of `LocalConfigCacheCorruptException` hits under `src/**/bin/**/*.xml` are **generated doc
+ build artifacts** (`GenerateDocumentationFile=true`), not source.
+
+### 7.3 ⚠ Test-file drift — the plan's delete list is incomplete
+
+Plan `…phase1.md:484-485` names only `GenerationSealedCacheTests.cs` and
+`ResilientConfigReaderTests.cs`. Actual test files in
+`tests/Core/ZB.MOM.WW.OtOpcUa.Configuration.Tests/`:
+
+| File | Status |
+|---|---|
+| `GenerationSealedCacheTests.cs` | in plan ✅ |
+| `ResilientConfigReaderTests.cs` | in plan ✅ — note `StaleConfigFlagTests` is co-located here at `:323` |
+| **`LiteDbConfigCacheTests.cs`** | **MISSING from the plan** — will fail to compile if the source is deleted (`:125` calls `new LiteDB.LiteDatabase(...)` directly) |
+
+See D-3.
+
+### 7.3b Outcome (Task 9, executed 2026-07-20)
+
+Deleted: the 6 `LocalCache/` sources and **all three** test files (including
+`LiteDbConfigCacheTests.cs`, which the plan omitted — D-3). `LiteDB` removed from both
+`Configuration.csproj` and `Directory.Packages.props`. The stale XML-doc reference at
+`ILdapGroupRoleMappingService.cs:14` was rewritten to state the present truth (no fallback exists;
+reviving it means an admin-side cache on LocalDb, not restoring the old pipeline) rather than
+deleted, per the DoD's "explanatory prose may remain". Configuration tests: 92/92 green.
+
+**Follow-up found while executing:** `Polly.Core` in `Configuration.csproj` is now orphaned too —
+`ResilientConfigReader` was its only real consumer in that project (remaining `Polly` mentions are
+comments). Left in place deliberately: removing it is outside Task 9's scope and could break a
+consumer relying on the transitive flow. Worth a separate cleanup.
+
+### 7.4 Package removal
+
+- `PackageReference`: `Configuration.csproj:29` — only consumer.
+- `PackageVersion`: `Directory.Packages.props:35` — `LiteDB` `5.0.21`.
+
+Both safe to delete; no transitive consumer.
+
+---
+
+## 8. Test projects, rig, packages
+
+### 8.1 Test project layout
+
+| Question | Answer |
+|---|---|
+| `tests/Server/ZB.MOM.WW.OtOpcUa.Host.Tests/` | **DOES NOT EXIST** — there is no Host *unit*-test project at all |
+| Closest Host project | `Host.IntegrationTests` — despite the name it carries plenty of pure unit tests (`LdapOptionsValidatorTests`, `ServerHistorianOptionsValidatorTests`, `ResilienceInvokerFactoryRegistrationTests`, …) |
+| `tests/Server/ZB.MOM.WW.OtOpcUa.Runtime.Tests/` | **EXISTS** |
+| DriverHostActor tests | `Runtime.Tests/Drivers/` — 11 `DriverHostActor*Tests.cs` files. **xunit v2** (`xunit` 2.9.3), references **`Akka.TestKit.Xunit2`** ✅ |
+| `Host.IntegrationTests` xunit version | **xunit.v3** 1.1.0; SDK `Microsoft.NET.Sdk.Web` |
+
+`Directory.Packages.props` documents that AdminUI.Tests / ControlPlane.Tests / **Runtime.Tests** are
+the three xunit-v2 holdouts, held there by `Akka.TestKit.Xunit2` (verified 2026-07-13: no xunit.v3
+Akka TestKit exists). This matches the design's "actor tests live in an xunit2 TestKit project".
+
+### 8.2 ⚠ `WebApplicationFactory` is **deliberately not used** — Task 11 must be rewritten
+
+The only mention of the type in the entire `tests/` tree is a comment explaining why it is avoided:
+
+`Host.IntegrationTests/TwoNodeClusterHarness.cs:48`
+```
+/// Why not WebApplicationFactory<Program>? Program.cs reads OTOPCUA_ROLES ...
+```
+
+The actual pattern builds the host directly — `TwoNodeClusterHarness.cs:329`:
+```csharp
+var builder = WebApplication.CreateBuilder(new WebApplicationOptions { Args = [] });
+```
+`Microsoft.AspNetCore.Mvc.Testing` is still referenced (csproj `:15`) but only for transitive
+test-host plumbing. **Copy `TwoNodeClusterHarness`, not `WebApplicationFactory`.** See D-5.
+
+`Host.IntegrationTests` also has its own `docker-compose.yml` (sql `14331`, ldap `3894`) gated by
+`DockerFixtureAvailability.cs`.
+
+### 8.3 docker-dev rig topology — plan's site-a/site-b assumption is **correct**
+
+Single Akka mesh (`otopcua`), seeded by `central-1`. Tenant separation is by `ServerCluster.ClusterId`
+rows (MAIN / SITE-A / SITE-B) inside **one shared ConfigDb** — not separate meshes.
+
+| Service | Roles | Cluster |
+|---|---|---|
+| `central-1`, `central-2` | `admin,driver` | MAIN |
+| `site-a-1`, `site-a-2` | `driver` | SITE-A |
+| `site-b-1`, `site-b-2` | `driver` | SITE-B |
+| `sql`, `migrator`, `cluster-seed`, `traefik` | — | infra |
+
+**There is no admin-only service** — the "admin" nodes are fused `admin,driver`. All six host nodes
+share the `&otopcua-host` YAML anchor.
+
+⚠ **NO HOST NODE HAS A WRITABLE VOLUME.** The only `volumes:` in the file are `sql` →
+`otopcua-mssql-data`, plus two read-only binds (`./seed:ro`, `./traefik-dynamic.yml:ro`). All six
+nodes write to the container's ephemeral layer, destroyed on recreate. This is already an accepted
+pattern (per-container `Secrets:SqlitePath=otopcua-secrets.db`, replicated over Akka rather than
+persisted). **Task 13 must add six named volumes** — there is nothing to piggyback on. See D-8.
+
+**Env style: confirmed double-underscore `Section__Key`**, with array indices and dictionary keys as
+a third segment (`Cluster__SeedNodes__0`, `Security__Ldap__GroupToRole__ReadOnly`). Exceptions that
+are plain SCREAMING_SNAKE (read directly, not config-bound): `OTOPCUA_ROLES`,
+`ZB_SECRETS_MASTER_KEY`, `GALAXY_MXGW_API_KEY`, `OTOPCUA_CONFIG_CONNECTION`, `ASPNETCORE_URLS`.
+
+**Ports.** Host-published: `14330`→sql, `4840`-`4845`→the six nodes' OPC UA, `9200`→traefik web,
+`8089`→traefik dashboard. Container-internal on every node: `4053` (Akka remoting), `9000`
+(Kestrel), `4840` (OPC UA). Also reserved on this machine: `14331`/`3894` (Host.IntegrationTests
+compose), `80`/`8080` (sibling scadabridge/scadalink stack), `10.100.0.35:3893` (shared GLAuth).
+
+**`9001` (the plan's choice) is free** container-internal and need not be published. ✅
+
+### 8.4 Package versions — `Directory.Packages.props`
+
+| Package | Version | Line |
+|---|---|---|
+| `Grpc.Core.Api` | 2.76.0 | 32 |
+| `Grpc.Net.Client` | 2.76.0 | 33 |
+| **`Grpc.AspNetCore`** | **NOT PRESENT** | — |
+| `Google.Protobuf` | 3.34.1 | 31 |
+| `Microsoft.Data.Sqlite` | 10.0.7 | ~55 |
+| `SQLitePCLRaw.bundle_e_sqlite3` | 2.1.12 | ~108 |
+| `LiteDB` | 5.0.21 | 35 |
+
+Both Grpc floors and the Protobuf floor are already satisfied — **no bumps needed**, only the new
+`Grpc.AspNetCore` entry (use 2.76.0 to match the train; watch for NU1605/NU1608 if it wants a
+Protobuf newer than the pinned 3.34.1).
+
+⚠ **`SQLitePCLRaw` is a surgical direct pin, not a transitive one.**
+`CentralPackageTransitivePinningEnabled` is deliberately **off** (it breaks the Roslyn version
+split). The 2.1.12 pin is promoted by a **direct `PackageReference` in `Core.AlarmHistorian`**
+(`Core.AlarmHistorian.csproj:15` + `:21`) — today the only Sqlite consumer in `src/`.
+
+> **Any new project that pulls `Microsoft.Data.Sqlite` MUST also take a direct
+> ``,** or it silently resolves the
+> vulnerable 2.1.11 native bundle. This applies to **Host** and **Runtime** (both gain `ZB.MOM.WW.LocalDb`,
+> which depends on `Microsoft.Data.Sqlite`) and to every test project that builds a real `ILocalDb`.
+> Copy the `Core.AlarmHistorian.csproj:15,21` pair exactly. The plan's Task 1 mentions
+> `SQLitePCLRaw.lib.e_sqlite3` "if the audit flags it" — the correct package is
+> **`bundle_e_sqlite3`**, and it is **required, not conditional**.
+
+---
+
+## 9. Deviations from the plan
+
+Recorded per the tasks.json `deviation` convention. **D-1, D-3, D-5, D-6 change the work
+materially.**
+
+| # | Plan says | Reality | Resolution |
+|---|---|---|---|
+| **D-1** | Cache keyed by `ClusterId`; "cluster identity" is a recon item with a STOP condition | `ClusterId` is stable, but **not resolvable at the boot seam** — it lives inside the artifact or in the unreachable central DB (§4) | Keep `cluster_id` as PK (pair sharing is the point). **Write:** resolve via `DeploymentArtifact.ResolveClusterScope`, `"__single"` sentinel for the degenerate case. **Read:** unkeyed `ORDER BY applied_at_utc DESC LIMIT 1`, warn if >1 row. No STOP. |
+| **D-2** | Cache write "fire-and-forget (`PipeTo`-style or guarded `Task.Run`)" | The actor has **no async DB IO at all**; a `PipeTo(Self)` needs handlers in all three states or it dead-letters after `Become(Steady)` (§5.4) | Synchronous write in its own try/catch at `:1439`. Matches the file's style; failure still cannot fail the apply. |
+| **D-3** | Task 9 deletes 2 test files | **3** test files reference the subsystem — `LiteDbConfigCacheTests.cs` is missing from the list (§7.3) | Delete all three. Also fix the stale XML-doc mention at `ILdapGroupRoleMappingService.cs:14`. |
+| **D-4** | Cache the artifact after a successful apply | `ReconcileDrivers`/`PushDesiredSubscriptions` **swallow** DB errors, so a "successful" apply can have loaded an **empty blob** and started zero drivers (§2.3) | Gate the write on a non-empty blob; surface the blob from `ReconcileDrivers` instead of a third read. **Load-bearing — without it the cache can be poisoned with an empty config.** |
+| **D-5** | Task 11 uses `WebApplicationFactory` | Deliberately **not used** in this repo; `TwoNodeClusterHarness.cs:48` documents why (Program.cs reads `OTOPCUA_ROLES` from the environment) (§8.2) | Rewrite Task 11 against the `TwoNodeClusterHarness` direct-host-build pattern. |
+| **D-6** | Task 5 re-binds `ASPNETCORE_URLS ?? "http://+:9000"` | `Install-Services.ps1:185-186` sets `ASPNETCORE_URLS` **only when `hasAdmin`** — the fallback would bind a driver-only node to a port it never served (§6.4) | Re-bind only when a URL is actually configured; when absent, add **only** the sync listener. Also note `TwoNodeClusterHarness.cs:331` already calls `UseKestrel(Listen(…, 0))`. |
+| **D-7** | Task 10 driver-gates the health check | `AddOtOpcUaHealth()` takes no args and is called unconditionally at `Program.cs:365`; no registration-time gating exists anywhere (§6.7) | Register unconditionally; return `Healthy` when LocalDb is absent — the `ActiveNodeHealthCheck` precedent. Avoids a signature change and satisfies "default-OFF must not degrade a plain node". |
+| **D-8** | Task 13 "existing data mount" | **No host node has any writable volume** (§8.3) | Add six named volumes. Note: `9001` is free container-internal. Meter allowlist (§6.6) is **required**, not conditional. |
+
+### Follow-ups logged, not fixed here
+
+1. **`TryRecoverFromStale` marks `Applied` without reconciling drivers** (§3.5) — pre-existing,
+ documented in-code at `:1692-1695`. Out of Phase 1 scope.
+2. **Duplicate blob read per apply** (`:1472` and `:1543`, §1.1) — D-4's fix removes one of them as a
+ side effect.
+3. **`Install-Services.ps1` gives driver-only nodes no `ASPNETCORE_URLS`** (§6.4) — pre-existing;
+ D-6 works around it rather than fixing it.
diff --git a/src/Core/ZB.MOM.WW.OtOpcUa.Commons/Interfaces/NodeDiagnosticsSnapshot.cs b/src/Core/ZB.MOM.WW.OtOpcUa.Commons/Interfaces/NodeDiagnosticsSnapshot.cs
index e31b62ff..a2fafebd 100644
--- a/src/Core/ZB.MOM.WW.OtOpcUa.Commons/Interfaces/NodeDiagnosticsSnapshot.cs
+++ b/src/Core/ZB.MOM.WW.OtOpcUa.Commons/Interfaces/NodeDiagnosticsSnapshot.cs
@@ -14,8 +14,16 @@ public sealed record DriverInstanceDiagnostics(
/// Per-node diagnostics returned by IFleetDiagnosticsClient. Populated by the node's
/// local DriverHostActor via a request/response over Akka.
///
+///
+/// True when the node booted its configuration from the node-local artifact cache because the
+/// central ConfigDb was unreachable. Such a node looks entirely healthy — full address space,
+/// live values — but its configuration is frozen and no deployment can reach it, so the state
+/// needs to be explicitly visible rather than inferred. Defaults to false, which is also what
+/// every node that booted normally reports.
+///
public sealed record NodeDiagnosticsSnapshot(
NodeId NodeId,
RevisionHash? CurrentRevision,
IReadOnlyList Drivers,
- DateTime AsOfUtc);
+ DateTime AsOfUtc,
+ bool RunningFromCache = false);
diff --git a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/GenerationSealedCache.cs b/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/GenerationSealedCache.cs
deleted file mode 100644
index 220b9f9b..00000000
--- a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/GenerationSealedCache.cs
+++ /dev/null
@@ -1,197 +0,0 @@
-using LiteDB;
-
-namespace ZB.MOM.WW.OtOpcUa.Configuration.LocalCache;
-
-///
-/// Generation-sealed LiteDB cache per docs/v2/plan.md and Phase 6.1
-/// Stream D.1. Each published generation writes one read-only LiteDB file under
-/// <cache-root>/<clusterId>/<generationId>.db. A per-cluster
-/// CURRENT text file holds the currently-active generation id; it is updated
-/// atomically (temp file + ) only after
-/// the sealed file is fully written.
-///
-///
-/// Mixed-generation reads are impossible: any read opens the single file pointed to
-/// by CURRENT, which is a coherent snapshot. Corruption of the CURRENT file or the
-/// sealed file surfaces as — the reader
-/// fails closed rather than silently falling back to an older generation. Recovery path
-/// is to re-fetch from the central DB (and the Phase 6.1 Stream C UsingStaleConfig
-/// flag goes true until that succeeds).
-///
-/// This cache is the read-path fallback when the central DB is unreachable. The
-/// write path (draft edits, publish) bypasses the cache and fails hard on DB outage per
-/// Stream D.2 — inconsistent writes are worse than a temporary inability to edit.
-///
-public sealed class GenerationSealedCache
-{
- private const string CollectionName = "generation";
- private const string CurrentPointerFileName = "CURRENT";
- private readonly string _cacheRoot;
-
- // Private per-database BsonMapper with the entity pre-registered. LiteDB's default
- // BsonMapper.Global is a process-wide singleton whose lazy per-type member registration is
- // not thread-safe across concurrently-constructed LiteDatabase instances; a seal racing a
- // read (or this cache racing LiteDbConfigCache) corrupts the global mapper, surfacing as
- // "Member … not found on BsonMapper" or a bogus duplicate-_id insert.
- private static BsonMapper BuildMapper()
- {
- var mapper = new BsonMapper();
- mapper.Entity();
- return mapper;
- }
-
- /// Root directory for all clusters' sealed caches.
- public string CacheRoot => _cacheRoot;
-
- /// Initializes a new instance of the GenerationSealedCache class.
- /// The root directory for the cache.
- public GenerationSealedCache(string cacheRoot)
- {
- ArgumentException.ThrowIfNullOrWhiteSpace(cacheRoot);
- _cacheRoot = cacheRoot;
- Directory.CreateDirectory(_cacheRoot);
- }
-
- ///
- /// Seal a generation: write the snapshot to <cluster>/<generationId>.db,
- /// mark the file read-only, then atomically publish the CURRENT pointer. Existing
- /// sealed files for prior generations are preserved (prune separately).
- ///
- /// The generation snapshot to seal.
- /// The cancellation token.
- /// A task representing the asynchronous operation.
- public async Task SealAsync(GenerationSnapshot snapshot, CancellationToken ct = default)
- {
- ArgumentNullException.ThrowIfNull(snapshot);
- ct.ThrowIfCancellationRequested();
-
- var clusterDir = Path.Combine(_cacheRoot, snapshot.ClusterId);
- Directory.CreateDirectory(clusterDir);
- var sealedPath = Path.Combine(clusterDir, $"{snapshot.GenerationId}.db");
-
- if (File.Exists(sealedPath))
- {
- // Already sealed — idempotent. Treat as no-op + update pointer in case an earlier
- // seal succeeded but the pointer update failed (crash recovery).
- WritePointerAtomically(clusterDir, snapshot.GenerationId);
- return;
- }
-
- var tmpPath = sealedPath + ".tmp";
- try
- {
- using (var db = new LiteDatabase(new ConnectionString { Filename = tmpPath, Upgrade = false }, BuildMapper()))
- {
- var col = db.GetCollection(CollectionName);
- col.Insert(snapshot);
- }
-
- File.Move(tmpPath, sealedPath);
- File.SetAttributes(sealedPath, File.GetAttributes(sealedPath) | FileAttributes.ReadOnly);
- WritePointerAtomically(clusterDir, snapshot.GenerationId);
- }
- catch
- {
- try { if (File.Exists(tmpPath)) File.Delete(tmpPath); } catch { /* best-effort */ }
- throw;
- }
-
- await Task.CompletedTask;
- }
-
- ///
- /// Read the current sealed snapshot for . Throws
- /// when the pointer is missing
- /// (first-boot-no-snapshot case) or when the sealed file is corrupt. Never silently
- /// falls back to a prior generation.
- ///
- /// The cluster ID to read the snapshot for.
- /// The cancellation token.
- /// A task representing the asynchronous operation containing the generation snapshot.
- public Task ReadCurrentAsync(string clusterId, CancellationToken ct = default)
- {
- ArgumentException.ThrowIfNullOrWhiteSpace(clusterId);
- ct.ThrowIfCancellationRequested();
-
- var clusterDir = Path.Combine(_cacheRoot, clusterId);
- var pointerPath = Path.Combine(clusterDir, CurrentPointerFileName);
- if (!File.Exists(pointerPath))
- throw new GenerationCacheUnavailableException(
- $"No sealed generation for cluster '{clusterId}' at '{clusterDir}'. First-boot case: the central DB must be reachable at least once before cache fallback is possible.");
-
- long generationId;
- try
- {
- var text = File.ReadAllText(pointerPath).Trim();
- generationId = long.Parse(text, System.Globalization.CultureInfo.InvariantCulture);
- }
- catch (Exception ex)
- {
- throw new GenerationCacheUnavailableException(
- $"CURRENT pointer at '{pointerPath}' is corrupt or unreadable.", ex);
- }
-
- var sealedPath = Path.Combine(clusterDir, $"{generationId}.db");
- if (!File.Exists(sealedPath))
- throw new GenerationCacheUnavailableException(
- $"CURRENT points at generation {generationId} but '{sealedPath}' is missing — fails closed rather than serving an older generation.");
-
- try
- {
- using var db = new LiteDatabase(new ConnectionString { Filename = sealedPath, ReadOnly = true }, BuildMapper());
- var col = db.GetCollection(CollectionName);
- var snapshot = col.FindAll().FirstOrDefault()
- ?? throw new GenerationCacheUnavailableException(
- $"Sealed file '{sealedPath}' contains no snapshot row — file is corrupt.");
- return Task.FromResult(snapshot);
- }
- catch (GenerationCacheUnavailableException) { throw; }
- catch (Exception ex) when (ex is LiteException or InvalidDataException or IOException
- or NotSupportedException or FormatException)
- {
- throw new GenerationCacheUnavailableException(
- $"Sealed file '{sealedPath}' is corrupt or unreadable — fails closed rather than falling back to an older generation.", ex);
- }
- }
-
- /// Return the generation id the CURRENT pointer points at, or null if no pointer exists.
- /// The cluster ID to get the current generation ID for.
- /// The generation ID, or null if no pointer exists.
- public long? TryGetCurrentGenerationId(string clusterId)
- {
- ArgumentException.ThrowIfNullOrWhiteSpace(clusterId);
- var pointerPath = Path.Combine(_cacheRoot, clusterId, CurrentPointerFileName);
- if (!File.Exists(pointerPath)) return null;
- try
- {
- return long.Parse(File.ReadAllText(pointerPath).Trim(), System.Globalization.CultureInfo.InvariantCulture);
- }
- catch
- {
- return null;
- }
- }
-
- private static void WritePointerAtomically(string clusterDir, long generationId)
- {
- var pointerPath = Path.Combine(clusterDir, CurrentPointerFileName);
- var tmpPath = pointerPath + ".tmp";
- File.WriteAllText(tmpPath, generationId.ToString(System.Globalization.CultureInfo.InvariantCulture));
- if (File.Exists(pointerPath))
- File.Replace(tmpPath, pointerPath, destinationBackupFileName: null);
- else
- File.Move(tmpPath, pointerPath);
- }
-}
-
-/// Sealed cache is unreachable — caller must fail closed.
-public sealed class GenerationCacheUnavailableException : Exception
-{
- /// Initializes a new instance of the GenerationCacheUnavailableException class.
- /// The error message.
- public GenerationCacheUnavailableException(string message) : base(message) { }
- /// Initializes a new instance of the GenerationCacheUnavailableException class with an inner exception.
- /// The error message.
- /// The inner exception.
- public GenerationCacheUnavailableException(string message, Exception inner) : base(message, inner) { }
-}
diff --git a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/GenerationSnapshot.cs b/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/GenerationSnapshot.cs
deleted file mode 100644
index b883719d..00000000
--- a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/GenerationSnapshot.cs
+++ /dev/null
@@ -1,20 +0,0 @@
-namespace ZB.MOM.WW.OtOpcUa.Configuration.LocalCache;
-
-///
-/// A self-contained snapshot of one generation — enough to rebuild the address space on a node
-/// that has lost DB connectivity. The payload is the JSON-serialized sp_GetGenerationContent
-/// result; the local cache doesn't inspect the shape, it just round-trips bytes.
-///
-public sealed class GenerationSnapshot
-{
- /// Gets or sets the auto-generated LiteDB ID.
- public int Id { get; set; } // LiteDB auto-ID
- /// Gets or sets the cluster identifier.
- public required string ClusterId { get; set; }
- /// Gets or sets the generation identifier.
- public required long GenerationId { get; set; }
- /// Gets or sets the time this snapshot was cached.
- public required DateTime CachedAt { get; set; }
- /// Gets or sets the JSON-serialized payload content.
- public required string PayloadJson { get; set; }
-}
diff --git a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/ILocalConfigCache.cs b/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/ILocalConfigCache.cs
deleted file mode 100644
index 9483de8d..00000000
--- a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/ILocalConfigCache.cs
+++ /dev/null
@@ -1,32 +0,0 @@
-namespace ZB.MOM.WW.OtOpcUa.Configuration.LocalCache;
-
-///
-/// Per-node local cache of the most-recently-applied generation(s). Used to bootstrap the
-/// address space when the central DB is unreachable (degraded-but-running).
-///
-///
-/// Concurrency contract: implementations must serialize writes — specifically,
-/// for the same (ClusterId, GenerationId) from concurrent
-/// callers must not produce duplicate rows. Reads may run concurrently with reads and writes.
-/// The implementation enforces this via an instance-level
-/// around the find-then-insert/update window.
-///
-public interface ILocalConfigCache
-{
- /// Retrieves the most recent generation snapshot for the specified cluster.
- /// The cluster identifier.
- /// The cancellation token.
- /// The most recent generation snapshot, or null if none exists.
- Task GetMostRecentAsync(string clusterId, CancellationToken ct = default);
- /// Stores a generation snapshot in the local cache.
- /// The generation snapshot to store.
- /// The cancellation token.
- /// A task that represents the asynchronous operation.
- Task PutAsync(GenerationSnapshot snapshot, CancellationToken ct = default);
- /// Removes old generations, keeping only the most recent N.
- /// The cluster identifier.
- /// The number of latest generations to keep.
- /// The cancellation token.
- /// A task that represents the asynchronous operation.
- Task PruneOldGenerationsAsync(string clusterId, int keepLatest = 10, CancellationToken ct = default);
-}
diff --git a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/LiteDbConfigCache.cs b/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/LiteDbConfigCache.cs
deleted file mode 100644
index 4a597b12..00000000
--- a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/LiteDbConfigCache.cs
+++ /dev/null
@@ -1,129 +0,0 @@
-using LiteDB;
-
-namespace ZB.MOM.WW.OtOpcUa.Configuration.LocalCache;
-
-///
-/// LiteDB-backed . One file per node (default
-/// config_cache.db), one collection per snapshot. Corruption surfaces as
-/// on construction or read — callers should
-/// delete and re-fetch from the central DB.
-///
-public sealed class LiteDbConfigCache : ILocalConfigCache, IDisposable
-{
- private const string CollectionName = "generations";
-
- // LiteDB's default BsonMapper.Global is a process-wide singleton whose per-type member
- // registration is lazy and NOT thread-safe across concurrently-constructed LiteDatabase
- // instances. When several caches (this one + GenerationSealedCache) initialise in parallel
- // the global mapper races, surfacing as "Member ClusterId not found on BsonMapper" or a
- // bogus "duplicate key _id = 0" (the int auto-id mapping was lost so Insert writes a literal
- // 0 twice). Give each database a private, pre-registered mapper so member
- // resolution happens once, single-threaded, at construction and never touches the global.
- private static BsonMapper BuildMapper()
- {
- var mapper = new BsonMapper();
- mapper.Entity();
- return mapper;
- }
-
- private readonly LiteDatabase _db;
- private readonly ILiteCollection _col;
- // PutAsync is a find-then-insert/update; without serialization, two concurrent puts for the
- // same (ClusterId, GenerationId) can both observe `existing is null` and both Insert,
- // producing duplicate rows. Serialize writes through this semaphore so
- // the read-modify-write block is atomic for a given instance. LiteDB itself only locks the
- // page-level write, not the find-then-insert window.
- private readonly SemaphoreSlim _writeGate = new(initialCount: 1, maxCount: 1);
-
- /// Initializes a new instance of the class.
- /// Path to the LiteDB database file.
- public LiteDbConfigCache(string dbPath)
- {
- // LiteDB can be tolerant of header-only corruption at construction time (it may overwrite
- // the header and "recover"), so we force a write + read probe to fail fast on real corruption.
- try
- {
- _db = new LiteDatabase(new ConnectionString { Filename = dbPath, Upgrade = true }, BuildMapper());
- _col = _db.GetCollection(CollectionName);
- _col.EnsureIndex(s => s.ClusterId);
- _col.EnsureIndex(s => s.GenerationId);
- _ = _col.Count();
- }
- catch (Exception ex) when (ex is LiteException or InvalidDataException or IOException
- or NotSupportedException or UnauthorizedAccessException
- or ArgumentOutOfRangeException or FormatException)
- {
- _db?.Dispose();
- throw new LocalConfigCacheCorruptException(
- $"LiteDB cache at '{dbPath}' is corrupt or unreadable — delete the file and refetch from the central DB.",
- ex);
- }
- }
-
- ///
- public Task GetMostRecentAsync(string clusterId, CancellationToken ct = default)
- {
- ct.ThrowIfCancellationRequested();
- var snapshot = _col
- .Find(s => s.ClusterId == clusterId)
- .OrderByDescending(s => s.GenerationId)
- .FirstOrDefault();
- return Task.FromResult(snapshot);
- }
-
- ///
- public async Task PutAsync(GenerationSnapshot snapshot, CancellationToken ct = default)
- {
- ct.ThrowIfCancellationRequested();
- // Serialize the find-then-insert/update so concurrent callers do not observe a stale
- // `existing is null` and both Insert. LiteDB's per-call lock is not enough — the
- // read and the write are independent calls.
- await _writeGate.WaitAsync(ct).ConfigureAwait(false);
- try
- {
- // upsert by (ClusterId, GenerationId) — replace in place if already cached
- var existing = _col
- .Find(s => s.ClusterId == snapshot.ClusterId && s.GenerationId == snapshot.GenerationId)
- .FirstOrDefault();
-
- if (existing is null)
- _col.Insert(snapshot);
- else
- {
- snapshot.Id = existing.Id;
- _col.Update(snapshot);
- }
- }
- finally
- {
- _writeGate.Release();
- }
- }
-
- ///
- public Task PruneOldGenerationsAsync(string clusterId, int keepLatest = 10, CancellationToken ct = default)
- {
- ct.ThrowIfCancellationRequested();
- var doomed = _col
- .Find(s => s.ClusterId == clusterId)
- .OrderByDescending(s => s.GenerationId)
- .Skip(keepLatest)
- .Select(s => s.Id)
- .ToList();
-
- foreach (var id in doomed)
- _col.Delete(id);
-
- return Task.CompletedTask;
- }
-
- /// Releases all resources used by the cache.
- public void Dispose()
- {
- _writeGate.Dispose();
- _db.Dispose();
- }
-}
-
-public sealed class LocalConfigCacheCorruptException(string message, Exception inner)
- : Exception(message, inner);
diff --git a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/ResilientConfigReader.cs b/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/ResilientConfigReader.cs
deleted file mode 100644
index 2c55b199..00000000
--- a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/ResilientConfigReader.cs
+++ /dev/null
@@ -1,140 +0,0 @@
-using System.Text.RegularExpressions;
-using Microsoft.Extensions.Logging;
-using Polly;
-using Polly.Retry;
-using Polly.Timeout;
-
-namespace ZB.MOM.WW.OtOpcUa.Configuration.LocalCache;
-
-///
-/// Wraps a central-DB fetch function with Phase 6.1 Stream D.2 resilience:
-/// timeout 2 s → retry 3× jittered → fallback to sealed cache. Maintains the
-/// — fresh on central-DB success, stale on cache fallback.
-///
-///
-/// Read-path only per plan. The write path (draft save, publish) bypasses this
-/// wrapper entirely and fails hard on DB outage so inconsistent writes never land.
-///
-/// Fallback is triggered by any exception the fetch raises (central-DB
-/// unreachable, SqlException, timeout). If the sealed cache also fails (no pointer,
-/// corrupt file, etc.), surfaces — caller
-/// must fail the current request (InitializeAsync for a driver, etc.).
-///
-public sealed class ResilientConfigReader
-{
- private readonly GenerationSealedCache _cache;
- private readonly StaleConfigFlag _staleFlag;
- private readonly ResiliencePipeline _pipeline;
- private readonly ILogger _logger;
-
- /// Initializes a resilient config reader with the given cache and options.
- /// The sealed cache for fallback.
- /// The stale config flag to manage.
- /// The logger instance.
- /// The timeout for central fetch (default 2s).
- /// The number of retries (default 3).
- public ResilientConfigReader(
- GenerationSealedCache cache,
- StaleConfigFlag staleFlag,
- ILogger logger,
- TimeSpan? timeout = null,
- int retryCount = 3)
- {
- _cache = cache;
- _staleFlag = staleFlag;
- _logger = logger;
- var builder = new ResiliencePipelineBuilder()
- .AddTimeout(new TimeoutStrategyOptions { Timeout = timeout ?? TimeSpan.FromSeconds(2) });
-
- if (retryCount > 0)
- {
- builder.AddRetry(new RetryStrategyOptions
- {
- MaxRetryAttempts = retryCount,
- BackoffType = DelayBackoffType.Exponential,
- UseJitter = true,
- Delay = TimeSpan.FromMilliseconds(100),
- MaxDelay = TimeSpan.FromSeconds(1),
- // Handle ALL exceptions including OperationCanceledException. A SQL command-level
- // timeout surfaces as TaskCanceledException (derives from OperationCanceledException)
- // when the caller's token is NOT cancelled, and must be retried just like any other
- // transient error. Polly itself checks the cancellation token between retries and
- // stops with OperationCanceledException on genuine caller cancellation regardless of
- // this predicate.
- ShouldHandle = new PredicateBuilder().Handle(),
- });
- }
-
- _pipeline = builder.Build();
- }
-
- ///
- /// Redacts connection-string fragments (Password, User Id, Pwd, etc.)
- /// that a caller's exception message could carry. Conservative regex pass — anything
- /// matching Key=Value with a known credential key gets its value replaced.
- ///
- private static readonly Regex SecretsRegex = new(
- @"(?ix)\b(Password|Pwd|User\s*Id|Uid|AccessToken|Authorization|Api[-_]?Key)\s*=\s*[^;,)\s]*",
- RegexOptions.Compiled);
-
- /// Redacts sensitive credential information from a message.
- /// The message to scrub.
- /// The message with redacted credentials.
- internal static string ScrubSecrets(string? message)
- {
- if (string.IsNullOrEmpty(message)) return message ?? string.Empty;
- // Replace the entire matched fragment (key + value) with a redaction marker so the
- // key name itself doesn't leak — log scrapers grep for "Password=" too.
- return SecretsRegex.Replace(message, "[redacted credential]");
- }
-
- ///
- /// Executes a central fetch through the resilience pipeline. On full failure
- /// (post-retry), reads the sealed cache and extracts the requested shape.
- ///
- /// The type of configuration to read.
- /// The cluster ID to fetch for.
- /// Function to fetch from central DB.
- /// Function to extract the config from a snapshot.
- /// Cancellation token.
- /// The configuration of type T.
- public async ValueTask ReadAsync(
- string clusterId,
- Func> centralFetch,
- Func fromSnapshot,
- CancellationToken cancellationToken)
- {
- ArgumentException.ThrowIfNullOrWhiteSpace(clusterId);
- ArgumentNullException.ThrowIfNull(centralFetch);
- ArgumentNullException.ThrowIfNull(fromSnapshot);
-
- try
- {
- var result = await _pipeline.ExecuteAsync(centralFetch, cancellationToken).ConfigureAwait(false);
- _staleFlag.MarkFresh();
- return result;
- }
- // Catch all exceptions that are NOT genuine caller cancellations. A SQL command-level
- // timeout surfaces as TaskCanceledException (derives from OperationCanceledException)
- // but the caller's token is NOT cancelled — we must fall back to the sealed cache for
- // that case, not propagate. Only rethrow if the caller actually requested cancellation.
- catch (Exception ex) when (ex is not OperationCanceledException || !cancellationToken.IsCancellationRequested)
- {
- // Do NOT pass the raw exception object — it carries the stack
- // and inner-exception chain, and SqlException/wrapping delegates can surface
- // connection-string fragments (Password=…, User Id=…) embedded in messages.
- // Log only the exception type and a scrubbed message so secrets stay out of logs.
- _logger.LogWarning(
- "Central-DB read failed after retries ({ExceptionType}: {SanitizedMessage}); falling back to sealed cache for cluster {ClusterId}",
- ex.GetType().Name,
- ScrubSecrets(ex.Message),
- clusterId);
- // GenerationCacheUnavailableException surfaces intentionally — fails the caller's
- // operation. StaleConfigFlag stays unchanged; the flag only flips when we actually
- // served a cache snapshot.
- var snapshot = await _cache.ReadCurrentAsync(clusterId, cancellationToken).ConfigureAwait(false);
- _staleFlag.MarkStale();
- return fromSnapshot(snapshot);
- }
- }
-}
diff --git a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/StaleConfigFlag.cs b/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/StaleConfigFlag.cs
deleted file mode 100644
index 35e7e237..00000000
--- a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/StaleConfigFlag.cs
+++ /dev/null
@@ -1,20 +0,0 @@
-namespace ZB.MOM.WW.OtOpcUa.Configuration.LocalCache;
-
-///
-/// Thread-safe UsingStaleConfig signal per Phase 6.1 Stream D.3. Flips true whenever
-/// a read falls back to a sealed cache snapshot; flips false on the next successful central-DB
-/// round-trip. Surfaced on /healthz body and on the Admin /hosts page.
-///
-public sealed class StaleConfigFlag
-{
- private int _stale;
-
- /// True when the last config read was served from the sealed cache, not the central DB.
- public bool IsStale => Volatile.Read(ref _stale) != 0;
-
- /// Mark the current config as stale (a read fell back to the cache).
- public void MarkStale() => Volatile.Write(ref _stale, 1);
-
- /// Mark the current config as fresh (a central-DB read succeeded).
- public void MarkFresh() => Volatile.Write(ref _stale, 0);
-}
diff --git a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/Services/ILdapGroupRoleMappingService.cs b/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/Services/ILdapGroupRoleMappingService.cs
index 5f0193dd..1a21d2a5 100644
--- a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/Services/ILdapGroupRoleMappingService.cs
+++ b/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/Services/ILdapGroupRoleMappingService.cs
@@ -10,10 +10,18 @@ namespace ZB.MOM.WW.OtOpcUa.Configuration.Services;
/// Phase 6.2 compliance check on control/data-plane separation).
///
///
-/// Per Phase 6.2 Stream A.2 this service is expected to run behind the Phase 6.1
-/// ResilientConfigReader pipeline (timeout → retry → fallback-to-cache) so a
-/// transient DB outage during sign-in falls back to the sealed snapshot rather than
-/// denying every login.
+///
+/// This service has no local-cache fallback: a DB outage during sign-in denies logins.
+///
+///
+/// The Phase 6.1 ResilientConfigReader pipeline (timeout → retry →
+/// fallback-to-sealed-snapshot) this was once expected to run behind was never wired to
+/// anything and was deleted along with the rest of the dormant LiteDB local cache, which
+/// ZB.MOM.WW.LocalDb supersedes. LocalDb caches the deployed-configuration artifact
+/// for driver-role nodes; it does not currently cover admin-plane reads like this one.
+/// Reviving the fallback means adding an admin-side cache on LocalDb, not restoring the
+/// old pipeline.
+///
///
public interface ILdapGroupRoleMappingService
{
diff --git a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/ZB.MOM.WW.OtOpcUa.Configuration.csproj b/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/ZB.MOM.WW.OtOpcUa.Configuration.csproj
index 4c8b187f..ce45b41c 100644
--- a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/ZB.MOM.WW.OtOpcUa.Configuration.csproj
+++ b/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/ZB.MOM.WW.OtOpcUa.Configuration.csproj
@@ -26,7 +26,6 @@
-
diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/KestrelHttpBinding.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/KestrelHttpBinding.cs
new file mode 100644
index 00000000..dcf343d2
--- /dev/null
+++ b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/KestrelHttpBinding.cs
@@ -0,0 +1,102 @@
+using System.Net;
+using Microsoft.AspNetCore.Server.Kestrel.Core;
+
+namespace ZB.MOM.WW.OtOpcUa.Host.Configuration;
+
+///
+/// One HTTP endpoint the host was asked to serve, parsed out of the configured URL list so it
+/// can be re-bound explicitly.
+///
+/// The host component as written — +, *, a hostname, or an IP.
+/// The port.
+/// True when the URL used the https scheme.
+public sealed record KestrelHttpBinding(string Host, int Port, bool IsSecure)
+{
+ ///
+ /// Parses a semicolon-separated URL list (the ASPNETCORE_URLS / urls format)
+ /// into bindings. Unparseable entries are skipped rather than throwing — a malformed URL
+ /// must not take the process down at startup.
+ ///
+ /// The configured URL list, or null/empty when none is configured.
+ /// The parsed bindings, in the order given; empty when nothing is configured.
+ public static IReadOnlyList Parse(string? urls)
+ {
+ if (string.IsNullOrWhiteSpace(urls))
+ return [];
+
+ var result = new List();
+ foreach (var raw in urls.Split(';', StringSplitOptions.RemoveEmptyEntries | StringSplitOptions.TrimEntries))
+ {
+ // Uri cannot parse the "+"/"*" wildcard hosts Kestrel accepts, so substitute a
+ // placeholder host purely to get scheme/port out, then keep the original host token.
+ var isWildcard = raw.Contains("://+", StringComparison.Ordinal)
+ || raw.Contains("://*", StringComparison.Ordinal);
+ var probe = isWildcard
+ ? raw.Replace("://+", "://placeholder", StringComparison.Ordinal)
+ .Replace("://*", "://placeholder", StringComparison.Ordinal)
+ : raw;
+
+ if (!Uri.TryCreate(probe, UriKind.Absolute, out var uri))
+ continue;
+
+ var host = isWildcard
+ ? raw.Contains("://+", StringComparison.Ordinal) ? "+" : "*"
+ : uri.Host;
+
+ result.Add(new KestrelHttpBinding(host, uri.Port, uri.Scheme == Uri.UriSchemeHttps));
+ }
+
+ return result;
+ }
+
+ ///
+ /// Parses the bare-port list format of ASPNETCORE_HTTP_PORTS / HTTP_PORTS (and
+ /// the HTTPS variants) into wildcard (+, all-interfaces) bindings — the shape
+ /// Kestrel itself gives those variables. Modern .NET base images (aspnet:8.0+) set
+ /// ASPNETCORE_HTTP_PORTS=8080 as the container default instead of
+ /// ASPNETCORE_URLS, so a node that never sets URLS explicitly still has a real
+ /// configured surface here that must be re-bound, not replaced with Kestrel's localhost:5000.
+ ///
+ /// A ;- or ,-separated list of bare ports, or null/empty.
+ /// True to mark the parsed bindings https (the HTTPS_PORTS vars).
+ /// One all-interfaces binding per parseable port; empty when nothing is configured.
+ public static IReadOnlyList ParsePorts(string? ports, bool isSecure)
+ {
+ if (string.IsNullOrWhiteSpace(ports))
+ return [];
+
+ var result = new List();
+ foreach (var token in ports.Split([';', ','], StringSplitOptions.RemoveEmptyEntries | StringSplitOptions.TrimEntries))
+ {
+ // Skip a malformed port rather than throwing — a bad env var must not down startup.
+ if (int.TryParse(token, out var port) && port is > 0 and <= 65535)
+ result.Add(new KestrelHttpBinding("+", port, isSecure));
+ }
+
+ return result;
+ }
+
+ ///
+ /// Applies this binding to Kestrel, choosing the narrowest listen call the host component
+ /// allows.
+ ///
+ ///
+ /// A hostname that is neither a literal IP nor localhost cannot be reliably mapped to
+ /// a local interface, so it falls back to —
+ /// the safe superset. Narrowing it and guessing wrong would silently stop serving.
+ ///
+ /// The Kestrel options being configured.
+ public void Apply(KestrelServerOptions options)
+ {
+ ArgumentNullException.ThrowIfNull(options);
+
+ if (Host is "+" or "*")
+ options.ListenAnyIP(Port);
+ else if (IPAddress.TryParse(Host, out var address))
+ options.Listen(address, Port);
+ else if (string.Equals(Host, "localhost", StringComparison.OrdinalIgnoreCase))
+ options.ListenLocalhost(Port);
+ else
+ options.ListenAnyIP(Port);
+ }
+}
diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbRegistration.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbRegistration.cs
new file mode 100644
index 00000000..c23fa6bc
--- /dev/null
+++ b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbRegistration.cs
@@ -0,0 +1,101 @@
+using Microsoft.Extensions.Configuration;
+using Microsoft.Extensions.DependencyInjection;
+using ZB.MOM.WW.LocalDb;
+using ZB.MOM.WW.LocalDb.Replication;
+using ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache;
+
+namespace ZB.MOM.WW.OtOpcUa.Host.Configuration;
+
+///
+/// Single registration point for the node-local LocalDb subsystem: the embedded SQLite store
+/// that caches the deployed-configuration artifact, plus the optional gRPC replication that
+/// mirrors it to the node's redundant pair peer.
+///
+///
+///
+/// Exists as a named extension rather than inline AddZbLocalDb calls in
+/// Program.cs so the wiring is covered by a DI resolution test — Program.cs is
+/// top-level statements and cannot be exercised directly, which is exactly how a
+/// "registered but never resolvable" defect ships unnoticed. This family has shipped that
+/// defect three times (Secrets 0.2.0, Secrets 0.2.2, ScadaBridge #22).
+///
+///
+/// Driver-role nodes only. Admin-only nodes have no deployed configuration to cache,
+/// and registering here would impose the LocalDb:Path-required-or-no-boot constraint
+/// on them for nothing.
+///
+///
+/// Storage is unconditional; replication is default-OFF. A node with no
+/// PeerAddress and no SyncListenPort is simply a fast local SQLite file —
+/// AddZbLocalDbReplication registers the engine but it stays idle. Replication is
+/// opt-in per pair because there is no production distribution story for the sync
+/// ApiKey yet (the same open question the Secrets KEK has).
+///
+///
+public static class LocalDbRegistration
+{
+ /// Configuration section holding LocalDbOptions.
+ public const string LocalDbSectionPath = "LocalDb";
+
+ /// Configuration section holding ReplicationOptions.
+ public const string ReplicationSectionPath = "LocalDb:Replication";
+
+ ///
+ /// Configuration key carrying the dedicated h2c sync listener's port. Zero or absent means
+ /// no listener is bound and the passive sync endpoint is not mapped.
+ ///
+ public const string SyncListenPortKey = "LocalDb:SyncListenPort";
+
+ ///
+ /// The port the dedicated cleartext-HTTP/2 sync listener should bind, or 0 when the
+ /// node should not listen at all. Defaults to 0 — absent configuration must mean
+ /// "off".
+ ///
+ /// The application configuration.
+ /// The configured sync port, or 0.
+ public static int SyncListenPort(IConfiguration configuration)
+ {
+ ArgumentNullException.ThrowIfNull(configuration);
+ return configuration.GetValue(SyncListenPortKey, defaultValue: 0);
+ }
+
+ ///
+ /// Registers the local database (with the deployment-cache schema applied and registered for
+ /// replication) and the replication engine.
+ ///
+ ///
+ ///
+ /// runs inside AddZbLocalDb's singleton factory,
+ /// before the first caller receives the ILocalDb — so every consumer is guaranteed
+ /// to see the tables created and their capture triggers installed.
+ ///
+ ///
+ /// AddZbLocalDbReplication is called unconditionally rather than behind an
+ /// Enabled flag because the library already models "off" as "no
+ /// PeerAddress": the initiator background service idles and the passive endpoint
+ /// is only reachable if Program.cs maps it, which it does only when
+ /// is set. Adding a second gate on top would give two
+ /// ways to express the same thing and a way for them to disagree.
+ ///
+ ///
+ /// The service collection to add to.
+ /// The application configuration.
+ /// The same instance, for chaining.
+ public static IServiceCollection AddOtOpcUaLocalDb(
+ this IServiceCollection services,
+ IConfiguration configuration)
+ {
+ ArgumentNullException.ThrowIfNull(services);
+ ArgumentNullException.ThrowIfNull(configuration);
+
+ services.AddZbLocalDb(configuration, LocalDbSetup.OnReady);
+ services.AddZbLocalDbReplication(configuration);
+
+ // The consumer-facing seam. WithOtOpcUaRuntimeActors resolves this optionally and threads it
+ // into DriverHostActor; registering the store without this leaves the cache tables present,
+ // replicating, and never written to.
+ services.AddSingleton();
+
+ return services;
+ }
+}
diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbSetup.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbSetup.cs
new file mode 100644
index 00000000..0ce2d3be
--- /dev/null
+++ b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbSetup.cs
@@ -0,0 +1,48 @@
+using ZB.MOM.WW.LocalDb;
+using ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache;
+
+namespace ZB.MOM.WW.OtOpcUa.Host.Configuration;
+
+///
+/// The onReady callback handed to AddZbLocalDb: creates the deployment-cache
+/// tables and opts them into replication.
+///
+///
+///
+/// Public rather than internal only so LocalDbSetupTests can drive the production
+/// callback directly. Initialising a test database from a hand-written copy of this schema
+/// would prove only that the test agrees with itself — the whole value of those tests is
+/// that they exercise this method.
+///
+///
+public static class LocalDbSetup
+{
+ ///
+ /// Initialises the local database: DDL first, then replication registration.
+ ///
+ ///
+ ///
+ /// THE ORDER IS LOAD-BEARING: DDL → RegisterReplicated → writes.
+ /// RegisterReplicated is what installs the three AFTER triggers that capture
+ /// changes into the oplog. Any row written before that call is never captured, so it
+ /// never reaches the peer — silently, and permanently, because nothing ever revisits
+ /// history. Phase 1 writes nothing here; when Phase 2 adds its store-and-forward
+ /// migrator, the migrator must run after both registrations for the same reason.
+ ///
+ ///
+ /// The freshly constructed local database.
+ public static void OnReady(ILocalDb db)
+ {
+ ArgumentNullException.ThrowIfNull(db);
+
+ // CreateConnection() hands back an already-open, pragma-configured connection carrying the
+ // zb_hlc_next() UDF the capture triggers need. Calling Open() on it would throw.
+ using (var connection = db.CreateConnection())
+ {
+ DeploymentCacheSchema.Apply(connection);
+ }
+
+ db.RegisterReplicated(DeploymentCacheSchema.ArtifactsTable);
+ db.RegisterReplicated(DeploymentCacheSchema.PointerTable);
+ }
+}
diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbSyncAuthInterceptor.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbSyncAuthInterceptor.cs
new file mode 100644
index 00000000..a16acd61
--- /dev/null
+++ b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbSyncAuthInterceptor.cs
@@ -0,0 +1,162 @@
+using System.Security.Cryptography;
+using System.Text;
+using Grpc.Core;
+using Grpc.Core.Interceptors;
+using Microsoft.Extensions.Logging;
+using Microsoft.Extensions.Options;
+using ZB.MOM.WW.LocalDb.Replication;
+
+namespace ZB.MOM.WW.OtOpcUa.Host.Configuration;
+
+///
+/// Gates the LocalDb passive sync endpoint. The replication library deliberately leaves inbound
+/// authentication to the host — its LocalDbSyncService verifies nothing — so without this
+/// interceptor anything that can reach a driver node's sync port could stream arbitrary rows
+/// into that node's deployment-artifact cache.
+///
+///
+///
+/// Why that matters here specifically. The cached artifact is what a driver node
+/// boots from when central SQL is unreachable. An attacker able to write it could choose
+/// the configuration a node comes up on during exactly the outage nobody is watching.
+///
+///
+/// Scoped by method path. Only calls under /localdb_sync.v1.LocalDbSync/ are
+/// gated; every other method on the shared gRPC pipeline passes through untouched.
+///
+///
+/// Fail-closed. With no LocalDb:Replication:ApiKey configured, NO sync stream
+/// is accepted, authenticated or not. That is the deliberate choice: the alternative —
+/// treating "no key" as "no auth required" — would silently expose the endpoint on exactly
+/// the default configuration every node ships with. An operator enabling replication must
+/// set the same key on both nodes of the pair, which is already required for the initiator
+/// to dial out (SyncBackgroundService sends Authorization: Bearer <key>).
+/// A key typo therefore stops convergence outright rather than degrading to unauthenticated.
+///
+///
+/// Comparison is over UTF-8 bytes, so
+/// a wrong key cannot be recovered byte-by-byte from response timing. Length differences are
+/// unavoidably observable and are not sensitive.
+///
+///
+public sealed class LocalDbSyncAuthInterceptor : Interceptor
+{
+ private const string ServicePrefix = "/localdb_sync.v1.LocalDbSync/";
+ private const string AuthorizationHeader = "authorization";
+ private const string BearerPrefix = "Bearer ";
+
+ private readonly IOptions _options;
+ private readonly ILogger _logger;
+
+ /// Creates the interceptor.
+ /// Replication options; ApiKey is the expected bearer token.
+ /// Logger for denial diagnostics.
+ public LocalDbSyncAuthInterceptor(
+ IOptions options,
+ ILogger logger)
+ {
+ ArgumentNullException.ThrowIfNull(options);
+ ArgumentNullException.ThrowIfNull(logger);
+
+ _options = options;
+ _logger = logger;
+ }
+
+ ///
+ public override Task UnaryServerHandler(
+ TRequest request,
+ ServerCallContext context,
+ UnaryServerMethod continuation)
+ {
+ Authorize(context);
+ return continuation(request, context);
+ }
+
+ ///
+ public override Task DuplexStreamingServerHandler(
+ IAsyncStreamReader requestStream,
+ IServerStreamWriter responseStream,
+ ServerCallContext context,
+ DuplexStreamingServerMethod continuation)
+ {
+ Authorize(context);
+ return continuation(requestStream, responseStream, context);
+ }
+
+ ///
+ public override Task ClientStreamingServerHandler(
+ IAsyncStreamReader requestStream,
+ ServerCallContext context,
+ ClientStreamingServerMethod continuation)
+ {
+ Authorize(context);
+ return continuation(requestStream, context);
+ }
+
+ ///
+ public override Task ServerStreamingServerHandler(
+ TRequest request,
+ IServerStreamWriter responseStream,
+ ServerCallContext context,
+ ServerStreamingServerMethod continuation)
+ {
+ Authorize(context);
+ return continuation(request, responseStream, context);
+ }
+
+ ///
+ /// Throws with if this
+ /// is a sync call that does not carry the configured bearer token. Non-sync calls return
+ /// immediately.
+ ///
+ private void Authorize(ServerCallContext context)
+ {
+ if (!context.Method.StartsWith(ServicePrefix, StringComparison.Ordinal))
+ return;
+
+ var expected = _options.Value.ApiKey;
+ if (string.IsNullOrEmpty(expected))
+ {
+ _logger.LogWarning(
+ "Rejected a LocalDb sync call to {Method}: no LocalDb:Replication:ApiKey is configured, " +
+ "so the passive sync endpoint is closed. Configure the same key on both nodes of the pair.",
+ context.Method);
+ throw new RpcException(new Status(
+ StatusCode.PermissionDenied,
+ "LocalDb sync is not accepting connections: no API key is configured on this node."));
+ }
+
+ var presented = ExtractBearerToken(context.RequestHeaders);
+ if (presented is null || !FixedTimeEquals(presented, expected))
+ {
+ _logger.LogWarning(
+ "Rejected a LocalDb sync call to {Method}: {Reason}.",
+ context.Method,
+ presented is null ? "no bearer token presented" : "bearer token did not match");
+ throw new RpcException(new Status(
+ StatusCode.PermissionDenied,
+ "LocalDb sync authentication failed."));
+ }
+ }
+
+ private static string? ExtractBearerToken(Metadata headers)
+ {
+ // gRPC lowercases header keys on the wire; compare case-insensitively anyway so a
+ // hand-built Metadata in a test behaves the same as a real request.
+ foreach (var entry in headers)
+ {
+ if (!string.Equals(entry.Key, AuthorizationHeader, StringComparison.OrdinalIgnoreCase))
+ continue;
+
+ var value = entry.Value;
+ if (value is not null && value.StartsWith(BearerPrefix, StringComparison.OrdinalIgnoreCase))
+ return value[BearerPrefix.Length..];
+ }
+
+ return null;
+ }
+
+ private static bool FixedTimeEquals(string presented, string expected)
+ => CryptographicOperations.FixedTimeEquals(
+ Encoding.UTF8.GetBytes(presented), Encoding.UTF8.GetBytes(expected));
+}
diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Host/Health/HealthEndpoints.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Health/HealthEndpoints.cs
index a2abe173..4ad7385e 100644
--- a/src/Server/ZB.MOM.WW.OtOpcUa.Host/Health/HealthEndpoints.cs
+++ b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Health/HealthEndpoints.cs
@@ -1,10 +1,15 @@
using Microsoft.AspNetCore.Routing;
using Microsoft.EntityFrameworkCore;
+using Microsoft.Extensions.Configuration;
using Microsoft.Extensions.DependencyInjection;
+using Microsoft.Extensions.Diagnostics.HealthChecks;
+using Microsoft.Extensions.Options;
using ZB.MOM.WW.Health;
using ZB.MOM.WW.Health.Akka;
using ZB.MOM.WW.Health.EntityFrameworkCore;
+using ZB.MOM.WW.LocalDb.Replication;
using ZB.MOM.WW.OtOpcUa.Configuration;
+using ZB.MOM.WW.OtOpcUa.Host.Configuration;
namespace ZB.MOM.WW.OtOpcUa.Host.Health;
@@ -36,7 +41,20 @@ public static class HealthEndpoints
"admin-leader",
failureStatus: null,
tags: new[] { ZbHealthTags.Active },
- args: "admin");
+ args: "admin")
+ // Registered unconditionally, not driver-gated. AddOtOpcUaHealth takes no role argument
+ // and runs on every node; the check itself resolves ISyncStatus optionally and reports
+ // Healthy when LocalDb is absent (admin-only graphs) or replication is default-OFF, so a
+ // plain node is never degraded by it. A factory registration keeps this no-arg signature
+ // while still reading ISyncStatus + options + the sync port from the container.
+ .Add(new HealthCheckRegistration(
+ "localdb-replication",
+ sp => new LocalDbReplicationHealthCheck(
+ sp.GetService(),
+ sp.GetRequiredService>(),
+ LocalDbRegistration.SyncListenPort(sp.GetRequiredService())),
+ failureStatus: null,
+ tags: new[] { ZbHealthTags.Active }));
return services;
}
diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Host/Health/LocalDbReplicationHealthCheck.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Health/LocalDbReplicationHealthCheck.cs
new file mode 100644
index 00000000..8919d179
--- /dev/null
+++ b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Health/LocalDbReplicationHealthCheck.cs
@@ -0,0 +1,126 @@
+using Microsoft.Extensions.Diagnostics.HealthChecks;
+using Microsoft.Extensions.Options;
+using ZB.MOM.WW.LocalDb.Replication;
+
+namespace ZB.MOM.WW.OtOpcUa.Host.Health;
+
+///
+/// Pure decision function for the LocalDb replication probe, factored out of
+/// so the whole matrix is table-testable without
+/// standing up a replication engine.
+///
+public static class LocalDbReplicationDecision
+{
+ ///
+ /// Maps the resolved replication facts to a health status.
+ ///
+ ///
+ /// True when this node participates in replication at all — it has a peer address to dial, or
+ /// a sync listener bound. False means default-OFF, and default-OFF must never degrade a node.
+ ///
+ /// Whether a sync session is currently running.
+ ///
+ /// The unacked oplog backlog, or when it could not be read. Null is
+ /// deliberately not treated as zero: a failed poll is "unknown", not "caught up".
+ ///
+ ///
+ /// Backlog at or above which a connected pair is judged to be falling behind.
+ ///
+ /// The status plus a human-readable reason.
+ public static (HealthStatus Status, string Description) Evaluate(
+ bool peerConfigured, bool connected, long? oplogBacklog, long degradedThreshold)
+ {
+ if (!peerConfigured)
+ return (HealthStatus.Healthy, "LocalDb replication is not configured on this node (default-OFF).");
+
+ if (!connected)
+ return (HealthStatus.Degraded, "LocalDb replication is configured but no sync session is connected.");
+
+ if (oplogBacklog is null)
+ return (HealthStatus.Degraded, "LocalDb replication is connected but its oplog backlog is unknown (the poll failed).");
+
+ if (oplogBacklog.Value >= degradedThreshold)
+ return (HealthStatus.Degraded,
+ $"LocalDb replication is connected but its oplog backlog ({oplogBacklog.Value}) is at or above the degraded threshold ({degradedThreshold}).");
+
+ return (HealthStatus.Healthy, $"LocalDb replication is connected; oplog backlog {oplogBacklog.Value}.");
+ }
+}
+
+///
+/// Reports the health of this node's LocalDb replication link. Registered unconditionally with
+/// the shared health pipeline; when LocalDb is not present at all (admin-only graphs) it reports
+/// rather than degrading — matching the
+/// ActiveNodeHealthCheck precedent of "not applicable ⇒ Healthy".
+///
+///
+/// A node running LocalDb with replication switched off (no peer, no listener) is also Healthy:
+/// that is the default posture for most of the fleet, and it must not look degraded. Only a node
+/// that is meant to be replicating and is not — or is connected but cannot confirm it is
+/// draining — is surfaced as Degraded. It never reports Unhealthy: a replication problem does not
+/// stop the node serving its address space, so it must not fail a readiness gate.
+///
+public sealed class LocalDbReplicationHealthCheck : IHealthCheck
+{
+ ///
+ /// Backlog at or above which a connected pair is reported Degraded. Well below the library's
+ /// MaxOplogRows default (1,000,000, where a snapshot resync is forced), so the probe
+ /// flags a pair that is falling behind long before the engine itself intervenes.
+ ///
+ public const long DefaultBacklogDegradedThreshold = 100_000;
+
+ private readonly ISyncStatus? _syncStatus;
+ private readonly ReplicationOptions _options;
+ private readonly int _syncListenPort;
+ private readonly long _degradedThreshold;
+
+ /// Creates the health check.
+ ///
+ /// The replication status singleton, or when the replication engine is
+ /// not registered (admin-only nodes) — in which case the check is a Healthy no-op.
+ ///
+ /// The bound replication options (peer address, etc.).
+ ///
+ /// The configured sync listener port; a value greater than zero counts as "replication
+ /// configured" even when this node is the passive side with no peer address.
+ ///
+ /// Backlog degraded threshold; defaults to .
+ public LocalDbReplicationHealthCheck(
+ ISyncStatus? syncStatus,
+ IOptions options,
+ int syncListenPort,
+ long degradedThreshold = DefaultBacklogDegradedThreshold)
+ {
+ ArgumentNullException.ThrowIfNull(options);
+
+ _syncStatus = syncStatus;
+ _options = options.Value;
+ _syncListenPort = syncListenPort;
+ _degradedThreshold = degradedThreshold;
+ }
+
+ ///
+ public Task CheckHealthAsync(
+ HealthCheckContext context, CancellationToken cancellationToken = default)
+ {
+ // No replication engine registered at all → not applicable → Healthy. Admin-only nodes and
+ // any graph that did not call AddOtOpcUaLocalDb land here.
+ if (_syncStatus is null)
+ return Task.FromResult(HealthCheckResult.Healthy(
+ "LocalDb replication is not present on this node."));
+
+ var peerConfigured =
+ !string.IsNullOrWhiteSpace(_options.PeerAddress) || _syncListenPort > 0;
+
+ var (status, description) = LocalDbReplicationDecision.Evaluate(
+ peerConfigured, _syncStatus.Connected, _syncStatus.OplogBacklog, _degradedThreshold);
+
+ var result = status switch
+ {
+ HealthStatus.Healthy => HealthCheckResult.Healthy(description),
+ _ => HealthCheckResult.Degraded(description),
+ };
+
+ return Task.FromResult(result);
+ }
+}
diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Host/Observability/ObservabilityExtensions.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Observability/ObservabilityExtensions.cs
index 2c89fd2f..171d9583 100644
--- a/src/Server/ZB.MOM.WW.OtOpcUa.Host/Observability/ObservabilityExtensions.cs
+++ b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Observability/ObservabilityExtensions.cs
@@ -1,4 +1,5 @@
using Microsoft.Extensions.Configuration;
+using ZB.MOM.WW.LocalDb.Replication;
using ZB.MOM.WW.OtOpcUa.Commons.Observability;
using ZB.MOM.WW.Telemetry;
@@ -27,7 +28,11 @@ public static class ObservabilityExtensions
return services.AddZbTelemetry(o =>
{
o.ServiceName = "otopcua";
- o.Meters = [OtOpcUaTelemetry.MeterName];
+ // o.Meters is a STRICT allowlist — ZbTelemetry adds only the named meters, with no
+ // wildcard. The LocalDb replication engine publishes under its own meter name, so
+ // without this entry its localdb.sync.* / localdb.oplog.depth series are silently absent
+ // from /metrics on driver nodes (the exact omission a ScadaBridge live gate caught).
+ o.Meters = [OtOpcUaTelemetry.MeterName, LocalDbMetrics.MeterName];
o.ActivitySources = [OtOpcUaTelemetry.ActivitySourceName];
if (Enum.TryParse(configuration["OtOpcUa:Telemetry:Exporter"], ignoreCase: true, out var exporter))
o.Exporter = exporter;
diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Host/Program.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Program.cs
index c332916e..b9de794b 100644
--- a/src/Server/ZB.MOM.WW.OtOpcUa.Host/Program.cs
+++ b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Program.cs
@@ -1,7 +1,9 @@
using Akka.Actor;
using Akka.Hosting;
+using Microsoft.AspNetCore.Server.Kestrel.Core;
using Microsoft.Extensions.DependencyInjection.Extensions;
using Serilog;
+using ZB.MOM.WW.LocalDb.Replication;
using Serilog.Events;
using ZB.MOM.WW.OtOpcUa.AdminUI;
using ZB.MOM.WW.OtOpcUa.AdminUI.Clients;
@@ -174,6 +176,20 @@ if (hasDriver)
builder.Configuration,
(_, sp) => GatewayHistorian.CreateAlarmWriter(serverHistorianOptions, sp));
+ // Node-local LocalDb: caches the deployed-configuration artifact so this node can boot from
+ // its last-known-good config when central SQL is unreachable, and optionally replicates that
+ // cache to its redundant pair peer. Driver-role only — admin-only nodes have nothing to cache,
+ // and registering here would impose LocalDb:Path-required-or-no-boot on them for nothing.
+ // Storage ships unconditionally; replication stays inert until LocalDb:Replication:PeerAddress
+ // (initiator) / LocalDb:SyncListenPort (listener) are set. See LocalDbRegistration.
+ builder.Services.AddOtOpcUaLocalDb(builder.Configuration);
+
+ // gRPC server plumbing for the passive sync endpoint mapped below. Registered only under
+ // hasDriver so admin-only nodes expose no sync surface at all. The interceptor is the ONLY
+ // inbound auth on that endpoint — the replication library's LocalDbSyncService verifies
+ // nothing — and it fail-closes when no ApiKey is configured.
+ builder.Services.AddGrpc(o => o.Interceptors.Add());
+
// Config-gated server-side HistoryRead backend. When the ServerHistorian section is enabled this
// overrides the NullHistorianDataSource default from AddOtOpcUaRuntime (last registration wins) with
// a read-only HistorianGateway-backed data source the node manager's HistoryRead overrides
@@ -365,6 +381,92 @@ builder.Services.AddOtOpcUaSecrets(builder.Configuration);
builder.Services.AddOtOpcUaHealth();
builder.Services.AddOtOpcUaObservability(builder.Configuration);
+// ---------------------------------------------------------------------------------------------
+// LocalDb sync listener (default-OFF).
+//
+// DANGER: any explicit Kestrel Listen* call makes Kestrel IGNORE ASPNETCORE_URLS/urls ENTIRELY
+// (it logs "Overriding address(es)"). The host has no ConfigureKestrel today and binds solely via
+// that configuration, so adding a listener naively would silently unbind the AdminUI + deploy API
+// behind Traefik. Everything the host was already asked to serve is therefore re-bound explicitly
+// in the same block.
+//
+// The listener is HTTP/2-ONLY on purpose: the sync client speaks prior-knowledge h2c, which a
+// cleartext Http1AndHttp2 endpoint cannot negotiate (there is no ALPN without TLS). Hence a
+// dedicated port rather than multiplexing onto the main one.
+//
+// When LocalDb:SyncListenPort is 0 (the default) none of this runs and URL binding is untouched.
+// ---------------------------------------------------------------------------------------------
+var syncListenPort = LocalDbRegistration.SyncListenPort(builder.Configuration);
+if (hasDriver && syncListenPort > 0)
+{
+ // Mirror Kestrel's own source precedence: URLS wins; else the HTTP_PORTS/HTTPS_PORTS bare-port
+ // vars; else Kestrel's localhost:5000 default. Missing the HTTP_PORTS leg is not academic — the
+ // aspnet:8.0+ base images set ASPNETCORE_HTTP_PORTS=8080 as the CONTAINER default in place of
+ // ASPNETCORE_URLS, so a driver node that never sets URLS is really serving :8080 on all
+ // interfaces. Falling straight through to localhost:5000 would silently move its health/metrics
+ // surface to loopback (found on the docker-dev live gate).
+ var configuredUrls = builder.Configuration["urls"]
+ ?? builder.Configuration["ASPNETCORE_URLS"];
+ var existingBindings = KestrelHttpBinding.Parse(configuredUrls);
+
+ if (existingBindings.Count == 0)
+ {
+ var httpPorts = builder.Configuration["ASPNETCORE_HTTP_PORTS"] ?? builder.Configuration["HTTP_PORTS"];
+ var httpsPorts = builder.Configuration["ASPNETCORE_HTTPS_PORTS"] ?? builder.Configuration["HTTPS_PORTS"];
+ existingBindings =
+ [
+ .. KestrelHttpBinding.ParsePorts(httpPorts, isSecure: false),
+ .. KestrelHttpBinding.ParsePorts(httpsPorts, isSecure: true),
+ ];
+ if (existingBindings.Count > 0)
+ Log.Information(
+ "LocalDb sync listener enabled; re-binding the HTTP_PORTS surface ({HttpPorts}/{HttpsPorts}) " +
+ "alongside the sync port.", httpPorts, httpsPorts);
+ }
+
+ // A driver-only Windows-service node gets NO URLS and NO *_PORTS (Install-Services.ps1 sets URLS
+ // only under $hasAdmin), so "nothing configured" is a real, supported state — not an error.
+ // Kestrel's own default is localhost:5000; re-state it explicitly, because taking over
+ // configuration means taking over the default too. Health probes hit this port.
+ if (existingBindings.Count == 0)
+ {
+ existingBindings = [new KestrelHttpBinding("localhost", 5000, IsSecure: false)];
+ Log.Information(
+ "LocalDb sync listener enabled with no URLs configured; re-binding Kestrel's default " +
+ "http://localhost:5000 alongside the sync port.");
+ }
+
+ // Re-binding an https endpoint would need its certificate configuration replayed too, which
+ // this host does not model (TLS is terminated at Traefik). Rather than silently serve it
+ // without TLS — or drop it — refuse to take over Kestrel at all and leave the existing
+ // surface exactly as it was. The operator gets a loud, actionable error instead of an
+ // AdminUI that stopped answering.
+ if (existingBindings.Any(b => b.IsSecure))
+ {
+ Log.Error(
+ "LocalDb:SyncListenPort is set to {Port} but this host serves HTTPS endpoint(s) ({Urls}). " +
+ "Binding the sync listener requires re-binding every existing endpoint explicitly, and the " +
+ "HTTPS certificate configuration cannot be replayed safely. The sync listener is DISABLED; " +
+ "terminate TLS upstream (as the docker-dev rig does) or leave replication off on this node.",
+ syncListenPort, configuredUrls);
+ syncListenPort = 0;
+ }
+ else
+ {
+ builder.WebHost.ConfigureKestrel(kestrel =>
+ {
+ foreach (var binding in existingBindings)
+ binding.Apply(kestrel);
+
+ kestrel.ListenAnyIP(syncListenPort, o => o.Protocols = HttpProtocols.Http2);
+ });
+
+ Log.Information(
+ "LocalDb sync listener bound on :{SyncPort} (h2c); re-bound existing endpoint(s) {Urls}.",
+ syncListenPort, configuredUrls ?? "http://localhost:5000 (Kestrel default)");
+ }
+}
+
var app = builder.Build();
// AddZbSerilog registers Serilog as the MEL logging provider but does NOT assign the static
@@ -395,6 +497,15 @@ if (hasAdmin)
app.MapOtOpcUaDeployApi(app.Configuration);
}
+// Passive LocalDb sync endpoint. Gated on the same port check that bound the listener, so a
+// default-OFF or admin-only graph carries no sync surface whatsoever. Mapping it would be safe
+// even unauthenticated — LocalDbSyncAuthInterceptor fail-closes with no ApiKey configured — but
+// not mapping it at all is the stronger default.
+if (hasDriver && syncListenPort > 0)
+{
+ app.MapZbLocalDbSync();
+}
+
app.MapOtOpcUaHealth();
app.MapOtOpcUaMetrics();
diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Host/ZB.MOM.WW.OtOpcUa.Host.csproj b/src/Server/ZB.MOM.WW.OtOpcUa.Host/ZB.MOM.WW.OtOpcUa.Host.csproj
index 636c7953..dd7fbd9a 100644
--- a/src/Server/ZB.MOM.WW.OtOpcUa.Host/ZB.MOM.WW.OtOpcUa.Host.csproj
+++ b/src/Server/ZB.MOM.WW.OtOpcUa.Host/ZB.MOM.WW.OtOpcUa.Host.csproj
@@ -33,6 +33,13 @@
+
+
+
+
diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Host/appsettings.json b/src/Server/ZB.MOM.WW.OtOpcUa.Host/appsettings.json
index a955665b..e7ca5cd6 100644
--- a/src/Server/ZB.MOM.WW.OtOpcUa.Host/appsettings.json
+++ b/src/Server/ZB.MOM.WW.OtOpcUa.Host/appsettings.json
@@ -59,6 +59,16 @@
"Capacity": 1000000,
"DeadLetterRetentionDays": 30
},
+ "LocalDb": {
+ "_comment": "Node-local SQLite store caching the deployed-configuration artifact, so a driver node can boot from its last-known-good config when central SQL is unreachable. Registered on driver-role nodes only (see LocalDbRegistration). Storage is unconditional; replication to the redundant pair peer is default-OFF.",
+ "Path": "./data/otopcua-localdb.db",
+ "_SyncListenPortComment": "Port for the dedicated cleartext-HTTP/2 (h2c) listener serving the passive sync endpoint. 0 = no listener bound and the endpoint is not mapped. Must be a port distinct from the main HTTP listener: prior-knowledge h2c cannot share a cleartext Http1AndHttp2 port.",
+ "SyncListenPort": 0,
+ "Replication": {
+ "_comment": "Empty = passive/off. Set PeerAddress on exactly ONE node of the pair (the stream is bidirectional, so the other side still pushes its own deltas back). ApiKey must be BYTE-IDENTICAL on both nodes — the host interceptor fail-closes, so a typo silently stops convergence rather than degrading to unauthenticated. Supply it via ${secret:...} in production, never committed.",
+ "_MaxBatchSizeComment": "Row count, NOT bytes, against gRPC's 4 MB default message cap. Artifact chunk rows are ~171 KB base64, so keep this at 16 on the pair (16 x 171 KB ~= 2.7 MB worst case)."
+ }
+ },
"Deployment": {
"_comment": "R2-11 (05/CONV-2): deploy-gate TagConfig strictness. Warn (default) = non-blocking warnings logged + appended to the deployment result message; Error = a config with a typo'd enum or unparseable TagConfig is rejected at the draft gate. Running servers are untouched; the gate only sees re-deploys.",
"TagConfigValidationMode": "Warn"
diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/DeploymentCacheSchema.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/DeploymentCacheSchema.cs
new file mode 100644
index 00000000..14d29b9a
--- /dev/null
+++ b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/DeploymentCacheSchema.cs
@@ -0,0 +1,70 @@
+using Microsoft.Data.Sqlite;
+
+namespace ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache;
+
+///
+/// DDL for the node-local deployment-artifact cache: the chunked artifact table plus the
+/// per-cluster current-deployment pointer.
+///
+///
+///
+/// Deliberately depends on nothing but so it can be applied
+/// to any connection — the host's LocalDbSetup.OnReady in production, and a bare
+/// connection in a test — without dragging the DI graph along.
+///
+///
+/// Why the artifact is chunked base64 TEXT and not one BLOB row. Two independent
+/// constraints force it. RegisterReplicated rejects BLOB columns outright; and the
+/// replication engine batches by row count against gRPC's 4 MB default message cap,
+/// so a single multi-megabyte artifact row would simply never be deliverable — at any
+/// batch size. A 128 KiB raw chunk is ≈ 171 KB once base64-encoded, which keeps a
+/// worst-case batch comfortably inside the cap.
+///
+///
+/// Why no autoincrement PKs. Convergence is last-writer-wins keyed on the primary
+/// key. Two nodes independently allocating rowid 7 for different rows would silently
+/// overwrite one another. Every key here is TEXT or a composite of caller-supplied values.
+///
+///
+public static class DeploymentCacheSchema
+{
+ /// Table holding the artifact bytes, split into base64 chunks.
+ public const string ArtifactsTable = "deployment_artifacts";
+
+ /// Table holding one current-deployment pointer per cluster.
+ public const string PointerTable = "deployment_pointer";
+
+ ///
+ /// Creates both cache tables if they do not already exist. Idempotent.
+ ///
+ ///
+ /// An already-open connection. ILocalDb.CreateConnection() hands out open,
+ /// pragma-configured connections — do not call Open() on one.
+ ///
+ public static void Apply(SqliteConnection connection)
+ {
+ ArgumentNullException.ThrowIfNull(connection);
+
+ using var cmd = connection.CreateCommand();
+ cmd.CommandText = """
+ CREATE TABLE IF NOT EXISTS deployment_artifacts (
+ deployment_id TEXT NOT NULL,
+ chunk_index INTEGER NOT NULL,
+ cluster_id TEXT NOT NULL,
+ revision_hash TEXT NOT NULL,
+ chunk_count INTEGER NOT NULL,
+ chunk_base64 TEXT NOT NULL,
+ cached_at_utc TEXT NOT NULL,
+ PRIMARY KEY (deployment_id, chunk_index)
+ );
+ CREATE TABLE IF NOT EXISTS deployment_pointer (
+ cluster_id TEXT NOT NULL PRIMARY KEY,
+ deployment_id TEXT NOT NULL,
+ revision_hash TEXT NOT NULL,
+ artifact_sha256 TEXT NOT NULL,
+ applied_at_utc TEXT NOT NULL
+ );
+ """;
+ cmd.ExecuteNonQuery();
+ }
+}
diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/IDeploymentArtifactCache.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/IDeploymentArtifactCache.cs
new file mode 100644
index 00000000..7d4a1c74
--- /dev/null
+++ b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/IDeploymentArtifactCache.cs
@@ -0,0 +1,68 @@
+namespace ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache;
+
+///
+/// Node-local durable cache of the deployment artifact this node last applied.
+///
+///
+///
+/// Exists so a driver node can boot into its last-known-good address space when the central
+/// configuration database is unreachable. Without it, a control-plane outage that outlives a
+/// node restart leaves that node with no address space at all — the plant loses visibility
+/// for a reason that has nothing to do with the plant.
+///
+///
+public interface IDeploymentArtifactCache
+{
+ ///
+ /// Persists as the cluster's current deployment.
+ ///
+ ///
+ ///
+ /// Idempotent per : re-storing the same deployment
+ /// replaces its chunks rather than accumulating orphans. Retention is bounded to the two
+ /// newest deployments per cluster, so a long-lived node cannot grow its local database
+ /// without limit.
+ ///
+ ///
+ Task StoreAsync(string clusterId, string deploymentId, string revisionHash,
+ byte[] artifact, CancellationToken ct = default);
+
+ ///
+ /// Reads the cached artifact for , or
+ /// when nothing is cached or the cached bytes fail their integrity check.
+ ///
+ ///
+ ///
+ /// A failed integrity check is deliberately indistinguishable from a miss. Booting a
+ /// plant from a truncated address space is strictly worse than booting from none: the
+ /// missing half looks like a deliberate configuration rather than an error.
+ ///
+ ///
+ Task GetCurrentAsync(string clusterId, CancellationToken ct = default);
+
+ ///
+ /// Reads the single cached pointer without knowing the cluster id.
+ ///
+ ///
+ ///
+ /// The boot seam needs this. A node's cluster id is only derivable from an artifact it
+ /// has already loaded, or from the central database — which is precisely what is
+ /// unreachable in the scenario this cache exists to survive. So the cold path reads the
+ /// newest pointer unkeyed.
+ ///
+ ///
+ /// More than one pointer row means the node was re-homed between clusters. The newest
+ /// wins, but the choice is logged as a warning naming both clusters — a node silently
+ /// booting a neighbouring cluster's configuration is exactly the failure that must never
+ /// be quiet.
+ ///
+ ///
+ Task GetCurrentUnkeyedAsync(CancellationToken ct = default);
+}
+
+///
+/// A deployment artifact recovered from the node-local cache, with the metadata needed to decide
+/// whether it is still current once the control plane is reachable again.
+///
+public sealed record CachedDeploymentArtifact(
+ string DeploymentId, string RevisionHash, byte[] Artifact, DateTimeOffset AppliedAtUtc);
diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/LocalDbDeploymentArtifactCache.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/LocalDbDeploymentArtifactCache.cs
new file mode 100644
index 00000000..a29c08e3
--- /dev/null
+++ b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/LocalDbDeploymentArtifactCache.cs
@@ -0,0 +1,324 @@
+using System.Globalization;
+using System.Security.Cryptography;
+using Microsoft.Extensions.Logging;
+using ZB.MOM.WW.LocalDb;
+
+namespace ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache;
+
+///
+/// backed by the replicated LocalDb deployment-cache
+/// tables.
+///
+///
+///
+/// Why the artifact is stored as base64 chunks. See
+/// — replication rejects BLOB columns and batches by row
+/// count against gRPC's message cap, so a whole-artifact row could never be delivered.
+/// Everything in this class that looks like ceremony (chunk indices, a chunk count, a
+/// separate SHA over the raw bytes) exists to make that split safely reversible.
+///
+///
+public sealed class LocalDbDeploymentArtifactCache : IDeploymentArtifactCache
+{
+ ///
+ /// Raw bytes per chunk, before base64 expansion.
+ ///
+ ///
+ /// 128 KiB raw is ≈ 171 KB encoded, which keeps a worst-case replication batch inside gRPC's
+ /// 4 MB default cap with room to spare.
+ ///
+ private const int ChunkSize = 128 * 1024;
+
+ ///
+ /// How many deployments per cluster survive a store.
+ ///
+ ///
+ /// Two, not one: the previous artifact is what a rollback would need, and keeping it costs
+ /// one artifact's worth of disk. Three would only add a generation nobody rolls back to.
+ ///
+ private const int RetainedDeployments = 2;
+
+ private readonly ILocalDb _db;
+ private readonly ILogger _logger;
+
+ public LocalDbDeploymentArtifactCache(ILocalDb db, ILogger logger)
+ {
+ ArgumentNullException.ThrowIfNull(db);
+ ArgumentNullException.ThrowIfNull(logger);
+
+ _db = db;
+ _logger = logger;
+ }
+
+ ///
+ public async Task StoreAsync(string clusterId, string deploymentId, string revisionHash,
+ byte[] artifact, CancellationToken ct = default)
+ {
+ ArgumentException.ThrowIfNullOrWhiteSpace(clusterId);
+ ArgumentException.ThrowIfNullOrWhiteSpace(deploymentId);
+ ArgumentException.ThrowIfNullOrWhiteSpace(revisionHash);
+ ArgumentNullException.ThrowIfNull(artifact);
+
+ var sha = Convert.ToHexString(SHA256.HashData(artifact));
+ var cachedAtUtc = FormatTimestamp(DateTimeOffset.UtcNow);
+ var chunkCount = (artifact.Length + ChunkSize - 1) / ChunkSize;
+
+ // One transaction for the whole store. A pointer that commits without its chunks — or
+ // chunks that commit without the pointer — is a cache that reads back as a corrupt hit
+ // rather than a clean miss, which is the one outcome this type exists to prevent.
+ await using var tx = await _db.BeginTransactionAsync(ct);
+
+ // Delete-then-insert rather than upsert-per-chunk: a re-store with FEWER chunks than last
+ // time would otherwise leave the old tail behind, and those orphans read back as a
+ // chunk-count mismatch forever.
+ await tx.ExecuteAsync(
+ "DELETE FROM deployment_artifacts WHERE deployment_id = @DeploymentId",
+ new { DeploymentId = deploymentId },
+ ct);
+
+ for (var index = 0; index < chunkCount; index++)
+ {
+ var offset = index * ChunkSize;
+ var length = Math.Min(ChunkSize, artifact.Length - offset);
+
+ await tx.ExecuteAsync(
+ """
+ INSERT INTO deployment_artifacts
+ (deployment_id, chunk_index, cluster_id, revision_hash, chunk_count,
+ chunk_base64, cached_at_utc)
+ VALUES
+ (@DeploymentId, @ChunkIndex, @ClusterId, @RevisionHash, @ChunkCount,
+ @ChunkBase64, @CachedAtUtc)
+ """,
+ new
+ {
+ DeploymentId = deploymentId,
+ ChunkIndex = index,
+ ClusterId = clusterId,
+ RevisionHash = revisionHash,
+ ChunkCount = chunkCount,
+ ChunkBase64 = Convert.ToBase64String(artifact, offset, length),
+ CachedAtUtc = cachedAtUtc,
+ },
+ ct);
+ }
+
+ await tx.ExecuteAsync(
+ """
+ INSERT INTO deployment_pointer
+ (cluster_id, deployment_id, revision_hash, artifact_sha256, applied_at_utc)
+ VALUES
+ (@ClusterId, @DeploymentId, @RevisionHash, @Sha, @AppliedAtUtc)
+ ON CONFLICT(cluster_id) DO UPDATE SET
+ deployment_id = excluded.deployment_id,
+ revision_hash = excluded.revision_hash,
+ artifact_sha256 = excluded.artifact_sha256,
+ applied_at_utc = excluded.applied_at_utc
+ """,
+ new
+ {
+ ClusterId = clusterId,
+ DeploymentId = deploymentId,
+ RevisionHash = revisionHash,
+ Sha = sha,
+ AppliedAtUtc = cachedAtUtc,
+ },
+ ct);
+
+ // Prune inside the same transaction so the cache is never briefly unbounded.
+ //
+ // The `deployment_id <> @DeploymentId` clause is load-bearing, not belt-and-braces: it makes
+ // "the deployment the pointer names is always present" a structural invariant instead of a
+ // consequence of clock resolution. Without it, three stores landing inside one timestamp
+ // tick fall through to the `deployment_id DESC` tiebreak — and since real deployment ids are
+ // GUIDs, that ordering is effectively random, so the row just written can lose. The pointer
+ // would then name chunks that no longer exist, which reads back as a cache miss on every
+ // subsequent boot until the next deploy overwrites it. Silent and permanent: exactly the
+ // failure this cache exists to prevent.
+ await tx.ExecuteAsync(
+ $"""
+ DELETE FROM deployment_artifacts
+ WHERE cluster_id = @ClusterId
+ AND deployment_id <> @DeploymentId
+ AND deployment_id NOT IN (
+ SELECT deployment_id
+ FROM deployment_artifacts
+ WHERE cluster_id = @ClusterId
+ GROUP BY deployment_id
+ ORDER BY MAX(cached_at_utc) DESC, deployment_id DESC
+ LIMIT {RetainedDeployments}
+ )
+ """,
+ new { ClusterId = clusterId, DeploymentId = deploymentId },
+ ct);
+
+ await tx.CommitAsync(ct);
+ }
+
+ ///
+ public async Task GetCurrentAsync(
+ string clusterId, CancellationToken ct = default)
+ {
+ ArgumentException.ThrowIfNullOrWhiteSpace(clusterId);
+
+ var pointers = await _db.QueryAsync(
+ """
+ SELECT deployment_id, revision_hash, artifact_sha256, applied_at_utc
+ FROM deployment_pointer
+ WHERE cluster_id = @ClusterId
+ """,
+ ReadPointer,
+ new { ClusterId = clusterId },
+ ct);
+
+ if (pointers.Count == 0)
+ {
+ _logger.LogDebug("No cached deployment pointer for cluster {ClusterId}.", clusterId);
+ return null;
+ }
+
+ return await ReassembleAsync(clusterId, pointers[0], ct);
+ }
+
+ ///
+ public async Task GetCurrentUnkeyedAsync(CancellationToken ct = default)
+ {
+ // applied_at_utc is round-trip ISO-8601 UTC, so ordering it as TEXT is chronological —
+ // that is the reason the format is pinned rather than left to the current culture.
+ var pointers = await _db.QueryAsync(
+ """
+ SELECT cluster_id, deployment_id, revision_hash, artifact_sha256, applied_at_utc
+ FROM deployment_pointer
+ ORDER BY applied_at_utc DESC
+ """,
+ r => (ClusterId: r.GetString(0), Pointer: new PointerRow(
+ DeploymentId: r.GetString(1),
+ RevisionHash: r.GetString(2),
+ ArtifactSha256: r.GetString(3),
+ AppliedAtUtc: r.GetString(4))),
+ parameters: null,
+ ct);
+
+ if (pointers.Count == 0)
+ {
+ _logger.LogDebug("No cached deployment pointer of any cluster.");
+ return null;
+ }
+
+ if (pointers.Count > 1)
+ {
+ // The node was re-homed between clusters. Booting the newest is the only defensible
+ // choice, but it must never be a silent one — a wrong-cluster address space presents as
+ // a plausible configuration, so the operator needs the cluster names to spot it.
+ _logger.LogWarning(
+ "Local deployment cache holds pointers for {PointerCount} clusters ({ClusterIds}); " +
+ "booting the newest ({ChosenClusterId}). This node appears to have been re-homed — " +
+ "clear the stale cache if that is unexpected.",
+ pointers.Count,
+ string.Join(", ", pointers.Select(p => p.ClusterId)),
+ pointers[0].ClusterId);
+ }
+
+ return await ReassembleAsync(pointers[0].ClusterId, pointers[0].Pointer, ct);
+ }
+
+ ///
+ /// Rebuilds the artifact the pointer names, returning unless every
+ /// integrity check passes.
+ ///
+ private async Task ReassembleAsync(
+ string clusterId, PointerRow pointer, CancellationToken ct)
+ {
+ var chunks = await _db.QueryAsync(
+ """
+ SELECT chunk_count, chunk_base64
+ FROM deployment_artifacts
+ WHERE deployment_id = @DeploymentId
+ ORDER BY chunk_index
+ """,
+ r => (ChunkCount: r.GetInt32(0), Base64: r.GetString(1)),
+ new { pointer.DeploymentId },
+ ct);
+
+ // The chunk_count carried on every row is what makes a partial replica detectable. Without
+ // it a missing tail chunk is indistinguishable from a shorter artifact.
+ var expectedChunkCount = chunks.Count > 0 ? chunks[0].ChunkCount : 0;
+
+ if (chunks.Count != expectedChunkCount)
+ {
+ _logger.LogWarning(
+ "Cached deployment {DeploymentId} for cluster {ClusterId} is incomplete: found " +
+ "{FoundChunks} of {ExpectedChunks} chunks. Treating as a cache miss.",
+ pointer.DeploymentId, clusterId, chunks.Count, expectedChunkCount);
+ return null;
+ }
+
+ byte[] artifact;
+ try
+ {
+ artifact = Decode(chunks.Select(c => c.Base64));
+ }
+ catch (FormatException ex)
+ {
+ _logger.LogWarning(
+ ex,
+ "Cached deployment {DeploymentId} for cluster {ClusterId} has a chunk that is not " +
+ "valid base64. Treating as a cache miss.",
+ pointer.DeploymentId, clusterId);
+ return null;
+ }
+
+ var actualSha = Convert.ToHexString(SHA256.HashData(artifact));
+ if (!string.Equals(actualSha, pointer.ArtifactSha256, StringComparison.OrdinalIgnoreCase))
+ {
+ _logger.LogWarning(
+ "Cached deployment {DeploymentId} for cluster {ClusterId} failed its SHA-256 check " +
+ "(expected {ExpectedSha}, computed {ActualSha}). Treating as a cache miss.",
+ pointer.DeploymentId, clusterId, pointer.ArtifactSha256, actualSha);
+ return null;
+ }
+
+ if (!DateTimeOffset.TryParse(pointer.AppliedAtUtc, CultureInfo.InvariantCulture,
+ DateTimeStyles.RoundtripKind, out var appliedAtUtc))
+ {
+ _logger.LogWarning(
+ "Cached deployment {DeploymentId} for cluster {ClusterId} has an unparseable " +
+ "applied_at_utc ({AppliedAtUtc}). Treating as a cache miss.",
+ pointer.DeploymentId, clusterId, pointer.AppliedAtUtc);
+ return null;
+ }
+
+ return new CachedDeploymentArtifact(
+ pointer.DeploymentId, pointer.RevisionHash, artifact, appliedAtUtc);
+ }
+
+ private static byte[] Decode(IEnumerable chunksInOrder)
+ {
+ using var buffer = new MemoryStream();
+
+ foreach (var chunk in chunksInOrder)
+ {
+ var bytes = Convert.FromBase64String(chunk);
+ buffer.Write(bytes, 0, bytes.Length);
+ }
+
+ return buffer.ToArray();
+ }
+
+ private static PointerRow ReadPointer(Microsoft.Data.Sqlite.SqliteDataReader reader)
+ => new(
+ DeploymentId: reader.GetString(0),
+ RevisionHash: reader.GetString(1),
+ ArtifactSha256: reader.GetString(2),
+ AppliedAtUtc: reader.GetString(3));
+
+ ///
+ /// Round-trip ("O") UTC, so lexicographic ordering of the stored TEXT is chronological
+ /// ordering — the retention query sorts on it directly.
+ ///
+ private static string FormatTimestamp(DateTimeOffset value)
+ => value.ToUniversalTime().ToString("O", CultureInfo.InvariantCulture);
+
+ private sealed record PointerRow(
+ string DeploymentId, string RevisionHash, string ArtifactSha256, string AppliedAtUtc);
+}
diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Drivers/DriverHostActor.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Drivers/DriverHostActor.cs
index 4a87e53c..e3b49eba 100644
--- a/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Drivers/DriverHostActor.cs
+++ b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Drivers/DriverHostActor.cs
@@ -23,6 +23,7 @@ using ZB.MOM.WW.OtOpcUa.Core.ScriptedAlarms;
using ZB.MOM.WW.OtOpcUa.Core.Scripting;
using ZB.MOM.WW.OtOpcUa.Core.VirtualTags;
using ZB.MOM.WW.OtOpcUa.OpcUaServer;
+using ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache;
using ZB.MOM.WW.OtOpcUa.Runtime.ScriptedAlarms;
using ZB.MOM.WW.OtOpcUa.Runtime.VirtualTags;
using CommonsNodeId = ZB.MOM.WW.OtOpcUa.Commons.Types.NodeId;
@@ -57,6 +58,31 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers
/// Publishing interval handed to each driver's SubscribeBulk pass after an apply.
private static readonly TimeSpan SubscriptionPublishingInterval = TimeSpan.FromSeconds(1);
+ ///
+ /// Cache key used when an artifact carries no cluster scoping (single-cluster or unscoped
+ /// deployments). A literal rather than an empty string so the row is visibly deliberate when
+ /// someone reads the table during an incident.
+ ///
+ private const string SingleClusterCacheKey = "__single";
+
+ ///
+ /// Node-local cache of applied deployment artifacts, or null when this node has none
+ /// (admin-only graphs, and tests that do not exercise it).
+ ///
+ private readonly IDeploymentArtifactCache? _deploymentArtifactCache;
+
+ ///
+ /// True once this node has booted a cached artifact because central SQL was unreachable.
+ ///
+ ///
+ /// Surfaced on because a node running from cache looks
+ /// completely healthy from the outside — it serves a full address space with live values.
+ /// The difference is that its configuration is frozen at whatever the cache held, and no
+ /// deployment can reach it. Without an explicit signal that is invisible until someone
+ /// wonders why a deploy "succeeded" everywhere but did not take effect here.
+ ///
+ private bool _isRunningFromCache;
+
private readonly IDbContextFactory _dbFactory;
private readonly CommonsNodeId _localNode;
private readonly IActorRef? _coordinatorOverride;
@@ -339,11 +365,18 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers
ScriptRootLogger? scriptRootLogger = null,
IActorRef? scriptedAlarmHostOverride = null,
IDriverCapabilityInvokerFactory? invokerFactory = null,
- Func? driverMemberCountProvider = null) =>
+ Func? driverMemberCountProvider = null,
+ IDeploymentArtifactCache? deploymentArtifactCache = null) =>
+ // WARNING: this forwarding list is POSITIONAL, and Props.Create compiles it into an
+ // expression tree. Six IActorRef? parameters and several interface-typed ones mean a
+ // mis-ordered argument is usually type-compatible and therefore compiles clean, then binds
+ // the wrong dependency at runtime. New parameters go LAST in all three places — this
+ // signature, the constructor's, and this call — and nothing else moves.
Akka.Actor.Props.Create(() => new DriverHostActor(
dbFactory, localNode, coordinator, driverFactory, localRoles, dependencyMux, opcUaPublishActor,
healthPublisher, virtualTagEvaluator, historyWriter, virtualTagHostOverride,
- loggerFactory, scriptRootLogger, scriptedAlarmHostOverride, invokerFactory, driverMemberCountProvider));
+ loggerFactory, scriptRootLogger, scriptedAlarmHostOverride, invokerFactory, driverMemberCountProvider,
+ deploymentArtifactCache));
/// Initializes a new DriverHostActor with the specified dependencies.
/// Database context factory for configuration database access.
@@ -372,6 +405,10 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers
/// Test seam (archreview 03/S4): overrides the count of Up
/// driver-role cluster members the Primary gate reads while the role is unknown. When null the
/// default reads Cluster.Get(Context.System).State.Members (0 on a non-cluster ActorRefProvider).
+ /// Optional node-local cache of applied deployment artifacts.
+ /// When supplied, each successful apply stores its artifact so the node can boot from its
+ /// last-known-good configuration while central SQL is unreachable. Null on admin-only nodes and in
+ /// tests that do not exercise the cache — caching is then simply skipped.
public DriverHostActor(
IDbContextFactory dbFactory,
CommonsNodeId localNode,
@@ -388,8 +425,10 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers
ScriptRootLogger? scriptRootLogger = null,
IActorRef? scriptedAlarmHostOverride = null,
IDriverCapabilityInvokerFactory? invokerFactory = null,
- Func? driverMemberCountProvider = null)
+ Func? driverMemberCountProvider = null,
+ IDeploymentArtifactCache? deploymentArtifactCache = null)
{
+ _deploymentArtifactCache = deploymentArtifactCache;
_dbFactory = dbFactory;
_localNode = localNode;
_coordinatorOverride = coordinator;
@@ -541,7 +580,7 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers
// space were lost on restart. Re-spawn + re-materialise + re-subscribe from the
// applied deployment so a restarted/rebuilt node restores its served state instead
// of waiting for a config change (whose identical-config revision would no-op).
- RestoreApplied(new DeploymentId(latest.DeploymentId));
+ RestoreApplied(new DeploymentId(latest.DeploymentId), revision);
break;
case NodeDeploymentStatus.Applying:
_log.Warning("DriverHost {Node}: found orphan Applying row for deployment {Id}; replaying",
@@ -560,10 +599,122 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers
catch (Exception ex)
{
_log.Warning(ex, "DriverHost {Node}: ConfigDb unreachable on bootstrap; entering Stale", _localNode);
- Become(Stale);
+
+ // Central is unreachable, so try the node-local cache before giving up. On a hit this
+ // node serves its last-known-good configuration through the outage instead of coming up
+ // with an empty address space. On a miss, behaviour is exactly what it always was.
+ if (!TryBootFromCache())
+ Become(Stale);
}
}
+ ///
+ /// Last-resort boot path: apply the artifact cached by a previous successful deploy (or
+ /// replicated from this node's pair peer) when central SQL cannot be reached.
+ ///
+ ///
+ /// when a cached artifact was applied and the actor is now Steady;
+ /// when the caller should fall through to Stale.
+ ///
+ ///
+ ///
+ /// The read is unkeyed. The cache is keyed by ClusterId so a pair shares one
+ /// entry, but ClusterId is only derivable from an artifact you already hold or from the
+ /// central DB — and at this seam we have neither. So the newest pointer row wins, which
+ /// is correct in every real topology because a node belongs to one cluster and its peer
+ /// replicates that same cluster's row. A node re-homed between clusters is the only
+ /// ambiguous case, and the cache logs a warning naming both.
+ ///
+ ///
+ /// This does not make the node current._currentRevision is set from the
+ /// cached artifact, so a subsequent dispatch of that same revision correctly no-ops,
+ /// while any NEW revision still requires central — the cache holds the past, not the
+ /// future. Dispatch handling is unchanged.
+ ///
+ ///
+ /// Never throws: a fault here must degrade to Stale, which is exactly where the node
+ /// would have been without a cache at all.
+ ///
+ ///
+ private bool TryBootFromCache()
+ {
+ if (_deploymentArtifactCache is null)
+ return false;
+
+ try
+ {
+ var cached = _deploymentArtifactCache.GetCurrentUnkeyedAsync().GetAwaiter().GetResult();
+ if (cached is null)
+ {
+ _log.Info(
+ "DriverHost {Node}: no cached deployment available; entering Stale with no configuration.",
+ _localNode);
+ return false;
+ }
+
+ var deploymentId = DeploymentId.Parse(cached.DeploymentId);
+ var revision = RevisionHash.Parse(cached.RevisionHash);
+
+ // Steady, not Stale: this node is serving a real configuration. The retry-db timer that
+ // Stale would have started does not run here, so recovery rides on the next dispatch —
+ // matching how a normally-booted node behaves.
+ _currentRevision = revision;
+ Become(Steady);
+
+ ApplyCachedArtifact(deploymentId, cached.Artifact);
+
+ _isRunningFromCache = true;
+ _log.Warning(
+ "DriverHost {Node}: RUNNING FROM CACHE — central ConfigDb is unreachable, so this node " +
+ "booted deployment {Id} (rev {Rev}) cached at {CachedAtUtc:o} from its local database. " +
+ "Configuration changes cannot be applied until the ConfigDb is reachable again.",
+ _localNode, deploymentId, revision, cached.AppliedAtUtc);
+
+ return true;
+ }
+ catch (Exception ex)
+ {
+ _log.Error(ex,
+ "DriverHost {Node}: failed to boot from the local deployment cache; falling back to Stale.",
+ _localNode);
+ return false;
+ }
+ }
+
+ ///
+ /// Applies an artifact already in hand — no ConfigDb read, no ACK.
+ ///
+ ///
+ /// Mirrors , but sources the artifact from the cache rather than
+ /// re-reading it from a database that is by definition unreachable here. It also skips
+ /// UpsertNodeDeploymentState and SendAck for the same reason: both write to or
+ /// depend on central.
+ ///
+ private void ApplyCachedArtifact(DeploymentId deploymentId, byte[] blob)
+ {
+ var correlation = CorrelationId.NewId();
+
+ var specs = DeploymentArtifact.ParseDriverInstances(blob, _localNode.Value);
+ var snapshots = _children.ToDictionary(
+ kv => kv.Key,
+ kv => new DriverChildSnapshot(kv.Value.DriverType, kv.Value.LastConfigJson, kv.Value.ResilienceConfig),
+ StringComparer.Ordinal);
+ var plan = DriverSpawnPlanner.Compute(snapshots, specs);
+
+ foreach (var id in plan.ToStop) StopChild(id);
+ foreach (var spec in plan.ToApplyDelta) ApplyChildDelta(spec);
+ foreach (var spec in plan.ToSpawn) SpawnChild(spec);
+
+ // Hand the cached blob to the rebuild so it materialises the client-facing address space from
+ // it directly. Passing only the deploymentId would drive a ConfigDb read — unreachable here by
+ // definition — and the rebuild would no-op, leaving OPC UA clients browsing an empty server
+ // while the drivers below poll happily. (Found on the docker-dev live gate.)
+ _opcUaPublishActor?.Tell(
+ new ZB.MOM.WW.OtOpcUa.Runtime.OpcUa.OpcUaPublishActor.RebuildAddressSpace(correlation, deploymentId, blob));
+
+ PushDesiredSubscriptionsFromArtifact(deploymentId, blob);
+ }
+
private void Steady()
{
Receive(HandleDispatchFromSteady);
@@ -1393,7 +1544,8 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers
NodeId: _localNode,
CurrentRevision: _currentRevision,
Drivers: drivers,
- AsOfUtc: DateTime.UtcNow);
+ AsOfUtc: DateTime.UtcNow,
+ RunningFromCache: _isRunningFromCache);
Sender.Tell(snapshot);
}
@@ -1426,7 +1578,7 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers
try
{
- ReconcileDrivers(deploymentId);
+ var appliedBlob = ReconcileDrivers(deploymentId);
_currentRevision = revision;
UpsertNodeDeploymentState(deploymentId, NodeDeploymentStatus.Applied, failureReason: null);
SendAck(deploymentId, ApplyAckOutcome.Applied, failureReason: null, correlation);
@@ -1437,6 +1589,12 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers
// SubscribeBulk pass: hand each driver its desired tag references so live values flow into
// the just-rebuilt address space instead of staying BadWaitingForInitialData.
PushDesiredSubscriptions(deploymentId);
+ CacheAppliedArtifact(deploymentId, revision, appliedBlob);
+ // Reaching here means the artifact came from central, so the node is no longer serving a
+ // cache-sourced configuration. Note this only clears on a real apply: a dispatch of the
+ // revision already booted from cache short-circuits in HandleDispatchFromSteady without
+ // touching the ConfigDb, and the flag correctly stays set.
+ _isRunningFromCache = false;
OtOpcUaTelemetry.DeploymentApplied.Add(1, new KeyValuePair("outcome", "ack"));
_log.Info("DriverHost {Node}: applied deployment {Id} (rev {Rev}, children={Count})",
_localNode, deploymentId, revision, _children.Count);
@@ -1464,7 +1622,20 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers
/// configured can't materialise any of the requested
/// types, this is effectively a no-op.
///
- private void ReconcileDrivers(DeploymentId deploymentId)
+ ///
+ /// The artifact blob that was reconciled, or when it could not be
+ /// loaded at all.
+ ///
+ ///
+ /// Returning the blob rather than swallowing it is load-bearing for the artifact cache. This
+ /// method catches its own DB failures, logs a warning and returns WITHOUT rethrowing — so
+ /// proceeds to its success path, ACKs Applied and logs success
+ /// having spawned zero drivers. A cache write that trusted "the apply succeeded" would then
+ /// persist an empty artifact as this node's last-known-good configuration, and the node would
+ /// later boot from it into an empty address space. The caller distinguishes the two cases by
+ /// this return value.
+ ///
+ private byte[]? ReconcileDrivers(DeploymentId deploymentId)
{
byte[] blob;
try
@@ -1479,7 +1650,7 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers
{
_log.Warning(ex, "DriverHost {Node}: failed to load artifact for {Id}; skipping reconcile",
_localNode, deploymentId);
- return;
+ return null;
}
var specs = DeploymentArtifact.ParseDriverInstances(blob, _localNode.Value);
@@ -1500,6 +1671,74 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers
// sequenced AFTER ReinitializeAsync rebuilds DependencyRefs — a synchronous re-register HERE would
// re-read the STALE ref set (ApplyDelta runs asynchronously on the child), so it is deliberately NOT
// done inline.
+
+ return blob;
+ }
+
+ ///
+ /// Store a successfully applied artifact in the node-local cache, so this node can boot from
+ /// it while central SQL is unreachable.
+ ///
+ ///
+ ///
+ /// Never throws. By the time this runs the deployment is already recorded Applied
+ /// in central SQL and an Applied ACK has been sent to the coordinator. An exception
+ /// escaping here would unwind into 's catch and send a second,
+ /// contradictory Failed ACK for a deployment the fleet already believes is live.
+ ///
+ ///
+ /// An empty or unloadable blob is skipped, not cached. See
+ /// : a DB failure there degrades silently, so "the apply
+ /// succeeded" does not imply "a real configuration was applied". Caching an empty
+ /// artifact would make it this node's last-known-good and boot it into an empty address
+ /// space during the next outage — strictly worse than having no cache at all.
+ ///
+ ///
+ /// Runs synchronously on the actor thread. That matches every other DB call on this path
+ /// (all of which already block), and the alternative — piping the result back as a
+ /// self-message — would need a handler registered in Steady, Applying AND Stale or it
+ /// dead-letters after the Become(Steady) in the enclosing finally.
+ ///
+ ///
+ private void CacheAppliedArtifact(DeploymentId deploymentId, RevisionHash revision, byte[]? blob)
+ {
+ if (_deploymentArtifactCache is null)
+ return;
+
+ if (blob is null || blob.Length == 0)
+ {
+ _log.Warning(
+ "DriverHost {Node}: not caching deployment {Id} — its artifact was empty or could not " +
+ "be loaded, so it is not a usable last-known-good configuration.",
+ _localNode, deploymentId);
+ return;
+ }
+
+ try
+ {
+ // The node's ClusterId is carried inside the artifact (ClusterNode rows), not in config
+ // or on IClusterRoleInfo. A single-cluster or unscoped artifact has none, so it shares
+ // one sentinel key — the pair still converges on a single pointer row either way.
+ var scope = DeploymentArtifact.ResolveClusterScope(blob, _localNode.Value);
+ var clusterId = scope.Mode == ClusterFilterMode.ScopeTo && !string.IsNullOrWhiteSpace(scope.ClusterId)
+ ? scope.ClusterId
+ : SingleClusterCacheKey;
+
+ _deploymentArtifactCache
+ .StoreAsync(clusterId, deploymentId.ToString(), revision.Value, blob)
+ .GetAwaiter()
+ .GetResult();
+
+ _log.Debug("DriverHost {Node}: cached deployment {Id} (rev {Rev}, cluster {ClusterId}, {Bytes} bytes)",
+ _localNode, deploymentId, revision, clusterId, blob.Length);
+ }
+ catch (Exception ex)
+ {
+ _log.Error(ex,
+ "DriverHost {Node}: failed to cache applied deployment {Id}. The apply itself succeeded " +
+ "and is unaffected; this node simply has no local fallback for that revision.",
+ _localNode, deploymentId);
+ }
}
///
@@ -1511,14 +1750,24 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers
/// drivers, rebuilds the address space from the applied artifact, and re-pushes SubscribeBulk.
/// No re-ack: the deployment is already Applied.
///
- private void RestoreApplied(DeploymentId deploymentId)
+ private void RestoreApplied(DeploymentId deploymentId, RevisionHash? revision)
{
var correlation = CorrelationId.NewId();
try
{
- ReconcileDrivers(deploymentId);
- _opcUaPublishActor?.Tell(new ZB.MOM.WW.OtOpcUa.Runtime.OpcUa.OpcUaPublishActor.RebuildAddressSpace(correlation, deploymentId));
+ var appliedBlob = ReconcileDrivers(deploymentId);
+ _opcUaPublishActor?.Tell(new ZB.MOM.WW.OtOpcUa.Runtime.OpcUa.OpcUaPublishActor.RebuildAddressSpace(correlation, deploymentId, appliedBlob));
PushDesiredSubscriptions(deploymentId);
+ // Populate the node-local cache from the artifact this restored node is now serving. The
+ // cache invariant is "holds what the node currently serves"; without this only a FRESH
+ // apply (ApplyAndAck) ever wrote it, so a node whose cache was lost — a wiped/fresh volume,
+ // a disk failure — would recover its served state here yet stay cache-less, unable to
+ // boot-from-cache on the NEXT central outage until some future new deploy happened to land.
+ // Replication does not heal that gap either: a peer's already-acked rows are pruned from its
+ // oplog, so a fully-wiped node is never back-filled. Re-caching on restore is what closes
+ // it. (Found on the docker-dev live gate.)
+ if (revision is { } rev)
+ CacheAppliedArtifact(deploymentId, rev, appliedBlob);
_log.Info("DriverHost {Node}: restored served state for applied deployment {Id} on bootstrap", _localNode, deploymentId);
}
catch (Exception ex)
@@ -1552,6 +1801,19 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers
return;
}
+ PushDesiredSubscriptionsFromArtifact(deploymentId, blob);
+ }
+
+ ///
+ /// The artifact-processing half of , split out so the
+ /// boot-from-cache path can drive it with bytes already in hand.
+ ///
+ ///
+ /// Separated because the cache path runs precisely when the ConfigDb read above cannot
+ /// succeed — re-reading the artifact there would fail by definition.
+ ///
+ private void PushDesiredSubscriptionsFromArtifact(DeploymentId deploymentId, byte[] blob)
+ {
AddressSpaceComposition composition;
try
{
diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/OpcUa/OpcUaPublishActor.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/OpcUa/OpcUaPublishActor.cs
index c8d9d7d3..f19127ab 100644
--- a/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/OpcUa/OpcUaPublishActor.cs
+++ b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/OpcUa/OpcUaPublishActor.cs
@@ -73,7 +73,17 @@ public sealed class OpcUaPublishActor : ReceiveActor, IWithTimers
/// applied config + the SubscribeBulk pass. It is null only for legacy/dev callers, which
/// fall back to the latest sealed deployment (lags a not-yet-sealed apply by one revision).
///
- public sealed record RebuildAddressSpace(CorrelationId Correlation, DeploymentId? DeploymentId = null);
+ /// Correlation id for tracing this rebuild.
+ /// The applied deployment whose artifact to materialise, or null for the latest-sealed fallback.
+ ///
+ /// The artifact bytes already in hand, used INSTEAD of loading them from the ConfigDb. This is
+ /// what makes boot-from-cache actually serve its address space: on a central-SQL outage the
+ /// host has the cached blob but alone would drive a ConfigDb read
+ /// that cannot succeed, leaving the rebuild a no-op and clients browsing an empty server. Null
+ /// on the normal path, where loading from the ConfigDb by id is correct.
+ ///
+ public sealed record RebuildAddressSpace(
+ CorrelationId Correlation, DeploymentId? DeploymentId = null, byte[]? Artifact = null);
/// Inject driver-discovered nodes (FixedTree) under an equipment at runtime (post-connect).
/// The OPC UA NodeId of the equipment root folder to inject the
@@ -345,13 +355,15 @@ public sealed class OpcUaPublishActor : ReceiveActor, IWithTimers
try
{
- // Prefer the artifact of the deployment the host just applied — at apply time it is not
- // yet Sealed, so LoadLatestArtifact would return the PREVIOUS revision and materialise a
- // stale composition (variables that don't match the SubscribeBulk refs). Fall back to
- // latest-sealed only for legacy callers that don't carry a DeploymentId.
- var artifact = msg.DeploymentId is { } depId
- ? LoadArtifact(depId)
- : LoadLatestArtifact();
+ // An in-hand artifact (boot-from-cache) wins: the host already has the cached bytes, and
+ // the ConfigDb read the DeploymentId path would do is exactly what is unreachable during
+ // the outage this exists to survive. Otherwise prefer the artifact of the deployment the
+ // host just applied — at apply time it is not yet Sealed, so LoadLatestArtifact would
+ // return the PREVIOUS revision and materialise a stale composition (variables that don't
+ // match the SubscribeBulk refs). Fall back to latest-sealed only for legacy callers that
+ // carry neither.
+ var artifact = msg.Artifact
+ ?? (msg.DeploymentId is { } depId ? LoadArtifact(depId) : LoadLatestArtifact());
var composition = _localNode is { } ln
? DeploymentArtifact.ParseComposition(artifact, ln.Value,
inconsistency => _log.Warning("OpcUaPublish {Node}: cross-cluster binding — {Message}", ln, inconsistency))
diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/ServiceCollectionExtensions.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/ServiceCollectionExtensions.cs
index d3d7a4ec..0cddd870 100644
--- a/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/ServiceCollectionExtensions.cs
+++ b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/ServiceCollectionExtensions.cs
@@ -15,6 +15,7 @@ using ZB.MOM.WW.OtOpcUa.Core.AlarmHistorian;
using ZB.MOM.WW.OtOpcUa.Core.Scripting;
using ZB.MOM.WW.OtOpcUa.Core.VirtualTags;
using ZB.MOM.WW.OtOpcUa.OpcUaServer;
+using ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache;
using ZB.MOM.WW.OtOpcUa.Runtime.Drivers;
using ZB.MOM.WW.OtOpcUa.Runtime.Health;
using ZB.MOM.WW.OtOpcUa.Runtime.Historian;
@@ -221,6 +222,11 @@ public static class ServiceCollectionExtensions
var serviceLevel = resolver.GetService() ?? NullServiceLevelPublisher.Instance;
var loggerFactory = resolver.GetService() ?? NullLoggerFactory.Instance;
var healthPublisher = resolver.GetService() ?? NullDriverHealthPublisher.Instance;
+ // Node-local deployment-artifact cache. Registered by the Host's AddOtOpcUaLocalDb on
+ // driver-role nodes only; deliberately left null elsewhere (admin-only graphs, test
+ // harnesses) rather than given a null-object, so DriverHostActor skips caching outright
+ // instead of pretending to cache into a sink that drops everything.
+ var deploymentArtifactCache = resolver.GetService();
// Root script logger backs the ScriptedAlarm host's engine + script logging. Registered in
// Host DI inside the hasDriver block; may be absent in some role configs / test harnesses,
// in which case the DriverHostActor gracefully skips spawning the ScriptedAlarm host.
@@ -338,7 +344,8 @@ public static class ServiceCollectionExtensions
historyWriter: historyWriter,
loggerFactory: loggerFactory,
scriptRootLogger: scriptRootLogger,
- invokerFactory: invokerFactory),
+ invokerFactory: invokerFactory,
+ deploymentArtifactCache: deploymentArtifactCache),
DriverHostActorName);
registry.Register(driverHost);
diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/ZB.MOM.WW.OtOpcUa.Runtime.csproj b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/ZB.MOM.WW.OtOpcUa.Runtime.csproj
index 98eb9cce..dc524c84 100644
--- a/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/ZB.MOM.WW.OtOpcUa.Runtime.csproj
+++ b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/ZB.MOM.WW.OtOpcUa.Runtime.csproj
@@ -8,6 +8,9 @@
+
+
diff --git a/tests/Core/ZB.MOM.WW.OtOpcUa.Configuration.Tests/GenerationSealedCacheTests.cs b/tests/Core/ZB.MOM.WW.OtOpcUa.Configuration.Tests/GenerationSealedCacheTests.cs
deleted file mode 100644
index d7ed3121..00000000
--- a/tests/Core/ZB.MOM.WW.OtOpcUa.Configuration.Tests/GenerationSealedCacheTests.cs
+++ /dev/null
@@ -1,168 +0,0 @@
-using Shouldly;
-using Xunit;
-using ZB.MOM.WW.OtOpcUa.Configuration.LocalCache;
-
-namespace ZB.MOM.WW.OtOpcUa.Configuration.Tests;
-
-[Trait("Category", "Unit")]
-public sealed class GenerationSealedCacheTests : IDisposable
-{
- private readonly string _root = Path.Combine(Path.GetTempPath(), $"otopcua-sealed-{Guid.NewGuid():N}");
-
- /// Cleans up temporary directory after test execution.
- public void Dispose()
- {
- try
- {
- if (!Directory.Exists(_root)) return;
- // Remove ReadOnly attribute first so Directory.Delete can clean sealed files.
- foreach (var f in Directory.EnumerateFiles(_root, "*", SearchOption.AllDirectories))
- File.SetAttributes(f, FileAttributes.Normal);
- Directory.Delete(_root, recursive: true);
- }
- catch { /* best-effort cleanup */ }
- }
-
- private GenerationSnapshot MakeSnapshot(string clusterId, long generationId, string payload = "{\"sample\":true}") =>
- new()
- {
- ClusterId = clusterId,
- GenerationId = generationId,
- CachedAt = DateTime.UtcNow,
- PayloadJson = payload,
- };
-
- /// Verifies that reading a snapshot on first boot with no existing snapshot throws.
- [Fact]
- public async Task FirstBoot_NoSnapshot_ReadThrows()
- {
- var cache = new GenerationSealedCache(_root);
-
- await Should.ThrowAsync(
- () => cache.ReadCurrentAsync("cluster-a"));
- }
-
- /// Verifies that sealed snapshots can be read back correctly.
- [Fact]
- public async Task SealThenRead_RoundTrips()
- {
- var cache = new GenerationSealedCache(_root);
- var snapshot = MakeSnapshot("cluster-a", 42, "{\"hello\":\"world\"}");
-
- await cache.SealAsync(snapshot);
-
- var read = await cache.ReadCurrentAsync("cluster-a");
- read.GenerationId.ShouldBe(42);
- read.ClusterId.ShouldBe("cluster-a");
- read.PayloadJson.ShouldBe("{\"hello\":\"world\"}");
- }
-
- /// Verifies that sealed files are marked read-only on disk.
- [Fact]
- public async Task SealedFile_IsReadOnly_OnDisk()
- {
- var cache = new GenerationSealedCache(_root);
- await cache.SealAsync(MakeSnapshot("cluster-a", 5));
-
- var sealedPath = Path.Combine(_root, "cluster-a", "5.db");
- File.Exists(sealedPath).ShouldBeTrue();
- var attrs = File.GetAttributes(sealedPath);
- attrs.HasFlag(FileAttributes.ReadOnly).ShouldBeTrue("sealed file must be read-only");
- }
-
- /// Verifies that the current generation pointer advances when a new generation is sealed.
- [Fact]
- public async Task SealingTwoGenerations_PointerAdvances_ToLatest()
- {
- var cache = new GenerationSealedCache(_root);
- await cache.SealAsync(MakeSnapshot("cluster-a", 1));
- await cache.SealAsync(MakeSnapshot("cluster-a", 2));
-
- cache.TryGetCurrentGenerationId("cluster-a").ShouldBe(2);
- var read = await cache.ReadCurrentAsync("cluster-a");
- read.GenerationId.ShouldBe(2);
- }
-
- /// Verifies that prior generation files are preserved after a new seal.
- [Fact]
- public async Task PriorGenerationFile_Survives_AfterNewSeal()
- {
- var cache = new GenerationSealedCache(_root);
- await cache.SealAsync(MakeSnapshot("cluster-a", 1));
- await cache.SealAsync(MakeSnapshot("cluster-a", 2));
-
- File.Exists(Path.Combine(_root, "cluster-a", "1.db")).ShouldBeTrue(
- "prior generations preserved for audit; pruning is separate");
- File.Exists(Path.Combine(_root, "cluster-a", "2.db")).ShouldBeTrue();
- }
-
- /// Verifies that reading a corrupt sealed file fails safely.
- [Fact]
- public async Task CorruptSealedFile_ReadFailsClosed()
- {
- var cache = new GenerationSealedCache(_root);
- await cache.SealAsync(MakeSnapshot("cluster-a", 7));
-
- // Corrupt the sealed file: clear read-only, truncate, leave pointer intact.
- var sealedPath = Path.Combine(_root, "cluster-a", "7.db");
- File.SetAttributes(sealedPath, FileAttributes.Normal);
- File.WriteAllBytes(sealedPath, [0x00, 0x01, 0x02]);
-
- await Should.ThrowAsync(
- () => cache.ReadCurrentAsync("cluster-a"));
- }
-
- /// Verifies that reading with a missing sealed file fails safely.
- [Fact]
- public async Task MissingSealedFile_ReadFailsClosed()
- {
- var cache = new GenerationSealedCache(_root);
- await cache.SealAsync(MakeSnapshot("cluster-a", 3));
-
- // Delete the sealed file but leave the pointer — corruption scenario.
- var sealedPath = Path.Combine(_root, "cluster-a", "3.db");
- File.SetAttributes(sealedPath, FileAttributes.Normal);
- File.Delete(sealedPath);
-
- await Should.ThrowAsync(
- () => cache.ReadCurrentAsync("cluster-a"));
- }
-
- /// Verifies that reading with a corrupt pointer file fails safely.
- [Fact]
- public async Task CorruptPointerFile_ReadFailsClosed()
- {
- var cache = new GenerationSealedCache(_root);
- await cache.SealAsync(MakeSnapshot("cluster-a", 9));
-
- var pointerPath = Path.Combine(_root, "cluster-a", "CURRENT");
- File.WriteAllText(pointerPath, "not-a-number");
-
- await Should.ThrowAsync(
- () => cache.ReadCurrentAsync("cluster-a"));
- }
-
- /// Verifies that sealing the same generation twice is idempotent.
- [Fact]
- public async Task SealSameGenerationTwice_IsIdempotent()
- {
- var cache = new GenerationSealedCache(_root);
- await cache.SealAsync(MakeSnapshot("cluster-a", 11));
- await cache.SealAsync(MakeSnapshot("cluster-a", 11, "{\"v\":2}"));
-
- var read = await cache.ReadCurrentAsync("cluster-a");
- read.PayloadJson.ShouldBe("{\"sample\":true}", "sealed file is immutable; second seal no-ops");
- }
-
- /// Verifies that independent clusters do not interfere with each other.
- [Fact]
- public async Task IndependentClusters_DoNotInterfere()
- {
- var cache = new GenerationSealedCache(_root);
- await cache.SealAsync(MakeSnapshot("cluster-a", 1));
- await cache.SealAsync(MakeSnapshot("cluster-b", 10));
-
- (await cache.ReadCurrentAsync("cluster-a")).GenerationId.ShouldBe(1);
- (await cache.ReadCurrentAsync("cluster-b")).GenerationId.ShouldBe(10);
- }
-}
diff --git a/tests/Core/ZB.MOM.WW.OtOpcUa.Configuration.Tests/LiteDbConfigCacheTests.cs b/tests/Core/ZB.MOM.WW.OtOpcUa.Configuration.Tests/LiteDbConfigCacheTests.cs
deleted file mode 100644
index 0da9d589..00000000
--- a/tests/Core/ZB.MOM.WW.OtOpcUa.Configuration.Tests/LiteDbConfigCacheTests.cs
+++ /dev/null
@@ -1,192 +0,0 @@
-using Shouldly;
-using Xunit;
-using ZB.MOM.WW.OtOpcUa.Configuration.LocalCache;
-
-namespace ZB.MOM.WW.OtOpcUa.Configuration.Tests;
-
-[Trait("Category", "Unit")]
-public sealed class LiteDbConfigCacheTests : IDisposable
-{
- private readonly string _dbPath = Path.Combine(Path.GetTempPath(), $"otopcua-cache-test-{Guid.NewGuid():N}.db");
-
- /// Cleans up the temporary database file.
- public void Dispose()
- {
- if (File.Exists(_dbPath)) File.Delete(_dbPath);
- }
-
- private GenerationSnapshot Snapshot(string cluster, long gen) => new()
- {
- ClusterId = cluster,
- GenerationId = gen,
- CachedAt = DateTime.UtcNow,
- PayloadJson = $"{{\"g\":{gen}}}",
- };
-
- /// Verifies that payload is preserved through a write-then-read cycle.
- [Fact]
- public async Task Roundtrip_preserves_payload()
- {
- using var cache = new LiteDbConfigCache(_dbPath);
- var put = Snapshot("c-1", 42);
- await cache.PutAsync(put);
-
- var got = await cache.GetMostRecentAsync("c-1");
- got.ShouldNotBeNull();
- got!.GenerationId.ShouldBe(42);
- got.PayloadJson.ShouldBe(put.PayloadJson);
- }
-
- /// Verifies that GetMostRecentAsync returns the latest generation when multiple exist.
- [Fact]
- public async Task GetMostRecent_returns_latest_when_multiple_generations_present()
- {
- using var cache = new LiteDbConfigCache(_dbPath);
- foreach (var g in new long[] { 10, 20, 15 })
- await cache.PutAsync(Snapshot("c-1", g));
-
- var got = await cache.GetMostRecentAsync("c-1");
- got!.GenerationId.ShouldBe(20);
- }
-
- /// Verifies that GetMostRecentAsync returns null for an unknown cluster.
- [Fact]
- public async Task GetMostRecent_returns_null_for_unknown_cluster()
- {
- using var cache = new LiteDbConfigCache(_dbPath);
- (await cache.GetMostRecentAsync("ghost")).ShouldBeNull();
- }
-
- /// Verifies that Prune keeps the latest N generations and drops older ones.
- [Fact]
- public async Task Prune_keeps_latest_N_and_drops_older()
- {
- using var cache = new LiteDbConfigCache(_dbPath);
- for (long g = 1; g <= 15; g++)
- await cache.PutAsync(Snapshot("c-1", g));
-
- await cache.PruneOldGenerationsAsync("c-1", keepLatest: 10);
-
- (await cache.GetMostRecentAsync("c-1"))!.GenerationId.ShouldBe(15);
-
- // Drop them one by one and count — should be exactly 10 remaining
- var count = 0;
- while (await cache.GetMostRecentAsync("c-1") is not null)
- {
- count++;
- await cache.PruneOldGenerationsAsync("c-1", keepLatest: Math.Max(0, 10 - count));
- if (count > 20) break; // safety
- }
- count.ShouldBe(10);
- }
-
- /// Verifies that writing the same cluster/generation twice replaces rather than duplicates.
- [Fact]
- public async Task Put_same_cluster_generation_twice_replaces_not_duplicates()
- {
- using var cache = new LiteDbConfigCache(_dbPath);
- var first = Snapshot("c-1", 1);
- first.PayloadJson = "{\"v\":1}";
- await cache.PutAsync(first);
-
- var second = Snapshot("c-1", 1);
- second.PayloadJson = "{\"v\":2}";
- await cache.PutAsync(second);
-
- (await cache.GetMostRecentAsync("c-1"))!.PayloadJson.ShouldBe("{\"v\":2}");
- }
-
- // ------------------------------------------------------------------------------------
- // Configuration-005 — concurrent PutAsync for the same (ClusterId, GenerationId) must
- // not produce duplicate rows. The original find-then-insert was non-atomic so two racing
- // callers could both observe `existing is null` and both Insert.
- // ------------------------------------------------------------------------------------
- /// Verifies that concurrent PutAsync calls for the same cluster and generation do not create duplicates.
- [Fact]
- public async Task PutAsync_concurrent_for_same_cluster_and_generation_does_not_duplicate()
- {
- using var cache = new LiteDbConfigCache(_dbPath);
- // Pre-seed gen=99 so prune keepLatest:1 has a sentinel that survives independent of
- // any potential duplicate (gen=42) row count.
- await cache.PutAsync(Snapshot("c-1", 99));
-
- // Many parallel writes for the same key. Without serialization, racing find-then-insert
- // would Insert multiple rows for the same (ClusterId, GenerationId=42).
- var tasks = Enumerable.Range(0, 64).Select(_ => Task.Run(async () =>
- {
- var s = Snapshot("c-1", 42);
- await cache.PutAsync(s);
- })).ToArray();
-
- await Task.WhenAll(tasks);
-
- // Count rows for gen=42 directly by inspecting the LiteDB file via a fresh handle.
- cache.Dispose();
- using var verify = new LiteDB.LiteDatabase(_dbPath);
- var col = verify.GetCollection("generations");
- var gen42Count = col.Find(s => s.ClusterId == "c-1" && s.GenerationId == 42).Count();
- gen42Count.ShouldBe(1,
- $"PutAsync must upsert atomically — found {gen42Count} rows for (c-1, gen=42) after 64 concurrent puts");
- }
-
- // ------------------------------------------------------------------------------------
- // Configuration-012 — the per-instance _writeGate (Configuration-005) does not protect
- // against LiteDB's process-wide BsonMapper.Global lazy-init race. Many cache INSTANCES
- // constructed + driven concurrently corrupt the shared global mapper, surfacing as
- // "Member ClusterId not found on BsonMapper" or a bogus "duplicate key _id = 0". A private
- // per-database mapper with the entity pre-registered fixes it.
- // ------------------------------------------------------------------------------------
- /// Verifies that many cache instances constructed and driven concurrently do not
- /// corrupt LiteDB's shared global BsonMapper — each Put/Get round-trips its own payload and
- /// no insert throws a member-not-found or duplicate-_id exception.
- [Fact]
- public async Task Concurrent_cache_instances_do_not_race_the_shared_bson_mapper()
- {
- var paths = new List();
- try
- {
- var outer = Enumerable.Range(0, 24).Select(i => Task.Run(async () =>
- {
- var path = Path.Combine(Path.GetTempPath(), $"otopcua-cache-mapperrace-{Guid.NewGuid():N}.db");
- lock (paths) paths.Add(path);
-
- using var cache = new LiteDbConfigCache(path);
- // Pre-seed a sentinel, then hammer one (cluster, gen) from many threads.
- await cache.PutAsync(Snapshot($"c-{i}", 99));
- var inner = Enumerable.Range(0, 16)
- .Select(_ => Task.Run(() => cache.PutAsync(Snapshot($"c-{i}", 42))))
- .ToArray();
- await Task.WhenAll(inner);
-
- var got = await cache.GetMostRecentAsync($"c-{i}");
- got.ShouldNotBeNull();
- got!.GenerationId.ShouldBe(99); // 99 > 42, latest by GenerationId
- })).ToArray();
-
- // The unfixed code throws LiteException / NotSupportedException out of these tasks under
- // the global-mapper race; the fixed code completes cleanly.
- await Task.WhenAll(outer);
- }
- finally
- {
- foreach (var p in paths)
- if (File.Exists(p)) File.Delete(p);
- }
- }
-
- /// Verifies that a corrupted cache file surfaces as LocalConfigCacheCorruptException.
- [Fact]
- public void Corrupt_file_surfaces_as_LocalConfigCacheCorruptException()
- {
- // Write a file large enough to look like a LiteDB page but with garbage contents so page
- // deserialization fails on the first read probe.
- File.WriteAllBytes(_dbPath, new byte[8192]);
- Array.Fill(File.ReadAllBytes(_dbPath), 0xAB);
- using (var fs = File.OpenWrite(_dbPath))
- {
- fs.Write(new byte[8192].Select(_ => (byte)0xAB).ToArray());
- }
-
- Should.Throw(() => new LiteDbConfigCache(_dbPath));
- }
-}
diff --git a/tests/Core/ZB.MOM.WW.OtOpcUa.Configuration.Tests/ResilientConfigReaderTests.cs b/tests/Core/ZB.MOM.WW.OtOpcUa.Configuration.Tests/ResilientConfigReaderTests.cs
deleted file mode 100644
index fd24bdbd..00000000
--- a/tests/Core/ZB.MOM.WW.OtOpcUa.Configuration.Tests/ResilientConfigReaderTests.cs
+++ /dev/null
@@ -1,355 +0,0 @@
-using Microsoft.Extensions.Logging;
-using Microsoft.Extensions.Logging.Abstractions;
-using Polly.Timeout;
-using Shouldly;
-using Xunit;
-using ZB.MOM.WW.OtOpcUa.Configuration.LocalCache;
-
-namespace ZB.MOM.WW.OtOpcUa.Configuration.Tests;
-
-[Trait("Category", "Unit")]
-public sealed class ResilientConfigReaderTests : IDisposable
-{
- private readonly string _root = Path.Combine(Path.GetTempPath(), $"otopcua-reader-{Guid.NewGuid():N}");
-
- /// Disposes temporary test files.
- public void Dispose()
- {
- try
- {
- if (!Directory.Exists(_root)) return;
- foreach (var f in Directory.EnumerateFiles(_root, "*", SearchOption.AllDirectories))
- File.SetAttributes(f, FileAttributes.Normal);
- Directory.Delete(_root, recursive: true);
- }
- catch { /* best-effort */ }
- }
-
- /// Verifies that successful central DB reads return value and mark fresh.
- [Fact]
- public async Task CentralDbSucceeds_ReturnsValue_MarksFresh()
- {
- var cache = new GenerationSealedCache(_root);
- var flag = new StaleConfigFlag { };
- flag.MarkStale(); // pre-existing stale state
- var reader = new ResilientConfigReader(cache, flag, NullLogger.Instance);
-
- var result = await reader.ReadAsync(
- "cluster-a",
- _ => ValueTask.FromResult("fresh-from-db"),
- _ => "from-cache",
- CancellationToken.None);
-
- result.ShouldBe("fresh-from-db");
- flag.IsStale.ShouldBeFalse("successful central-DB read clears stale flag");
- }
-
- /// Verifies that exhausted retries fall back to cache and mark stale.
- [Fact]
- public async Task CentralDbFails_ExhaustsRetries_FallsBackToCache_MarksStale()
- {
- var cache = new GenerationSealedCache(_root);
- await cache.SealAsync(new GenerationSnapshot
- {
- ClusterId = "cluster-a", GenerationId = 99, CachedAt = DateTime.UtcNow,
- PayloadJson = "{\"cached\":true}",
- });
- var flag = new StaleConfigFlag();
- var reader = new ResilientConfigReader(cache, flag, NullLogger.Instance,
- timeout: TimeSpan.FromSeconds(10), retryCount: 2);
- var attempts = 0;
-
- var result = await reader.ReadAsync(
- "cluster-a",
- _ =>
- {
- attempts++;
- throw new InvalidOperationException("SQL dead");
-#pragma warning disable CS0162
- return ValueTask.FromResult("never");
-#pragma warning restore CS0162
- },
- snap => snap.PayloadJson,
- CancellationToken.None);
-
- attempts.ShouldBe(3, "1 initial + 2 retries = 3 attempts");
- result.ShouldBe("{\"cached\":true}");
- flag.IsStale.ShouldBeTrue("cache fallback flips stale flag true");
- }
-
- /// Verifies that DB failure with unavailable cache throws.
- [Fact]
- public async Task CentralDbFails_AndCacheAlsoUnavailable_Throws()
- {
- var cache = new GenerationSealedCache(_root);
- var flag = new StaleConfigFlag();
- var reader = new ResilientConfigReader(cache, flag, NullLogger.Instance,
- timeout: TimeSpan.FromSeconds(10), retryCount: 0);
-
- await Should.ThrowAsync(async () =>
- {
- await reader.ReadAsync(
- "cluster-a",
- _ => throw new InvalidOperationException("SQL dead"),
- _ => "never",
- CancellationToken.None);
- });
-
- flag.IsStale.ShouldBeFalse("no snapshot ever served, so flag stays whatever it was");
- }
-
- /// Verifies that cancellation is not retried.
- [Fact]
- public async Task Cancellation_NotRetried()
- {
- var cache = new GenerationSealedCache(_root);
- var flag = new StaleConfigFlag();
- var reader = new ResilientConfigReader(cache, flag, NullLogger.Instance,
- timeout: TimeSpan.FromSeconds(10), retryCount: 5);
- using var cts = new CancellationTokenSource();
- cts.Cancel();
- var attempts = 0;
-
- await Should.ThrowAsync(async () =>
- {
- await reader.ReadAsync(
- "cluster-a",
- ct =>
- {
- attempts++;
- ct.ThrowIfCancellationRequested();
- return ValueTask.FromResult("ok");
- },
- _ => "cache",
- cts.Token);
- });
-
- attempts.ShouldBeLessThanOrEqualTo(1);
- }
-
- // ------------------------------------------------------------------------------------
- // Configuration-006 — command-timeout TaskCanceledException and TimeoutRejectedException
- // must fall back to the sealed cache, not propagate as caller cancellation.
- // ------------------------------------------------------------------------------------
-
- /// Verifies that command timeout TaskCanceledException falls back to cache.
- [Fact]
- public async Task CommandTimeout_TaskCanceledException_FallsBackToCache()
- {
- // A SQL command-level timeout surfaces as a TaskCanceledException thrown by the
- // delegate itself (not triggered by the caller's CancellationToken). It must be
- // treated as a transient failure and trigger the cache fallback, not be mistaken
- // for genuine caller cancellation and propagated.
- var cache = new GenerationSealedCache(_root);
- await cache.SealAsync(new GenerationSnapshot
- {
- ClusterId = "cluster-b", GenerationId = 7, CachedAt = DateTime.UtcNow,
- PayloadJson = "{\"from\":\"cache\"}",
- });
- var flag = new StaleConfigFlag();
- var reader = new ResilientConfigReader(cache, flag, NullLogger.Instance,
- timeout: TimeSpan.FromSeconds(10), retryCount: 0);
-
- // Simulate a command-level timeout: TaskCanceledException with no linked token.
- var result = await reader.ReadAsync(
- "cluster-b",
- _ => throw new TaskCanceledException("SQL command timeout (no caller token)"),
- snap => snap.PayloadJson,
- CancellationToken.None); // caller token is NOT cancelled
-
- result.ShouldBe("{\"from\":\"cache\"}",
- "command-timeout TaskCanceledException must fall back to sealed cache");
- flag.IsStale.ShouldBeTrue("cache fallback marks the stale flag");
- }
-
- /// Verifies that Polly timeout rejection falls back to cache.
- [Fact]
- public async Task PollyTimeout_TimeoutRejectedException_FallsBackToCache()
- {
- // When Polly's own timeout strategy fires it throws TimeoutRejectedException.
- // That should trigger the cache fallback just like any other transient error.
- var cache = new GenerationSealedCache(_root);
- await cache.SealAsync(new GenerationSnapshot
- {
- ClusterId = "cluster-c", GenerationId = 8, CachedAt = DateTime.UtcNow,
- PayloadJson = "{\"from\":\"polly-timeout-cache\"}",
- });
- var flag = new StaleConfigFlag();
- // Set an extremely short Polly timeout so the async delay triggers it.
- var reader = new ResilientConfigReader(cache, flag, NullLogger.Instance,
- timeout: TimeSpan.FromMilliseconds(10), retryCount: 0);
-
- var result = await reader.ReadAsync(
- "cluster-c",
- async ct =>
- {
- await Task.Delay(TimeSpan.FromSeconds(5), ct); // far exceeds 10 ms timeout
- return "never";
- },
- snap => snap.PayloadJson,
- CancellationToken.None);
-
- result.ShouldBe("{\"from\":\"polly-timeout-cache\"}",
- "Polly TimeoutRejectedException must fall back to sealed cache");
- flag.IsStale.ShouldBeTrue("cache fallback marks the stale flag");
- }
-
- // ------------------------------------------------------------------------------------
- // Configuration-010 — fallback warning log must scrub connection-string fragments and
- // must not include the full exception object (which carries the stack and any inner-
- // exception chain). Project rule: no credential or connection-string fragment in logs.
- // ------------------------------------------------------------------------------------
-
- /// Verifies that fallback warnings do not log exceptions or password fragments.
- [Fact]
- public async Task FallbackWarning_does_not_log_full_exception_object_or_password_fragment()
- {
- var cache = new GenerationSealedCache(_root);
- await cache.SealAsync(new GenerationSnapshot
- {
- ClusterId = "cluster-e", GenerationId = 1, CachedAt = DateTime.UtcNow,
- PayloadJson = "{\"ok\":true}",
- });
- var flag = new StaleConfigFlag();
- var capturing = new CapturingLogger();
- var reader = new ResilientConfigReader(cache, flag, capturing,
- timeout: TimeSpan.FromSeconds(10), retryCount: 0);
-
- // Simulated SqlException-style message carrying a connection-string fragment, the
- // kind of thing a poorly-wrapped delegate could surface.
- const string secretBearingMessage =
- "Login failed for user 'sa'. (Server=sql.example.com,1433;User Id=sa;Password=SuperSecret123!)";
-
- await reader.ReadAsync(
- "cluster-e",
- _ => throw new InvalidOperationException(secretBearingMessage),
- snap => snap.PayloadJson,
- CancellationToken.None);
-
- var warning = capturing.Records.ShouldHaveSingleItem();
- warning.LogLevel.ShouldBe(LogLevel.Warning);
-
- // The exception object passed as the first arg to LogWarning(ex, ...) drives the
- // formatter's stack-trace dump; capturing it lets us assert the scrubbing surface.
- warning.Exception.ShouldBeNull(
- "the warning must not attach the raw exception — it can carry connection-string fragments");
-
- // The rendered message must not echo password / user-id strings even if the caller
- // embedded them in the exception message.
- warning.RenderedMessage.ShouldNotContain("Password=", Case.Insensitive);
- warning.RenderedMessage.ShouldNotContain("SuperSecret123!");
- warning.RenderedMessage.ShouldNotContain("User Id=", Case.Insensitive);
- }
-
- /// Verifies that caller cancellation propagates rather than falling back.
- [Fact]
- public async Task CallerCancellation_Propagates_NotFallback()
- {
- // Explicit caller cancellation must NOT fall back to the sealed cache — the
- // caller said stop, so we must stop.
- var cache = new GenerationSealedCache(_root);
- await cache.SealAsync(new GenerationSnapshot
- {
- ClusterId = "cluster-d", GenerationId = 9, CachedAt = DateTime.UtcNow,
- PayloadJson = "{\"should\":\"not be returned\"}",
- });
- var flag = new StaleConfigFlag();
- var reader = new ResilientConfigReader(cache, flag, NullLogger.Instance,
- timeout: TimeSpan.FromSeconds(10), retryCount: 0);
- using var cts = new CancellationTokenSource();
- cts.Cancel();
-
- await Should.ThrowAsync(async () =>
- {
- await reader.ReadAsync(
- "cluster-d",
- ct =>
- {
- ct.ThrowIfCancellationRequested();
- return ValueTask.FromResult("ok");
- },
- _ => "cache-should-not-be-used",
- cts.Token);
- });
-
- flag.IsStale.ShouldBeFalse("no cache snapshot served on genuine cancellation");
- }
-}
-
-/// Represents a captured log record for testing.
-internal sealed record LogRecord(LogLevel LogLevel, string RenderedMessage, Exception? Exception);
-
-/// Captures log records for assertion in tests.
-internal sealed class CapturingLogger : ILogger
-{
- /// Gets the list of captured log records.
- public List Records { get; } = new();
-
- /// Begins a scope (no-op for testing).
- /// The type of the scope state.
- /// The scope state.
- /// A disposable scope handle.
- public IDisposable BeginScope(TState state) where TState : notnull => NullScope.Instance;
-
- /// Returns true to enable all log levels.
- /// The log level to check.
- /// True to indicate the log level is enabled.
- public bool IsEnabled(LogLevel logLevel) => true;
-
- /// Logs a message by capturing it.
- /// The type of the log state.
- /// The log level.
- /// The event identifier.
- /// The log state.
- /// The exception, if any.
- /// Function to format the log message.
- public void Log(LogLevel logLevel, EventId eventId, TState state, Exception? exception, Func formatter)
- {
- Records.Add(new LogRecord(logLevel, formatter(state, exception), exception));
- }
-
- /// No-op scope for testing.
- private sealed class NullScope : IDisposable
- {
- /// Gets the singleton instance.
- public static readonly NullScope Instance = new();
-
- /// Disposes the scope (no-op).
- public void Dispose() { }
- }
-}
-
-[Trait("Category", "Unit")]
-public sealed class StaleConfigFlagTests
-{
- /// Verifies that default state is fresh.
- [Fact]
- public void Default_IsFresh()
- {
- new StaleConfigFlag().IsStale.ShouldBeFalse();
- }
-
- /// Verifies that stale and fresh states toggle correctly.
- [Fact]
- public void MarkStale_ThenFresh_Toggles()
- {
- var flag = new StaleConfigFlag();
- flag.MarkStale();
- flag.IsStale.ShouldBeTrue();
- flag.MarkFresh();
- flag.IsStale.ShouldBeFalse();
- }
-
- /// Verifies that concurrent writes converge to the final state.
- [Fact]
- public void ConcurrentWrites_Converge()
- {
- var flag = new StaleConfigFlag();
- Parallel.For(0, 1000, i =>
- {
- if (i % 2 == 0) flag.MarkStale(); else flag.MarkFresh();
- });
- flag.MarkFresh();
- flag.IsStale.ShouldBeFalse();
- }
-}
diff --git a/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDb/LocalDbPairConvergenceTests.cs b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDb/LocalDbPairConvergenceTests.cs
new file mode 100644
index 00000000..bfa317d3
--- /dev/null
+++ b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDb/LocalDbPairConvergenceTests.cs
@@ -0,0 +1,178 @@
+using Xunit;
+using ZB.MOM.WW.LocalDb;
+using ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache;
+
+namespace ZB.MOM.WW.OtOpcUa.Host.IntegrationTests.LocalDb;
+
+///
+/// LocalDb Phase 1 (Task 12) — two driver nodes replicating the deployment-artifact cache over a
+/// real loopback h2c transport, through the real fail-closed interceptor.
+///
+///
+/// This is the test the whole phase exists to pass: does a redundant driver pair actually share
+/// the cached configuration a node boots from when central SQL is down? Every upstream piece —
+/// the schema, the DI wiring, the interceptor, the chunked cache — can be individually green
+/// while the pair still fails to converge.
+///
+[Collection("LocalDbPairConvergence")]
+public sealed class LocalDbPairConvergenceTests
+{
+ private const string Cluster = "cluster-1";
+
+ ///
+ /// A deterministic artifact large enough to force multiple 128 KiB chunks (≈ 3 here), so the
+ /// tests exercise the chunk-split/reassemble path rather than a single-row shortcut.
+ ///
+ private static byte[] MultiChunkArtifact(int seed, int size = 300 * 1024)
+ {
+ var bytes = new byte[size];
+ new Random(seed).NextBytes(bytes);
+ return bytes;
+ }
+
+ private static async Task ReadArtifactAsync(IDeploymentArtifactCache cache, string cluster)
+ {
+ var cached = await cache.GetCurrentAsync(cluster, TestContext.Current.CancellationToken);
+ return cached?.Artifact;
+ }
+
+ [Fact]
+ public async Task ArtifactStoredOnA_ConvergesToB_ByteIdentical()
+ {
+ await using var harness = new LocalDbPairHarness();
+ await harness.StartAsync();
+
+ var deploymentId = Guid.NewGuid().ToString("N");
+ var artifact = MultiChunkArtifact(seed: 1);
+ await harness.CacheA.StoreAsync(Cluster, deploymentId, "rev-1", artifact,
+ TestContext.Current.CancellationToken);
+
+ // B's cache reassembles byte-for-byte what A stored — proving the chunk rows, chunk_count,
+ // and pointer all replicated intact through the real transport + interceptor.
+ await LocalDbPairHarness.WaitUntilAsync(
+ async () =>
+ {
+ var onB = await ReadArtifactAsync(harness.CacheB, Cluster);
+ return onB is not null && onB.AsSpan().SequenceEqual(artifact);
+ },
+ "the artifact stored on node A to be byte-identical on node B");
+
+ // The pointer row on B carries A's origin HLC + node id, not a locally re-derived stamp: the
+ // row_version dumps for the pointer table are identical on both nodes. This is what
+ // distinguishes "B replicated A's row" from "both nodes happened to write equal content".
+ await LocalDbPairHarness.WaitUntilAsync(
+ async () =>
+ await LocalDbPairHarness.DumpRowVersionAsync(harness.A, DeploymentCacheSchema.PointerTable)
+ == await LocalDbPairHarness.DumpRowVersionAsync(harness.B, DeploymentCacheSchema.PointerTable),
+ "both nodes' pointer row_version (hlc + origin node id) to match");
+ }
+
+ [Fact]
+ public async Task RetentionPruneOnA_TombstonesReachB()
+ {
+ await using var harness = new LocalDbPairHarness();
+ await harness.StartAsync();
+
+ // Three deployments for one cluster. The cache retains the newest two, so the first is pruned
+ // on A — and that prune must replicate to B as tombstones, not linger as live chunks.
+ var dep1 = Guid.NewGuid().ToString("N");
+ var dep2 = Guid.NewGuid().ToString("N");
+ var dep3 = Guid.NewGuid().ToString("N");
+
+ await harness.CacheA.StoreAsync(Cluster, dep1, "rev-1", MultiChunkArtifact(1),
+ TestContext.Current.CancellationToken);
+ await harness.CacheA.StoreAsync(Cluster, dep2, "rev-2", MultiChunkArtifact(2),
+ TestContext.Current.CancellationToken);
+
+ // dep1's chunks are still present until the third store prunes them — confirm they reached B
+ // first, so the later disappearance is a real replicated prune rather than never-arrived.
+ await LocalDbPairHarness.WaitUntilAsync(
+ async () => await LocalDbPairHarness.ChunkCountAsync(harness.B, dep1) > 0,
+ "dep1's chunks to reach node B before the prune");
+
+ await harness.CacheA.StoreAsync(Cluster, dep3, "rev-3", MultiChunkArtifact(3),
+ TestContext.Current.CancellationToken);
+
+ // A pruned dep1 locally; B must follow via replicated deletes: dep1 gone, tombstones present,
+ // and the surviving newest (dep3) intact.
+ await LocalDbPairHarness.WaitUntilAsync(
+ async () =>
+ await LocalDbPairHarness.ChunkCountAsync(harness.B, dep1) == 0
+ && await LocalDbPairHarness.TombstoneCountAsync(harness.B, DeploymentCacheSchema.ArtifactsTable) > 0
+ && await LocalDbPairHarness.ChunkCountAsync(harness.B, dep3) > 0,
+ "node B to prune dep1 (with tombstones) while keeping dep3");
+ }
+
+ [Fact]
+ public async Task WritesWhileTransportDown_SurviveRejoin()
+ {
+ await using var harness = new LocalDbPairHarness();
+ await harness.StartAsync();
+
+ var dep1 = Guid.NewGuid().ToString("N");
+ var first = MultiChunkArtifact(seed: 10);
+ await harness.CacheA.StoreAsync(Cluster, dep1, "rev-1", first,
+ TestContext.Current.CancellationToken);
+ await LocalDbPairHarness.WaitUntilAsync(
+ async () => await ReadArtifactAsync(harness.CacheB, Cluster) is { } b && b.AsSpan().SequenceEqual(first),
+ "the first artifact to converge before the outage");
+
+ // B goes offline; A keeps deploying. B's database survives the host teardown (pre-constructed
+ // instance), so the rejoin is genuine rather than a fresh node.
+ await harness.StopPassiveAsync();
+
+ var dep2 = Guid.NewGuid().ToString("N");
+ var second = MultiChunkArtifact(seed: 11);
+ await harness.CacheA.StoreAsync(Cluster, dep2, "rev-2", second,
+ TestContext.Current.CancellationToken);
+
+ await harness.RestartPairAsync();
+
+ // Everything written during the outage catches up: B now serves the newer deployment.
+ await LocalDbPairHarness.WaitUntilAsync(
+ async () =>
+ {
+ var cached = await harness.CacheB.GetCurrentAsync(Cluster, TestContext.Current.CancellationToken);
+ return cached is not null
+ && cached.DeploymentId == dep2
+ && cached.Artifact.AsSpan().SequenceEqual(second);
+ },
+ "node B to catch up on the deployment written while it was offline");
+ }
+
+ [Fact]
+ public async Task WrongApiKey_NeverConverges_WithMatchingKeyPositiveControl()
+ {
+ // Negative half: A and B hold different keys, so B's interceptor fail-closes on A's stream.
+ await using (var mismatched = new LocalDbPairHarness(
+ apiKeyA: LocalDbPairHarness.DefaultApiKey, apiKeyB: "a-different-key-entirely"))
+ {
+ await mismatched.StartAsync();
+
+ var deploymentId = Guid.NewGuid().ToString("N");
+ await mismatched.CacheA.StoreAsync(Cluster, deploymentId, "rev-1", MultiChunkArtifact(20),
+ TestContext.Current.CancellationToken);
+
+ var converged = await LocalDbPairHarness.HeldWithinAsync(
+ async () => await ReadArtifactAsync(mismatched.CacheB, Cluster) is not null,
+ TimeSpan.FromSeconds(5));
+
+ Assert.False(converged,
+ "a key mismatch must stop the pair converging — the interceptor is fail-closed");
+ }
+
+ // Positive control: the SAME scenario with matching keys DOES converge. Without this, the
+ // negative assertion above would pass even if convergence were broken for an unrelated reason.
+ await using var matched = new LocalDbPairHarness();
+ await matched.StartAsync();
+
+ var controlId = Guid.NewGuid().ToString("N");
+ var controlArtifact = MultiChunkArtifact(seed: 21);
+ await matched.CacheA.StoreAsync(Cluster, controlId, "rev-1", controlArtifact,
+ TestContext.Current.CancellationToken);
+
+ await LocalDbPairHarness.WaitUntilAsync(
+ async () => await ReadArtifactAsync(matched.CacheB, Cluster) is { } b && b.AsSpan().SequenceEqual(controlArtifact),
+ "the matching-key control pair to converge");
+ }
+}
diff --git a/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDb/LocalDbPairHarness.cs b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDb/LocalDbPairHarness.cs
new file mode 100644
index 00000000..d2815590
--- /dev/null
+++ b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDb/LocalDbPairHarness.cs
@@ -0,0 +1,324 @@
+using System.Net;
+using Microsoft.AspNetCore.Builder;
+using Microsoft.AspNetCore.Hosting;
+using Microsoft.AspNetCore.Hosting.Server;
+using Microsoft.AspNetCore.Hosting.Server.Features;
+using Microsoft.AspNetCore.Server.Kestrel.Core;
+using Microsoft.Extensions.Configuration;
+using Microsoft.Extensions.DependencyInjection;
+using Microsoft.Extensions.Hosting;
+using Microsoft.Extensions.Logging.Abstractions;
+using Xunit;
+using ZB.MOM.WW.LocalDb;
+using ZB.MOM.WW.LocalDb.Replication;
+using ZB.MOM.WW.OtOpcUa.Host.Configuration;
+using ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache;
+
+namespace ZB.MOM.WW.OtOpcUa.Host.IntegrationTests.LocalDb;
+
+///
+/// Serializes the two-node convergence tests against one another: each stands up a real Kestrel
+/// h2c listener plus two SQLite files, and running them concurrently under CI contention is a
+/// flakiness risk. The rest of the assembly parallelizes normally.
+///
+[CollectionDefinition("LocalDbPairConvergence")]
+public sealed class LocalDbPairConvergenceCollection;
+
+///
+/// Two driver-role OtOpcUa nodes replicating the deployment-artifact cache over a REAL loopback
+/// gRPC transport (Kestrel h2c on 127.0.0.1), through the REAL fail-closed
+/// . Node A is the initiator (it dials the peer); node B
+/// is passive (it hosts MapZbLocalDbSync).
+///
+///
+///
+/// Both databases are initialised through the production —
+/// a hand-written schema here would prove only that the test agrees with itself. The tables,
+/// their primary keys, and the DDL→RegisterReplicated ordering under test are the ones
+/// the host actually runs.
+///
+///
+/// The two instances are owned by the harness and registered into the
+/// hosts as pre-constructed singletons. MS.DI does not dispose instances it did not create,
+/// so tearing a host down (the transport-loss scenario) leaves the databases intact and
+/// writable — which is exactly what lets node A accumulate writes while B is down.
+///
+///
+/// A real loopback socket (not an in-memory TestServer) is deliberate: killing the passive
+/// host disposes the server and closes the socket, faulting the initiator's active stream
+/// promptly. An in-memory TestServer does not model a connection drop and leaves the client
+/// hanging.
+///
+///
+/// The API keys are per-node so the wrong-key scenario can hand A and B different keys and
+/// assert non-convergence — with a matching-key positive control proving the same harness
+/// does converge when the keys agree (an absence assertion without a positive control passed
+/// vacuously in ScadaBridge).
+///
+/// Offline: no docker, no external services.
+///
+public sealed class LocalDbPairHarness : IAsyncDisposable
+{
+ /// The key both nodes share unless a test overrides one of them.
+ public const string DefaultApiKey = "otopcua-localdb-pair-convergence-key";
+
+ /// How long a scenario waits for the pair to agree (or to stay diverged) before deciding.
+ public static readonly TimeSpan ConvergeTimeout = TimeSpan.FromSeconds(30);
+
+ private readonly string _apiKeyA;
+ private readonly string _apiKeyB;
+
+ private readonly string _pathA = Path.Combine(Path.GetTempPath(), $"otopcua-pairA-{Guid.NewGuid():N}.db");
+ private readonly string _pathB = Path.Combine(Path.GetTempPath(), $"otopcua-pairB-{Guid.NewGuid():N}.db");
+
+ private readonly ServiceProvider _dbProviderA;
+ private readonly ServiceProvider _dbProviderB;
+
+ private IHost? _serverHost; // node B — passive
+ private IHost? _initiatorHost; // node A — dials the peer
+
+ static LocalDbPairHarness() =>
+ // Grpc.Net.Client dials the loopback server over HTTP/2 cleartext (h2c).
+ AppContext.SetSwitch("System.Net.Http.SocketsHttpHandler.Http2UnencryptedSupport", true);
+
+ /// Node A's replication key. Defaults to .
+ /// Node B's replication key. Defaults to .
+ public LocalDbPairHarness(string? apiKeyA = null, string? apiKeyB = null)
+ {
+ _apiKeyA = apiKeyA ?? DefaultApiKey;
+ _apiKeyB = apiKeyB ?? DefaultApiKey;
+
+ _dbProviderA = BuildDatabaseProvider(_pathA);
+ _dbProviderB = BuildDatabaseProvider(_pathB);
+ }
+
+ /// Node A — the initiator, which dials the peer.
+ public ILocalDb A => _dbProviderA.GetRequiredService();
+
+ /// Node B — the passive node, which listens.
+ public ILocalDb B => _dbProviderB.GetRequiredService();
+
+ /// The production artifact cache over node A's database.
+ public IDeploymentArtifactCache CacheA =>
+ new LocalDbDeploymentArtifactCache(A, NullLogger.Instance);
+
+ /// The production artifact cache over node B's database.
+ public IDeploymentArtifactCache CacheB =>
+ new LocalDbDeploymentArtifactCache(B, NullLogger.Instance);
+
+ // ---- lifecycle --------------------------------------------------------------------------
+
+ /// Forces both databases to construct (running OnReady) then brings the pair online.
+ public async Task StartAsync()
+ {
+ _ = A;
+ _ = B;
+
+ await StartPassiveAsync();
+ await StartInitiatorAsync();
+ }
+
+ /// Takes node B's listener down, leaving its database intact and writable.
+ public async Task StopPassiveAsync()
+ {
+ await StopHostAsync(_serverHost);
+ _serverHost = null;
+ }
+
+ ///
+ /// Brings node B back on a NEW loopback port and re-dials from A. The initiator re-reads the
+ /// peer address on each reconnect, so this is a genuine rejoin over the same databases.
+ ///
+ public async Task RestartPairAsync()
+ {
+ await StartPassiveAsync();
+ await StopHostAsync(_initiatorHost);
+ _initiatorHost = null;
+ await StartInitiatorAsync();
+ }
+
+ // ---- convergence helpers ----------------------------------------------------------------
+
+ /// Polls until true or the deadline passes; fails otherwise.
+ public static async Task WaitUntilAsync(Func> condition, string because)
+ {
+ var deadline = DateTime.UtcNow + ConvergeTimeout;
+ while (DateTime.UtcNow < deadline)
+ {
+ if (await condition())
+ return;
+ await Task.Delay(50);
+ }
+
+ Assert.Fail($"Timed out after {ConvergeTimeout.TotalSeconds:0}s waiting for: {because}");
+ }
+
+ /// Waits out a bounded window and returns whether ever held.
+ ///
+ /// For the negative half of the wrong-key scenario: it must NOT converge. A short, fixed
+ /// window keeps the test quick while still giving a matching-key control ample time to
+ /// converge (the control uses with the full timeout).
+ ///
+ public static async Task HeldWithinAsync(Func> condition, TimeSpan window)
+ {
+ var deadline = DateTime.UtcNow + window;
+ while (DateTime.UtcNow < deadline)
+ {
+ if (await condition())
+ return true;
+ await Task.Delay(50);
+ }
+
+ return false;
+ }
+
+ ///
+ /// Dumps the __localdb_row_version rows for one table, ordered by pk, as
+ /// pk_json|hlc|node_id|is_tombstone. Equal dumps on both nodes prove B holds A's
+ /// origin-stamped row (same HLC + node id), not a locally re-derived one.
+ ///
+ public static async Task DumpRowVersionAsync(ILocalDb db, string table)
+ {
+ var rows = await db.QueryAsync(
+ """
+ SELECT pk_json, hlc, node_id, is_tombstone
+ FROM __localdb_row_version
+ WHERE table_name = @Table
+ ORDER BY pk_json
+ """,
+ static r => $"{r.GetString(0)}|{r.GetInt64(1)}|{r.GetString(2)}|{r.GetInt64(3)}",
+ new { Table = table });
+ return string.Join("\n", rows);
+ }
+
+ /// Counts deployment_artifacts chunk rows for one deployment on a node.
+ public static async Task ChunkCountAsync(ILocalDb db, string deploymentId)
+ {
+ var rows = await db.QueryAsync(
+ "SELECT COUNT(*) FROM deployment_artifacts WHERE deployment_id = @DeploymentId",
+ static r => r.GetInt64(0),
+ new { DeploymentId = deploymentId });
+ return rows[0];
+ }
+
+ /// Counts tombstone rows in __localdb_row_version for one table on a node.
+ public static async Task TombstoneCountAsync(ILocalDb db, string table)
+ {
+ var rows = await db.QueryAsync(
+ """
+ SELECT COUNT(*) FROM __localdb_row_version
+ WHERE table_name = @Table AND is_tombstone = 1
+ """,
+ static r => r.GetInt64(0),
+ new { Table = table });
+ return rows[0];
+ }
+
+ // ---- fixture internals ------------------------------------------------------------------
+
+ ///
+ /// A provider owning one deployment-cache database, initialised through the host's own
+ /// — same schema, same registration order.
+ ///
+ private static ServiceProvider BuildDatabaseProvider(string path)
+ {
+ var config = new ConfigurationBuilder()
+ .AddInMemoryCollection(new Dictionary { ["LocalDb:Path"] = path })
+ .Build();
+
+ return new ServiceCollection()
+ .AddZbLocalDb(config, LocalDbSetup.OnReady)
+ .BuildServiceProvider();
+ }
+
+ private IConfiguration ReplicationConfig(string apiKey, string? peerAddress)
+ {
+ var values = new Dictionary
+ {
+ // Tight flush + bounded reconnect backoff so convergence is observable well inside the
+ // poll deadline. The 60 s production default would let the doubling backoff overrun it
+ // after a peer outage.
+ ["LocalDb:Replication:FlushInterval"] = "00:00:00.050",
+ ["LocalDb:Replication:ReconnectBackoffMax"] = "00:00:02",
+ ["LocalDb:Replication:ApiKey"] = apiKey,
+ };
+
+ if (peerAddress is not null)
+ values["LocalDb:Replication:PeerAddress"] = peerAddress;
+
+ return new ConfigurationBuilder().AddInMemoryCollection(values).Build();
+ }
+
+ /// Starts node B, the passive listener, behind the real auth interceptor.
+ private async Task StartPassiveAsync()
+ {
+ var config = ReplicationConfig(_apiKeyB, peerAddress: null);
+
+ _serverHost = await new HostBuilder()
+ .ConfigureWebHost(web =>
+ {
+ web.UseKestrel(o =>
+ o.Listen(IPAddress.Loopback, 0, listen => listen.Protocols = HttpProtocols.Http2));
+ web.ConfigureServices(services =>
+ {
+ services.AddLogging();
+ services.AddRouting();
+ // The REAL interceptor, not a stand-in. If it rejected legitimate peer traffic,
+ // every matching-key scenario would fail — which is the point.
+ services.AddGrpc(o => o.Interceptors.Add());
+ services.AddSingleton(B);
+ services.AddZbLocalDbReplication(config);
+ })
+ .Configure(app =>
+ {
+ app.UseRouting();
+ app.UseEndpoints(e => e.MapZbLocalDbSync());
+ });
+ })
+ .StartAsync();
+ }
+
+ /// Starts node A, which dials the passive node.
+ private async Task StartInitiatorAsync()
+ {
+ var config = ReplicationConfig(_apiKeyA, PassiveAddress());
+
+ _initiatorHost = await new HostBuilder()
+ .ConfigureServices(services =>
+ {
+ services.AddLogging();
+ services.AddSingleton(A);
+ services.AddZbLocalDbReplication(config);
+ })
+ .StartAsync();
+ }
+
+ private string PassiveAddress()
+ => _serverHost!.Services.GetRequiredService()
+ .Features.Get()!.Addresses.Single();
+
+ private static async Task StopHostAsync(IHost? host)
+ {
+ if (host is null)
+ return;
+ try { await host.StopAsync(TimeSpan.FromSeconds(5)); } catch { /* teardown */ }
+ host.Dispose();
+ }
+
+ public async ValueTask DisposeAsync()
+ {
+ await StopHostAsync(_initiatorHost);
+ await StopHostAsync(_serverHost);
+ await _dbProviderA.DisposeAsync();
+ await _dbProviderB.DisposeAsync();
+
+ Microsoft.Data.Sqlite.SqliteConnection.ClearAllPools();
+ foreach (var path in new[] { _pathA, _pathB })
+ {
+ foreach (var suffix in new[] { "", "-wal", "-shm" })
+ {
+ try { File.Delete(path + suffix); } catch { /* best effort */ }
+ }
+ }
+ }
+}
diff --git a/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbReplicationHealthCheckTests.cs b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbReplicationHealthCheckTests.cs
new file mode 100644
index 00000000..6674bf3e
--- /dev/null
+++ b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbReplicationHealthCheckTests.cs
@@ -0,0 +1,122 @@
+using Microsoft.Extensions.Diagnostics.HealthChecks;
+using Microsoft.Extensions.Options;
+using Shouldly;
+using Xunit;
+using ZB.MOM.WW.LocalDb.Replication;
+using ZB.MOM.WW.OtOpcUa.Host.Health;
+
+namespace ZB.MOM.WW.OtOpcUa.Host.IntegrationTests;
+
+///
+/// LocalDb Phase 1 (Task 10) — the replication health check.
+///
+///
+/// The load-bearing case is the default-OFF one: a plain driver node with no peer and no
+/// listener must report Healthy, or every unreplicated node in the fleet degrades the
+/// moment this check ships. Beyond that, "connected" is not sufficient for healthy — an unknown
+/// backlog (a failed oplog poll) must not read as zero.
+///
+public sealed class LocalDbReplicationHealthCheckTests
+{
+ // ---- Pure decision core (table-tested) --------------------------------------------------
+
+ [Fact]
+ public void Unconfigured_IsHealthy_SoDefaultOffNodesDoNotDegrade()
+ {
+ LocalDbReplicationDecision.Evaluate(
+ peerConfigured: false, connected: false, oplogBacklog: null, degradedThreshold: 100_000)
+ .Status.ShouldBe(HealthStatus.Healthy);
+ }
+
+ [Fact]
+ public void ConfiguredButNotConnected_IsDegraded()
+ {
+ LocalDbReplicationDecision.Evaluate(
+ peerConfigured: true, connected: false, oplogBacklog: null, degradedThreshold: 100_000)
+ .Status.ShouldBe(HealthStatus.Degraded);
+ }
+
+ [Fact]
+ public void ConnectedButBacklogUnknown_IsDegraded()
+ {
+ // null backlog = the oplog poll failed. Reporting Healthy here would call a pair that cannot
+ // read its own oplog perfectly fine.
+ LocalDbReplicationDecision.Evaluate(
+ peerConfigured: true, connected: true, oplogBacklog: null, degradedThreshold: 100_000)
+ .Status.ShouldBe(HealthStatus.Degraded);
+ }
+
+ [Fact]
+ public void ConnectedWithSmallBacklog_IsHealthy()
+ {
+ LocalDbReplicationDecision.Evaluate(
+ peerConfigured: true, connected: true, oplogBacklog: 12, degradedThreshold: 100_000)
+ .Status.ShouldBe(HealthStatus.Healthy);
+ }
+
+ [Fact]
+ public void ConnectedWithBacklogOverThreshold_IsDegraded()
+ {
+ // A backlog past the threshold means the peer is falling behind faster than it drains —
+ // connected, but not keeping up.
+ LocalDbReplicationDecision.Evaluate(
+ peerConfigured: true, connected: true, oplogBacklog: 100_001, degradedThreshold: 100_000)
+ .Status.ShouldBe(HealthStatus.Degraded);
+ }
+
+ // ---- IHealthCheck wrapper ---------------------------------------------------------------
+
+ [Fact]
+ public async Task Wrapper_WithNoSyncStatusRegistered_IsHealthy()
+ {
+ // Admin-only graphs never register the replication engine. The check is registered
+ // unconditionally (AddOtOpcUaHealth takes no role argument), so it must no-op to Healthy
+ // rather than throw or degrade when LocalDb is simply absent.
+ var check = new LocalDbReplicationHealthCheck(
+ syncStatus: null,
+ options: Options.Create(new ReplicationOptions()),
+ syncListenPort: 0);
+
+ var result = await check.CheckHealthAsync(new HealthCheckContext(), TestContext.Current.CancellationToken);
+
+ result.Status.ShouldBe(HealthStatus.Healthy);
+ }
+
+ [Fact]
+ public async Task Wrapper_ListenerConfiguredButNotConnected_IsDegraded()
+ {
+ // A passive node (no PeerAddress) still counts as "replication configured" when it is
+ // listening — otherwise a passive node that never gets dialled would look healthy while
+ // silently accepting nothing.
+ var check = new LocalDbReplicationHealthCheck(
+ syncStatus: new StubSyncStatus { Connected = false },
+ options: Options.Create(new ReplicationOptions { PeerAddress = "" }),
+ syncListenPort: 9001);
+
+ var result = await check.CheckHealthAsync(new HealthCheckContext(), TestContext.Current.CancellationToken);
+
+ result.Status.ShouldBe(HealthStatus.Degraded);
+ }
+
+ [Fact]
+ public async Task Wrapper_PeerConfiguredAndConnectedAndDraining_IsHealthy()
+ {
+ var check = new LocalDbReplicationHealthCheck(
+ syncStatus: new StubSyncStatus { Connected = true, OplogBacklog = 3 },
+ options: Options.Create(new ReplicationOptions { PeerAddress = "https://peer:9001" }),
+ syncListenPort: 9001);
+
+ var result = await check.CheckHealthAsync(new HealthCheckContext(), TestContext.Current.CancellationToken);
+
+ result.Status.ShouldBe(HealthStatus.Healthy);
+ }
+
+ private sealed class StubSyncStatus : ISyncStatus
+ {
+ public bool Connected { get; init; }
+ public string? PeerNodeId { get; init; }
+ public DateTimeOffset? LastSyncUtc { get; init; }
+ public long? OplogBacklog { get; init; }
+ public long ConnectionAttempts { get; init; }
+ }
+}
diff --git a/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbSetupTests.cs b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbSetupTests.cs
new file mode 100644
index 00000000..2db014c2
--- /dev/null
+++ b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbSetupTests.cs
@@ -0,0 +1,128 @@
+using Microsoft.Data.Sqlite;
+using Microsoft.Extensions.Configuration;
+using Microsoft.Extensions.DependencyInjection;
+using Shouldly;
+using Xunit;
+using ZB.MOM.WW.LocalDb;
+using ZB.MOM.WW.OtOpcUa.Host.Configuration;
+
+namespace ZB.MOM.WW.OtOpcUa.Host.IntegrationTests;
+
+///
+/// LocalDb Phase 1 (Task 2) — pins the deployment-cache schema and the load-bearing
+/// DDL → RegisterReplicated → writes ordering inside .
+///
+///
+///
+/// The ordering assertion is the point of this file. RegisterReplicated installs the
+/// capture triggers; rows written before it are never captured into the oplog and
+/// therefore never reach the peer — silently, and forever. A reordered OnReady passes
+/// every schema-shape test while shipping a replication path that quietly moves nothing.
+///
+///
+/// Built against a real temp-file rather than a hand-written schema:
+/// the library has no in-memory mode, and asserting against a schema the test itself wrote
+/// would only prove the test agrees with itself.
+///
+///
+public sealed class LocalDbSetupTests : IDisposable
+{
+ private readonly string _dbPath =
+ Path.Combine(Path.GetTempPath(), $"otopcua-localdb-setup-{Guid.NewGuid():N}.db");
+
+ private ServiceProvider? _provider;
+
+ private ILocalDb BuildDb()
+ {
+ var configuration = new ConfigurationBuilder()
+ .AddInMemoryCollection(new Dictionary { ["LocalDb:Path"] = _dbPath })
+ .Build();
+
+ _provider = new ServiceCollection()
+ .AddZbLocalDb(configuration, LocalDbSetup.OnReady)
+ .BuildServiceProvider();
+
+ return _provider.GetRequiredService();
+ }
+
+ [Fact]
+ public void OnReady_RegistersExactlyTheTwoDeploymentTables()
+ {
+ // Both directions are load-bearing. "No fewer" catches a dropped RegisterReplicated call
+ // (that table then never replicates). "No more" catches an accidental registration —
+ // every replicated table costs three triggers and oplog volume on every write.
+ var db = BuildDb();
+
+ db.ReplicatedTables.Keys.OrderBy(k => k, StringComparer.Ordinal)
+ .ShouldBe(["deployment_artifacts", "deployment_pointer"]);
+ }
+
+ [Fact]
+ public void DeploymentArtifacts_PkIsDeploymentIdPlusChunkIndex()
+ {
+ // The composite PK is what makes an artifact chunkable. A single-column PK here would
+ // make every chunk of a deployment collide onto one row under LWW.
+ var db = BuildDb();
+
+ db.ReplicatedTables["deployment_artifacts"].PkColumns
+ .ShouldBe(["deployment_id", "chunk_index"]);
+ }
+
+ [Fact]
+ public void DeploymentPointer_PkIsClusterId()
+ {
+ // One current-deployment pointer per cluster: the pair's two nodes converge on the same
+ // row rather than each keeping its own.
+ var db = BuildDb();
+
+ db.ReplicatedTables["deployment_pointer"].PkColumns.ShouldBe(["cluster_id"]);
+ }
+
+ [Fact]
+ public async Task RowsWrittenAfterOnReady_EnterTheOplog()
+ {
+ // THE ordering assertion. If DDL and RegisterReplicated were swapped — or a write were
+ // added to OnReady ahead of registration — the triggers would not exist at write time and
+ // this count would stay 0 while every other test in this file still passed.
+ var db = BuildDb();
+
+ await db.ExecuteAsync(
+ """
+ INSERT INTO deployment_pointer
+ (cluster_id, deployment_id, revision_hash, artifact_sha256, applied_at_utc)
+ VALUES (@ClusterId, @DeploymentId, @RevisionHash, @Sha, @AppliedAtUtc)
+ """,
+ new
+ {
+ ClusterId = "SITE-A",
+ DeploymentId = "0123456789abcdef0123456789abcdef",
+ RevisionHash = new string('a', 64),
+ Sha = new string('b', 64),
+ AppliedAtUtc = "2026-07-20T00:00:00.0000000Z",
+ },
+ TestContext.Current.CancellationToken);
+
+ var oplogRows = await db.QueryAsync(
+ "SELECT COUNT(*) FROM __localdb_oplog",
+ r => r.GetInt32(0),
+ parameters: null,
+ TestContext.Current.CancellationToken);
+
+ oplogRows[0].ShouldBeGreaterThanOrEqualTo(1);
+ }
+
+ public void Dispose()
+ {
+ _provider?.Dispose();
+
+ // No in-memory mode, so these are real files. Pooled connections keep a handle open past
+ // dispose; clearing the pools first is what makes the delete actually succeed.
+ SqliteConnection.ClearAllPools();
+
+ foreach (var path in new[] { _dbPath, $"{_dbPath}-wal", $"{_dbPath}-shm" })
+ {
+ if (File.Exists(path))
+ File.Delete(path);
+ }
+ }
+}
diff --git a/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbSyncAuthInterceptorTests.cs b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbSyncAuthInterceptorTests.cs
new file mode 100644
index 00000000..9a506376
--- /dev/null
+++ b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbSyncAuthInterceptorTests.cs
@@ -0,0 +1,172 @@
+using Grpc.Core;
+using Microsoft.Extensions.Logging.Abstractions;
+using Microsoft.Extensions.Options;
+using Shouldly;
+using Xunit;
+using ZB.MOM.WW.LocalDb.Replication;
+using ZB.MOM.WW.OtOpcUa.Host.Configuration;
+
+namespace ZB.MOM.WW.OtOpcUa.Host.IntegrationTests;
+
+///
+/// LocalDb Phase 1 (Task 3) — the passive sync endpoint's inbound gate.
+///
+///
+/// The replication library's LocalDbSyncService verifies nothing; inbound auth is
+/// explicitly the host's job. Without this interceptor, anything able to reach a driver node's
+/// sync port could stream arbitrary rows straight into that node's deployment-artifact cache —
+/// which is exactly what the node boots from when central SQL is unreachable.
+///
+public sealed class LocalDbSyncAuthInterceptorTests
+{
+ private const string SyncMethod = "/localdb_sync.v1.LocalDbSync/Sync";
+ private const string OtherMethod = "/otopcua.SomethingElse/Call";
+
+ private static LocalDbSyncAuthInterceptor CreateInterceptor(string? apiKey)
+ => new(
+ Options.Create(new ReplicationOptions { ApiKey = apiKey }),
+ NullLogger.Instance);
+
+ private static ServerCallContext CreateContext(string method, string? authorizationHeader)
+ {
+ var headers = new Metadata();
+ if (authorizationHeader is not null)
+ headers.Add("authorization", authorizationHeader);
+
+ return new FakeServerCallContext(method, headers);
+ }
+
+ ///
+ /// Minimal carrying just a method name and request headers —
+ /// the only two things the interceptor reads.
+ ///
+ ///
+ /// Hand-rolled rather than using Grpc.Core.Testing.TestServerCallContext: that type
+ /// ships in the retired native Grpc.Core package and does not exist on the grpc-dotnet
+ /// stack this solution runs on.
+ ///
+ private sealed class FakeServerCallContext(string method, Metadata requestHeaders)
+ : ServerCallContext
+ {
+ protected override string MethodCore => method;
+ protected override string HostCore => "localhost";
+ protected override string PeerCore => "ipv4:127.0.0.1:12345";
+ protected override DateTime DeadlineCore => DateTime.UtcNow.AddMinutes(1);
+ protected override Metadata RequestHeadersCore => requestHeaders;
+ protected override CancellationToken CancellationTokenCore => CancellationToken.None;
+ protected override Metadata ResponseTrailersCore { get; } = [];
+ protected override Status StatusCore { get; set; }
+ protected override WriteOptions? WriteOptionsCore { get; set; }
+
+ protected override AuthContext AuthContextCore { get; } =
+ new(null, new Dictionary>());
+
+ protected override ContextPropagationToken CreatePropagationTokenCore(
+ ContextPropagationOptions? options)
+ => throw new NotSupportedException();
+
+ protected override Task WriteResponseHeadersAsyncCore(Metadata responseHeaders)
+ => Task.CompletedTask;
+ }
+
+ /// Invokes the interceptor's unary path with a trivial continuation.
+ private static Task Invoke(
+ LocalDbSyncAuthInterceptor interceptor, ServerCallContext context)
+ => interceptor.UnaryServerHandler(
+ "request", context, (_, _) => Task.FromResult("ok"));
+
+ [Fact]
+ public async Task NonSyncMethod_PassesThrough_EvenWithNoKeyConfigured()
+ {
+ // The interceptor is registered on the whole AddGrpc pipeline, so it sees every call.
+ // It must be scoped strictly to the sync service.
+ var interceptor = CreateInterceptor(apiKey: null);
+ var context = CreateContext(OtherMethod, authorizationHeader: null);
+
+ (await Invoke(interceptor, context)).ShouldBe("ok");
+ }
+
+ [Fact]
+ public async Task SyncMethod_WithNoKeyConfigured_IsDenied_EvenWithABearerToken()
+ {
+ // Fail-closed. "No key configured" is the DEFAULT every node ships with, so treating it
+ // as "no auth required" would silently expose the endpoint on exactly the configuration
+ // that is most common. Presenting a token must not help.
+ var interceptor = CreateInterceptor(apiKey: null);
+ var context = CreateContext(SyncMethod, "Bearer anything-at-all");
+
+ var ex = await Should.ThrowAsync(() => Invoke(interceptor, context));
+ ex.StatusCode.ShouldBe(StatusCode.PermissionDenied);
+ }
+
+ [Fact]
+ public async Task SyncMethod_WithNoBearerToken_IsDenied()
+ {
+ var interceptor = CreateInterceptor("the-shared-key");
+ var context = CreateContext(SyncMethod, authorizationHeader: null);
+
+ var ex = await Should.ThrowAsync(() => Invoke(interceptor, context));
+ ex.StatusCode.ShouldBe(StatusCode.PermissionDenied);
+ }
+
+ [Fact]
+ public async Task SyncMethod_WithWrongBearerToken_IsDenied()
+ {
+ var interceptor = CreateInterceptor("the-shared-key");
+ var context = CreateContext(SyncMethod, "Bearer the-wrong-key");
+
+ var ex = await Should.ThrowAsync(() => Invoke(interceptor, context));
+ ex.StatusCode.ShouldBe(StatusCode.PermissionDenied);
+ }
+
+ [Fact]
+ public async Task SyncMethod_WithCorrectBearerToken_PassesThrough()
+ {
+ var interceptor = CreateInterceptor("the-shared-key");
+ var context = CreateContext(SyncMethod, "Bearer the-shared-key");
+
+ (await Invoke(interceptor, context)).ShouldBe("ok");
+ }
+
+ [Fact]
+ public async Task SyncMethod_WithCorrectKey_ButNoBearerScheme_IsDenied()
+ {
+ // A raw key with no "Bearer " prefix is not what SyncBackgroundService sends, and
+ // accepting it would widen the accepted credential shape for no reason.
+ var interceptor = CreateInterceptor("the-shared-key");
+ var context = CreateContext(SyncMethod, "the-shared-key");
+
+ var ex = await Should.ThrowAsync(() => Invoke(interceptor, context));
+ ex.StatusCode.ShouldBe(StatusCode.PermissionDenied);
+ }
+
+ [Fact]
+ public async Task SyncMethod_TokenComparison_IsNotAPrefixMatch()
+ {
+ // A prefix/StartsWith comparison would accept a truncated key and make the secret
+ // recoverable one character at a time.
+ var interceptor = CreateInterceptor("the-shared-key");
+ var context = CreateContext(SyncMethod, "Bearer the-shared-ke");
+
+ var ex = await Should.ThrowAsync(() => Invoke(interceptor, context));
+ ex.StatusCode.ShouldBe(StatusCode.PermissionDenied);
+ }
+
+ [Fact]
+ public async Task DuplexStreaming_IsGated_BecauseThatIsHowSyncActuallyRuns()
+ {
+ // The sync RPC is a bidirectional stream. Gating only the unary path would leave the real
+ // endpoint wide open while every unary test still passed.
+ var interceptor = CreateInterceptor("the-shared-key");
+ var context = CreateContext(SyncMethod, "Bearer the-wrong-key");
+
+ var ex = await Should.ThrowAsync(() =>
+ interceptor.DuplexStreamingServerHandler(
+ requestStream: null!,
+ responseStream: null!,
+ context,
+ (_, _, _) => Task.CompletedTask));
+
+ ex.StatusCode.ShouldBe(StatusCode.PermissionDenied);
+ }
+}
diff --git a/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbSyncListenerTests.cs b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbSyncListenerTests.cs
new file mode 100644
index 00000000..58e25082
--- /dev/null
+++ b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbSyncListenerTests.cs
@@ -0,0 +1,234 @@
+using System.Net;
+using Microsoft.AspNetCore.Builder;
+using Microsoft.AspNetCore.Server.Kestrel.Core;
+using Microsoft.Extensions.DependencyInjection;
+using Microsoft.Extensions.Hosting;
+using Shouldly;
+using Xunit;
+using ZB.MOM.WW.OtOpcUa.Host.Configuration;
+
+namespace ZB.MOM.WW.OtOpcUa.Host.IntegrationTests;
+
+///
+/// LocalDb Phase 1 (Task 5) — the dedicated h2c sync listener and, more importantly, the
+/// re-binding of the endpoints the host was already serving.
+///
+///
+///
+/// Why this file exists. The host binds exclusively via ASPNETCORE_URLS and has
+/// no ConfigureKestrel call. Any explicit Listen* makes Kestrel discard that
+/// configuration wholesale — it logs "Overriding address(es)" and serves only what was
+/// listed explicitly. Getting this wrong unbinds the AdminUI and the deploy API behind
+/// Traefik, and it fails silently: the process starts, logs look normal, and nothing answers.
+///
+///
+/// These tests drive a minimal WebApplication shaped exactly like Program.cs's
+/// block rather than booting the real host, which needs SQL Server, an Akka mesh and LDAP.
+/// What is under test is the Kestrel binding contract, and that is fully reproduced here.
+///
+///
+public sealed class LocalDbSyncListenerTests
+{
+ // ---------------------------------------------------------------------------------------
+ // KestrelHttpBinding.Parse
+ // ---------------------------------------------------------------------------------------
+
+ [Fact]
+ public void Parse_NullOrEmpty_YieldsNoBindings()
+ {
+ // "Nothing configured" is a real supported state: Install-Services.ps1 sets ASPNETCORE_URLS
+ // only for admin nodes, so a driver-only service genuinely has none.
+ KestrelHttpBinding.Parse(null).ShouldBeEmpty();
+ KestrelHttpBinding.Parse("").ShouldBeEmpty();
+ KestrelHttpBinding.Parse(" ").ShouldBeEmpty();
+ }
+
+ [Theory]
+ [InlineData("http://+:9000", "+", 9000)]
+ [InlineData("http://*:9000", "*", 9000)]
+ [InlineData("http://localhost:9000", "localhost", 9000)]
+ [InlineData("http://0.0.0.0:8080", "0.0.0.0", 8080)]
+ [InlineData("http://127.0.0.1:5001", "127.0.0.1", 5001)]
+ public void Parse_SingleUrl_ExtractsHostAndPort(string url, string expectedHost, int expectedPort)
+ {
+ // The "+" and "*" wildcards are Kestrel-isms that System.Uri cannot parse — the docker-dev
+ // rig uses "http://+:9000", so mishandling them would break every rig node.
+ var bindings = KestrelHttpBinding.Parse(url);
+
+ bindings.Count.ShouldBe(1);
+ bindings[0].Host.ShouldBe(expectedHost);
+ bindings[0].Port.ShouldBe(expectedPort);
+ bindings[0].IsSecure.ShouldBeFalse();
+ }
+
+ [Fact]
+ public void Parse_MultipleUrls_PreservesAllOfThem()
+ {
+ // Dropping one of several configured endpoints is the exact silent-unbind failure this
+ // whole mechanism exists to avoid.
+ var bindings = KestrelHttpBinding.Parse("http://+:9000;http://localhost:9100");
+
+ bindings.Count.ShouldBe(2);
+ bindings[0].Port.ShouldBe(9000);
+ bindings[1].Port.ShouldBe(9100);
+ }
+
+ [Fact]
+ public void Parse_HttpsUrl_IsFlaggedSecure()
+ {
+ // Program.cs refuses to take over Kestrel when any endpoint is HTTPS, because replaying
+ // certificate configuration is not modelled here. This flag is what drives that refusal.
+ KestrelHttpBinding.Parse("https://+:443")[0].IsSecure.ShouldBeTrue();
+ }
+
+ [Fact]
+ public void Parse_MalformedEntry_IsSkipped_NotThrown()
+ {
+ // A typo'd URL must not take the process down at startup.
+ KestrelHttpBinding.Parse("not a url;http://+:9000").ShouldHaveSingleItem().Port.ShouldBe(9000);
+ }
+
+ // ---- ParsePorts (ASPNETCORE_HTTP_PORTS / HTTPS_PORTS) ----------------------------------
+
+ [Fact]
+ public void ParsePorts_NullOrEmpty_YieldsNoBindings()
+ {
+ KestrelHttpBinding.ParsePorts(null, isSecure: false).ShouldBeEmpty();
+ KestrelHttpBinding.ParsePorts("", isSecure: false).ShouldBeEmpty();
+ KestrelHttpBinding.ParsePorts(" ", isSecure: false).ShouldBeEmpty();
+ }
+
+ [Fact]
+ public void ParsePorts_BarePort_BindsAllInterfaces()
+ {
+ // The aspnet:8.0+ container default is ASPNETCORE_HTTP_PORTS=8080, on all interfaces — not
+ // loopback. Re-binding it as a wildcard is what keeps a driver node's health/metrics surface
+ // reachable from other containers once the sync listener takes over Kestrel.
+ var binding = KestrelHttpBinding.ParsePorts("8080", isSecure: false).ShouldHaveSingleItem();
+ binding.Host.ShouldBe("+");
+ binding.Port.ShouldBe(8080);
+ binding.IsSecure.ShouldBeFalse();
+ }
+
+ [Fact]
+ public void ParsePorts_SeparatedList_AndSecureFlag()
+ {
+ KestrelHttpBinding.ParsePorts("8080;8081", isSecure: false).Select(b => b.Port).ShouldBe([8080, 8081]);
+ KestrelHttpBinding.ParsePorts("8080,8081", isSecure: false).Select(b => b.Port).ShouldBe([8080, 8081]);
+ KestrelHttpBinding.ParsePorts("443", isSecure: true).ShouldHaveSingleItem().IsSecure.ShouldBeTrue();
+ }
+
+ [Fact]
+ public void ParsePorts_MalformedPort_IsSkipped_NotThrown()
+ {
+ KestrelHttpBinding.ParsePorts("notaport;8080;0;70000", isSecure: false)
+ .ShouldHaveSingleItem().Port.ShouldBe(8080);
+ }
+
+ // ---------------------------------------------------------------------------------------
+ // The real Kestrel contract
+ // ---------------------------------------------------------------------------------------
+
+ [Fact]
+ public async Task ExplicitListen_ReBindsTheExistingSurface_AndAddsAnH2cListener()
+ {
+ // THE test. Both ports must answer simultaneously: the re-bound HTTP/1.1 surface (proving
+ // the ASPNETCORE_URLS override was compensated for) and the HTTP/2-only sync port (proving
+ // prior-knowledge h2c works, which a cleartext Http1AndHttp2 endpoint cannot do).
+ var httpPort = GetFreePort();
+ var syncPort = GetFreePort();
+
+ var builder = WebApplication.CreateBuilder();
+ builder.Environment.ApplicationName = typeof(LocalDbSyncListenerTests).Assembly.GetName().Name!;
+ builder.Services.AddGrpc();
+
+ // Exactly the shape Program.cs applies.
+ foreach (var binding in KestrelHttpBinding.Parse($"http://localhost:{httpPort}"))
+ {
+ var captured = binding;
+ builder.WebHost.ConfigureKestrel(k => captured.Apply(k));
+ }
+
+ builder.WebHost.ConfigureKestrel(k =>
+ k.ListenAnyIP(syncPort, o => o.Protocols = HttpProtocols.Http2));
+
+ await using var app = builder.Build();
+ app.MapGet("/healthz", () => "ok");
+ await app.StartAsync(TestContext.Current.CancellationToken);
+
+ try
+ {
+ using var http1 = new HttpClient();
+ var body = await http1.GetStringAsync(
+ $"http://localhost:{httpPort}/healthz", TestContext.Current.CancellationToken);
+ body.ShouldBe("ok");
+
+ // Prior-knowledge h2c against the sync port. A 404 is a perfectly good result — it
+ // proves the HTTP/2 connection was established and the request was routed, which is
+ // the only thing in question here.
+ using var http2 = new HttpClient
+ {
+ DefaultRequestVersion = HttpVersion.Version20,
+ DefaultVersionPolicy = HttpVersionPolicy.RequestVersionExact,
+ };
+ using var response = await http2.GetAsync(
+ $"http://localhost:{syncPort}/", TestContext.Current.CancellationToken);
+
+ response.Version.ShouldBe(HttpVersion.Version20);
+ }
+ finally
+ {
+ await app.StopAsync(TestContext.Current.CancellationToken);
+ }
+ }
+
+ [Fact]
+ public async Task ExplicitListen_WithoutReBinding_SilentlyDiscardsTheConfiguredUrls()
+ {
+ // POSITIVE CONTROL for the whole task. If Kestrel did NOT override configured URLs, the
+ // re-binding above would be pointless ceremony. This pins the actual behaviour: a single
+ // explicit Listen* call makes the ASPNETCORE_URLS port stop answering entirely — no
+ // exception, no failed startup, just silence. That is the regression being guarded against.
+ var configuredPort = GetFreePort();
+ var explicitPort = GetFreePort();
+
+ var builder = WebApplication.CreateBuilder();
+ builder.Environment.ApplicationName = typeof(LocalDbSyncListenerTests).Assembly.GetName().Name!;
+ builder.WebHost.UseUrls($"http://localhost:{configuredPort}");
+
+ // Deliberately NOT re-binding configuredPort — this is the mistake being demonstrated.
+ builder.WebHost.ConfigureKestrel(k => k.ListenAnyIP(explicitPort));
+
+ await using var app = builder.Build();
+ app.MapGet("/healthz", () => "ok");
+ await app.StartAsync(TestContext.Current.CancellationToken);
+
+ try
+ {
+ using var client = new HttpClient { Timeout = TimeSpan.FromSeconds(5) };
+
+ // The explicitly listed port serves.
+ (await client.GetStringAsync(
+ $"http://localhost:{explicitPort}/healthz",
+ TestContext.Current.CancellationToken)).ShouldBe("ok");
+
+ // The configured one does not — nothing is listening there at all.
+ await Should.ThrowAsync(() => client.GetStringAsync(
+ $"http://localhost:{configuredPort}/healthz",
+ TestContext.Current.CancellationToken));
+ }
+ finally
+ {
+ await app.StopAsync(TestContext.Current.CancellationToken);
+ }
+ }
+
+ private static int GetFreePort()
+ {
+ using var listener = new System.Net.Sockets.TcpListener(IPAddress.Loopback, 0);
+ listener.Start();
+ var port = ((IPEndPoint)listener.LocalEndpoint).Port;
+ listener.Stop();
+ return port;
+ }
+}
diff --git a/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbWiringTests.cs b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbWiringTests.cs
new file mode 100644
index 00000000..77060548
--- /dev/null
+++ b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbWiringTests.cs
@@ -0,0 +1,172 @@
+using Microsoft.AspNetCore.Builder;
+using Microsoft.Data.Sqlite;
+using Microsoft.Extensions.Configuration;
+using Microsoft.Extensions.DependencyInjection;
+using Microsoft.Extensions.Diagnostics.HealthChecks;
+using Microsoft.Extensions.Options;
+using Shouldly;
+using Xunit;
+using ZB.MOM.WW.LocalDb;
+using ZB.MOM.WW.LocalDb.Replication;
+using ZB.MOM.WW.OtOpcUa.Host.Configuration;
+using ZB.MOM.WW.OtOpcUa.Host.Health;
+using ZB.MOM.WW.OtOpcUa.Host.Observability;
+using ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache;
+
+namespace ZB.MOM.WW.OtOpcUa.Host.IntegrationTests;
+
+///
+/// LocalDb Phase 1 (Task 11) — DI pins over the real composition methods.
+///
+///
+///
+/// DI extension methods have shipped inert three times in this family (Secrets 0.2.0 / 0.2.2,
+/// ScadaBridge #22): a registration that compiles, deploys, and does nothing because the seam
+/// was never resolved from a built container. These tests build a real container from the
+/// same AddOtOpcUa* methods Program.cs calls and resolve through it.
+///
+///
+/// They deliberately do NOT boot the full host. WebApplicationFactory<Program>
+/// would start hosted services — Akka, the OPC UA server, ValidateOnStart — which need
+/// SQL Server and an Akka mesh (see TwoNodeClusterHarness on why the real host is not
+/// driven here). What is under test is DI resolution, so the container is built and resolved
+/// without ever starting it.
+///
+///
+public sealed class LocalDbWiringTests : IDisposable
+{
+ private readonly List _dbPaths = [];
+ private readonly List _apps = [];
+
+ ///
+ /// Builds a container the way Program.cs's driver branch does for LocalDb: the real
+ /// , health, and observability methods,
+ /// over a temp-file LocalDb:Path.
+ ///
+ private IServiceProvider BuildDriverGraph()
+ {
+ var dbPath = Path.Combine(Path.GetTempPath(), $"otopcua-wiring-{Guid.NewGuid():N}.db");
+ _dbPaths.Add(dbPath);
+
+ var builder = WebApplication.CreateBuilder(new WebApplicationOptions { Args = [] });
+ builder.Configuration.AddInMemoryCollection(new Dictionary
+ {
+ ["LocalDb:Path"] = dbPath,
+ });
+
+ // The exact LocalDb composition Program.cs runs under hasDriver.
+ builder.Services.AddOtOpcUaLocalDb(builder.Configuration);
+ builder.Services.AddGrpc(o => o.Interceptors.Add());
+ builder.Services.AddOtOpcUaHealth();
+ builder.Services.AddOtOpcUaObservability(builder.Configuration);
+
+ var app = builder.Build();
+ _apps.Add(app);
+ return app.Services;
+ }
+
+ ///
+ /// Builds an admin-only container: the shared registrations that run on every node, but NOT
+ /// — exactly as Program.cs gates it under
+ /// hasDriver.
+ ///
+ private IServiceProvider BuildAdminOnlyGraph()
+ {
+ var builder = WebApplication.CreateBuilder(new WebApplicationOptions { Args = [] });
+
+ // No LocalDb:Path configured — an admin-only node has no reason to set it.
+ builder.Services.AddOtOpcUaHealth();
+ builder.Services.AddOtOpcUaObservability(builder.Configuration);
+
+ var app = builder.Build();
+ _apps.Add(app);
+ return app.Services;
+ }
+
+ [Fact]
+ public void DriverGraph_LocalDbResolvesAsASingletonWithBothCacheTablesRegistered()
+ {
+ var sp = BuildDriverGraph();
+
+ var first = sp.GetService();
+ first.ShouldNotBeNull();
+ sp.GetService().ShouldBeSameAs(first); // singleton
+
+ // Exact set, both directions — a dropped registration replicates nothing; an extra one costs
+ // three triggers and oplog volume on every write.
+ first.ReplicatedTables.Keys.OrderBy(k => k, StringComparer.Ordinal)
+ .ShouldBe(["deployment_artifacts", "deployment_pointer"]);
+ }
+
+ [Fact]
+ public void DriverGraph_ArtifactCacheResolvesToTheLocalDbBackedImplementation()
+ {
+ var sp = BuildDriverGraph();
+
+ sp.GetService()
+ .ShouldBeOfType();
+ }
+
+ [Fact]
+ public void DriverGraph_DefaultOff_SyncStatusReportsNotConnectedAndNoPeer()
+ {
+ // The default-OFF posture pin: registering the replication engine must not connect it.
+ var sp = BuildDriverGraph();
+
+ var status = sp.GetService();
+ status.ShouldNotBeNull();
+ status.Connected.ShouldBeFalse();
+ status.PeerNodeId.ShouldBeNull();
+ }
+
+ [Fact]
+ public void DriverGraph_LocalDbReplicationHealthCheckIsRegistered()
+ {
+ var sp = BuildDriverGraph();
+
+ var registrations = sp.GetRequiredService>()
+ .Value.Registrations
+ .Select(r => r.Name)
+ .ToArray();
+
+ registrations.ShouldContain("localdb-replication");
+ }
+
+ [Fact]
+ public void AdminOnlyGraph_HasNoLocalDb_AndDoesNotDemandLocalDbPath()
+ {
+ // Boot must not require LocalDb:Path on a node that never registers LocalDb. Building the
+ // container without a configured path and resolving the shared services proves nothing in
+ // the always-on registrations reaches for ILocalDb or its options.
+ var sp = BuildAdminOnlyGraph();
+
+ sp.GetService().ShouldBeNull();
+ sp.GetService().ShouldBeNull();
+
+ // The unconditionally-registered health check must still resolve and construct — it is meant
+ // to no-op to Healthy when LocalDb is absent, not to fail because ISyncStatus is missing.
+ var registrations = sp.GetRequiredService>()
+ .Value.Registrations
+ .Where(r => r.Name == "localdb-replication")
+ .ToArray();
+ var localDbCheck = registrations.ShouldHaveSingleItem();
+ Should.NotThrow(() => localDbCheck.Factory(sp));
+ }
+
+ public void Dispose()
+ {
+ foreach (var app in _apps)
+ ((IDisposable)app).Dispose();
+
+ SqliteConnection.ClearAllPools();
+
+ foreach (var basePath in _dbPaths)
+ {
+ foreach (var path in new[] { basePath, $"{basePath}-wal", $"{basePath}-shm" })
+ {
+ if (File.Exists(path))
+ File.Delete(path);
+ }
+ }
+ }
+}
diff --git a/tests/Server/ZB.MOM.WW.OtOpcUa.Runtime.Tests/Deployment/LocalDbDeploymentArtifactCacheTests.cs b/tests/Server/ZB.MOM.WW.OtOpcUa.Runtime.Tests/Deployment/LocalDbDeploymentArtifactCacheTests.cs
new file mode 100644
index 00000000..86bb8154
--- /dev/null
+++ b/tests/Server/ZB.MOM.WW.OtOpcUa.Runtime.Tests/Deployment/LocalDbDeploymentArtifactCacheTests.cs
@@ -0,0 +1,291 @@
+using Microsoft.Data.Sqlite;
+using Microsoft.Extensions.Configuration;
+using Microsoft.Extensions.DependencyInjection;
+using Microsoft.Extensions.Logging;
+using Microsoft.Extensions.Logging.Abstractions;
+using Shouldly;
+using Xunit;
+using ZB.MOM.WW.LocalDb;
+using ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache;
+
+// NOT "...Tests.Deployment", even though the folder is Deployment/. A child namespace named
+// Deployment under ...Runtime.Tests shadows the Deployment EF entity type for every other file in
+// this assembly — the same CS0118 trap that forced the production namespace's rename.
+namespace ZB.MOM.WW.OtOpcUa.Runtime.Tests.DeploymentCache;
+
+///
+/// LocalDb Phase 1 (Task 6) — the chunked deployment-artifact cache.
+///
+///
+///
+/// Driven against a real temp-file rather than a fake. The behaviours
+/// under test — chunk reassembly order, SHA-256 integrity, retention pruning, upsert
+/// semantics — live entirely in SQL, so a mocked seam would assert nothing about them.
+///
+///
+/// This project cannot reference the Host, so the schema setup here mirrors
+/// LocalDbSetup.OnReady rather than calling it. The load-bearing
+/// DDL → RegisterReplicated ordering is pinned by the Host's own
+/// LocalDbSetupTests; what matters here is only that the tables exist.
+///
+///
+public sealed class LocalDbDeploymentArtifactCacheTests : IDisposable
+{
+ private const string ClusterA = "SITE-A";
+ private const string ClusterB = "SITE-B";
+
+ private readonly string _dbPath =
+ Path.Combine(Path.GetTempPath(), $"otopcua-artifact-cache-{Guid.NewGuid():N}.db");
+
+ private readonly ServiceProvider _provider;
+ private readonly ILocalDb _db;
+
+ public LocalDbDeploymentArtifactCacheTests()
+ {
+ var configuration = new ConfigurationBuilder()
+ .AddInMemoryCollection(new Dictionary { ["LocalDb:Path"] = _dbPath })
+ .Build();
+
+ _provider = new ServiceCollection()
+ .AddZbLocalDb(configuration, db =>
+ {
+ using (var connection = db.CreateConnection())
+ {
+ DeploymentCacheSchema.Apply(connection);
+ }
+
+ db.RegisterReplicated(DeploymentCacheSchema.ArtifactsTable);
+ db.RegisterReplicated(DeploymentCacheSchema.PointerTable);
+ })
+ .BuildServiceProvider();
+
+ _db = _provider.GetRequiredService();
+ }
+
+ private LocalDbDeploymentArtifactCache CreateCache(
+ ILogger? logger = null)
+ => new(_db, logger ?? NullLogger.Instance);
+
+ private static byte[] RandomArtifact(int length)
+ {
+ var bytes = new byte[length];
+ Random.Shared.NextBytes(bytes);
+ return bytes;
+ }
+
+ private async Task CountChunksAsync(string? deploymentId = null)
+ {
+ var sql = deploymentId is null
+ ? "SELECT COUNT(*) FROM deployment_artifacts"
+ : "SELECT COUNT(*) FROM deployment_artifacts WHERE deployment_id = @DeploymentId";
+
+ var rows = await _db.QueryAsync(sql, r => r.GetInt32(0),
+ deploymentId is null ? null : new { DeploymentId = deploymentId });
+
+ return rows[0];
+ }
+
+ private async Task> DistinctDeploymentIdsAsync()
+ => await _db.QueryAsync(
+ "SELECT DISTINCT deployment_id FROM deployment_artifacts ORDER BY deployment_id",
+ r => r.GetString(0));
+
+ [Fact]
+ public async Task StoreThenGetCurrent_RoundTripsAMultiChunkArtifactByteForByte()
+ {
+ // 300 KiB against a 128 KiB chunk forces three chunks, so this covers the reassembly
+ // ordering that a single-chunk artifact would never exercise.
+ var artifact = RandomArtifact(300 * 1024);
+ var cache = CreateCache();
+
+ await cache.StoreAsync(ClusterA, "dep-1", "rev-1", artifact);
+
+ var cached = await cache.GetCurrentAsync(ClusterA);
+
+ cached.ShouldNotBeNull();
+ cached.DeploymentId.ShouldBe("dep-1");
+ cached.RevisionHash.ShouldBe("rev-1");
+ cached.Artifact.ShouldBe(artifact);
+ (await CountChunksAsync("dep-1")).ShouldBe(3);
+ }
+
+ [Fact]
+ public async Task Store_KeepsOnlyTheTwoNewestDeploymentsPerCluster()
+ {
+ var cache = CreateCache();
+
+ await cache.StoreAsync(ClusterA, "dep-1", "rev-1", RandomArtifact(64));
+ await cache.StoreAsync(ClusterA, "dep-2", "rev-2", RandomArtifact(64));
+ await cache.StoreAsync(ClusterA, "dep-3", "rev-3", RandomArtifact(64));
+
+ // Bounded retention is what stops a node that redeploys daily from filling its disk with
+ // address spaces nobody will ever roll back to.
+ (await DistinctDeploymentIdsAsync()).ShouldBe(["dep-2", "dep-3"]);
+
+ var cached = await cache.GetCurrentAsync(ClusterA);
+ cached.ShouldNotBeNull();
+ cached.DeploymentId.ShouldBe("dep-3");
+ }
+
+ [Fact]
+ public async Task Store_NeverPrunesTheDeploymentThePointerNames()
+ {
+ // Regression: the prune ranks survivors by cached_at_utc and falls through to a
+ // deployment_id tiebreak. Real deployment ids are GUIDs, so that tiebreak is effectively
+ // random — the deployment just written could lose and have its chunks deleted while the
+ // pointer still named it. The result is a cache that misses on every boot until the next
+ // deploy: silent, permanent, and precisely the outage this cache exists to survive.
+ //
+ // Rather than trying to race three stores into one clock tick, this forces the general
+ // case: the two older deployments are back-dated INTO THE FUTURE so the newest store loses
+ // the ranking outright. The invariant must hold regardless of timestamps.
+ var cache = CreateCache();
+ var newest = RandomArtifact(1024);
+
+ await cache.StoreAsync(ClusterA, "dep-old-1", "rev-1", RandomArtifact(1024));
+ await cache.StoreAsync(ClusterA, "dep-old-2", "rev-2", RandomArtifact(1024));
+
+ await _db.ExecuteAsync(
+ "UPDATE deployment_artifacts SET cached_at_utc = @Future",
+ new { Future = "2099-01-01T00:00:00.0000000Z" });
+
+ await cache.StoreAsync(ClusterA, "dep-new", "rev-3", newest);
+
+ // The pointer names dep-new, so dep-new's chunks must still be there...
+ (await CountChunksAsync("dep-new")).ShouldBeGreaterThan(0);
+
+ // ...and it must actually read back, which is the behaviour that matters.
+ var cached = await cache.GetCurrentAsync(ClusterA);
+ cached.ShouldNotBeNull();
+ cached.DeploymentId.ShouldBe("dep-new");
+ cached.Artifact.ShouldBe(newest);
+ }
+
+ [Fact]
+ public async Task GetCurrent_ReturnsNullWhenAChunkIsCorrupt()
+ {
+ var cache = CreateCache();
+ await cache.StoreAsync(ClusterA, "dep-1", "rev-1", RandomArtifact(300 * 1024));
+
+ // Valid base64 of the wrong bytes: this passes decoding and the chunk-count check, so only
+ // the SHA-256 comparison can catch it. That is the check being pinned.
+ await _db.ExecuteAsync(
+ "UPDATE deployment_artifacts SET chunk_base64 = @Chunk WHERE deployment_id = @DeploymentId AND chunk_index = 1",
+ new { Chunk = Convert.ToBase64String(RandomArtifact(128 * 1024)), DeploymentId = "dep-1" });
+
+ (await cache.GetCurrentAsync(ClusterA)).ShouldBeNull();
+ }
+
+ [Fact]
+ public async Task GetCurrent_ReturnsNullWhenAChunkIsMissing()
+ {
+ var cache = CreateCache();
+ await cache.StoreAsync(ClusterA, "dep-1", "rev-1", RandomArtifact(300 * 1024));
+
+ // A partially replicated artifact is the realistic version of this: the pointer row arrives
+ // before the last chunk does. Reassembling what is present would yield a plausible-looking
+ // but silently truncated address space.
+ await _db.ExecuteAsync(
+ "DELETE FROM deployment_artifacts WHERE deployment_id = @DeploymentId AND chunk_index = 2",
+ new { DeploymentId = "dep-1" });
+
+ (await cache.GetCurrentAsync(ClusterA)).ShouldBeNull();
+ }
+
+ [Fact]
+ public async Task GetCurrent_ReturnsNullWhenNoPointerExists()
+ {
+ var cache = CreateCache();
+
+ (await cache.GetCurrentAsync(ClusterA)).ShouldBeNull();
+ }
+
+ [Fact]
+ public async Task Store_IsIdempotentForTheSameDeploymentId()
+ {
+ var cache = CreateCache();
+
+ await cache.StoreAsync(ClusterA, "dep-1", "rev-1", RandomArtifact(300 * 1024));
+
+ // The re-store is smaller, so a delete-before-insert that did not happen would leave the
+ // tail chunks of the first artifact behind — orphans that also break the chunk-count check.
+ var second = RandomArtifact(200 * 1024);
+ await cache.StoreAsync(ClusterA, "dep-1", "rev-1b", second);
+
+ (await CountChunksAsync("dep-1")).ShouldBe(2);
+ (await CountChunksAsync()).ShouldBe(2);
+
+ var cached = await cache.GetCurrentAsync(ClusterA);
+ cached.ShouldNotBeNull();
+ cached.RevisionHash.ShouldBe("rev-1b");
+ cached.Artifact.ShouldBe(second);
+ }
+
+ [Fact]
+ public async Task GetCurrentUnkeyed_ReturnsTheOnlyCachedArtifact()
+ {
+ var artifact = RandomArtifact(300 * 1024);
+ var cache = CreateCache();
+
+ await cache.StoreAsync(ClusterA, "dep-1", "rev-1", artifact);
+
+ var cached = await cache.GetCurrentUnkeyedAsync();
+
+ cached.ShouldNotBeNull();
+ cached.DeploymentId.ShouldBe("dep-1");
+ cached.Artifact.ShouldBe(artifact);
+ }
+
+ [Fact]
+ public async Task GetCurrentUnkeyed_TakesTheNewestPointerAndWarnsNamingBothClusters()
+ {
+ var newest = RandomArtifact(1024);
+ var logger = new CapturingLogger();
+ var cache = CreateCache(logger);
+
+ await cache.StoreAsync(ClusterA, "dep-a", "rev-a", RandomArtifact(1024));
+ await cache.StoreAsync(ClusterB, "dep-b", "rev-b", newest);
+
+ var cached = await cache.GetCurrentUnkeyedAsync();
+
+ cached.ShouldNotBeNull();
+ cached.DeploymentId.ShouldBe("dep-b");
+ cached.Artifact.ShouldBe(newest);
+
+ // A re-homed node booting a neighbouring cluster's configuration must be operator-visible.
+ // Naming both clusters is what turns "wrong tags appeared" into a one-line diagnosis.
+ var warning = logger.Entries.ShouldHaveSingleItem();
+ warning.Level.ShouldBe(LogLevel.Warning);
+ warning.Message.ShouldContain(ClusterA);
+ warning.Message.ShouldContain(ClusterB);
+ }
+
+ public void Dispose()
+ {
+ _provider.Dispose();
+
+ // Real files, and pooled connections outlive the provider — clearing the pools is what
+ // makes the delete actually succeed.
+ SqliteConnection.ClearAllPools();
+
+ foreach (var path in new[] { _dbPath, $"{_dbPath}-wal", $"{_dbPath}-shm" })
+ {
+ if (File.Exists(path))
+ File.Delete(path);
+ }
+ }
+
+ /// Minimal logger that records formatted messages so a test can assert on them.
+ private sealed class CapturingLogger : ILogger
+ {
+ public List<(LogLevel Level, string Message)> Entries { get; } = [];
+
+ public IDisposable? BeginScope(TState state) where TState : notnull => null;
+
+ public bool IsEnabled(LogLevel logLevel) => true;
+
+ public void Log(LogLevel logLevel, EventId eventId, TState state, Exception? exception,
+ Func formatter)
+ => Entries.Add((logLevel, formatter(state, exception)));
+ }
+}
diff --git a/tests/Server/ZB.MOM.WW.OtOpcUa.Runtime.Tests/Drivers/DriverHostActorArtifactCacheTests.cs b/tests/Server/ZB.MOM.WW.OtOpcUa.Runtime.Tests/Drivers/DriverHostActorArtifactCacheTests.cs
new file mode 100644
index 00000000..77f9972b
--- /dev/null
+++ b/tests/Server/ZB.MOM.WW.OtOpcUa.Runtime.Tests/Drivers/DriverHostActorArtifactCacheTests.cs
@@ -0,0 +1,273 @@
+using System.Text.Json;
+using Akka.Actor;
+using Microsoft.EntityFrameworkCore;
+using Shouldly;
+using Xunit;
+using ZB.MOM.WW.OtOpcUa.Commons.Messages.Deploy;
+using ZB.MOM.WW.OtOpcUa.Commons.Types;
+using ZB.MOM.WW.OtOpcUa.Configuration;
+using ZB.MOM.WW.OtOpcUa.Configuration.Entities;
+using ZB.MOM.WW.OtOpcUa.Configuration.Enums;
+using ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache;
+using ZB.MOM.WW.OtOpcUa.Runtime.Drivers;
+using ZB.MOM.WW.OtOpcUa.Runtime.Tests.Harness;
+
+namespace ZB.MOM.WW.OtOpcUa.Runtime.Tests.Drivers;
+
+///
+/// LocalDb Phase 1 (Task 7) — the write half of the deployment-artifact cache.
+///
+///
+/// What matters here is not that a store happens, but the conditions under which it must NOT:
+/// a failed apply, and — less obviously — a "successful" apply whose artifact never actually
+/// loaded. See ReconcileDrivers: it swallows its own DB failures, so apply success does
+/// not imply a real configuration was applied.
+///
+public sealed class DriverHostActorArtifactCacheTests : RuntimeActorTestBase
+{
+ private static readonly NodeId TestNode = NodeId.Parse("driver-test");
+ private static readonly RevisionHash RevA = RevisionHash.Parse(new string('a', 64));
+
+ [Fact]
+ public void SuccessfulApply_StoresTheArtifactInTheCache()
+ {
+ var db = NewInMemoryDbFactory();
+ var cache = new RecordingArtifactCache();
+ var (deploymentId, artifact) = SeedDeployment(db, RevA, clusterId: null);
+
+ var coordinator = CreateTestProbe();
+ var actor = Sys.ActorOf(DriverHostActor.Props(
+ db, TestNode, coordinator.Ref,
+ localRoles: new HashSet { "driver" },
+ deploymentArtifactCache: cache));
+
+ actor.Tell(new DispatchDeployment(deploymentId, RevA, CorrelationId.NewId()));
+ coordinator.ExpectMsg(TimeSpan.FromSeconds(5)).Outcome.ShouldBe(ApplyAckOutcome.Applied);
+
+ AwaitAssert(() => cache.Stores.Count.ShouldBe(1), duration: TimeSpan.FromSeconds(3));
+
+ var store = cache.Stores[0];
+ store.DeploymentId.ShouldBe(deploymentId.ToString());
+ store.RevisionHash.ShouldBe(RevA.Value);
+ store.Artifact.ShouldBe(artifact);
+ }
+
+ [Fact]
+ public void ApplyOfAClusterScopedArtifact_StoresUnderThatClusterId()
+ {
+ // The ClusterId is not config and not on IClusterRoleInfo — it is carried in the artifact's
+ // Nodes[] rows. Getting this wrong keys the pair's two nodes to different cache rows, and
+ // they would silently stop sharing a cache entry.
+ var db = NewInMemoryDbFactory();
+ var cache = new RecordingArtifactCache();
+ var (deploymentId, _) = SeedDeployment(db, RevA, clusterId: "SITE-A");
+
+ var coordinator = CreateTestProbe();
+ var actor = Sys.ActorOf(DriverHostActor.Props(
+ db, TestNode, coordinator.Ref,
+ localRoles: new HashSet { "driver" },
+ deploymentArtifactCache: cache));
+
+ actor.Tell(new DispatchDeployment(deploymentId, RevA, CorrelationId.NewId()));
+ coordinator.ExpectMsg(TimeSpan.FromSeconds(5));
+
+ AwaitAssert(() => cache.Stores.Count.ShouldBe(1), duration: TimeSpan.FromSeconds(3));
+ cache.Stores[0].ClusterId.ShouldBe("SITE-A");
+ }
+
+ [Fact]
+ public void UnscopedArtifact_StoresUnderTheSingleClusterSentinel()
+ {
+ var db = NewInMemoryDbFactory();
+ var cache = new RecordingArtifactCache();
+ var (deploymentId, _) = SeedDeployment(db, RevA, clusterId: null);
+
+ var coordinator = CreateTestProbe();
+ var actor = Sys.ActorOf(DriverHostActor.Props(
+ db, TestNode, coordinator.Ref,
+ localRoles: new HashSet { "driver" },
+ deploymentArtifactCache: cache));
+
+ actor.Tell(new DispatchDeployment(deploymentId, RevA, CorrelationId.NewId()));
+ coordinator.ExpectMsg(TimeSpan.FromSeconds(5));
+
+ AwaitAssert(() => cache.Stores.Count.ShouldBe(1), duration: TimeSpan.FromSeconds(3));
+ cache.Stores[0].ClusterId.ShouldBe("__single");
+ }
+
+ [Fact]
+ public void EmptyArtifact_IsNotCached()
+ {
+ // THE case that makes this cache safe to boot from. ReconcileDrivers catches its own DB
+ // failures, logs a warning and returns without rethrowing — so an apply can reach its
+ // success path, ACK Applied, and have started zero drivers. Persisting that as
+ // last-known-good would boot the node into an empty address space during the next outage:
+ // strictly worse than having no cache at all. An empty blob is the observable form of it.
+ var db = NewInMemoryDbFactory();
+ var cache = new RecordingArtifactCache();
+
+ var deploymentId = DeploymentId.NewId();
+ using (var ctx = db.CreateDbContext())
+ {
+ ctx.Deployments.Add(new Deployment
+ {
+ DeploymentId = deploymentId.Value,
+ RevisionHash = RevA.Value,
+ Status = DeploymentStatus.Sealed,
+ CreatedBy = "test",
+ SealedAtUtc = DateTime.UtcNow,
+ ArtifactBlob = Array.Empty(),
+ });
+ ctx.SaveChanges();
+ }
+
+ var coordinator = CreateTestProbe();
+ var actor = Sys.ActorOf(DriverHostActor.Props(
+ db, TestNode, coordinator.Ref,
+ localRoles: new HashSet { "driver" },
+ deploymentArtifactCache: cache));
+
+ actor.Tell(new DispatchDeployment(deploymentId, RevA, CorrelationId.NewId()));
+
+ // The apply still succeeds — an empty artifact is a legitimate no-op deployment.
+ coordinator.ExpectMsg(TimeSpan.FromSeconds(5)).Outcome.ShouldBe(ApplyAckOutcome.Applied);
+
+ // But nothing is cached from it.
+ Thread.Sleep(500);
+ cache.Stores.ShouldBeEmpty();
+ }
+
+ [Fact]
+ public void AThrowingCache_DoesNotFailTheApply()
+ {
+ // By the time the cache runs, an Applied ACK has already gone to the coordinator. If a cache
+ // failure escaped, ApplyAndAck's catch would send a SECOND, contradictory Failed ACK for a
+ // deployment the fleet already believes is live.
+ var db = NewInMemoryDbFactory();
+ var (deploymentId, _) = SeedDeployment(db, RevA, clusterId: null);
+
+ var coordinator = CreateTestProbe();
+ var actor = Sys.ActorOf(DriverHostActor.Props(
+ db, TestNode, coordinator.Ref,
+ localRoles: new HashSet { "driver" },
+ deploymentArtifactCache: new ThrowingArtifactCache()));
+
+ actor.Tell(new DispatchDeployment(deploymentId, RevA, CorrelationId.NewId()));
+
+ coordinator.ExpectMsg(TimeSpan.FromSeconds(5)).Outcome.ShouldBe(ApplyAckOutcome.Applied);
+
+ // Exactly one ACK, and it says Applied. A second Failed ACK arriving here would mean the
+ // cache failure had corrupted the deployment protocol.
+ coordinator.ExpectNoMsg(TimeSpan.FromMilliseconds(500));
+ }
+
+ [Fact]
+ public void NoCacheConfigured_AppliesNormally()
+ {
+ // Admin-only graphs and most test harnesses pass null. Caching is skipped, nothing else changes.
+ var db = NewInMemoryDbFactory();
+ var (deploymentId, _) = SeedDeployment(db, RevA, clusterId: null);
+
+ var coordinator = CreateTestProbe();
+ var actor = Sys.ActorOf(DriverHostActor.Props(
+ db, TestNode, coordinator.Ref,
+ localRoles: new HashSet { "driver" }));
+
+ actor.Tell(new DispatchDeployment(deploymentId, RevA, CorrelationId.NewId()));
+
+ coordinator.ExpectMsg(TimeSpan.FromSeconds(5)).Outcome.ShouldBe(ApplyAckOutcome.Applied);
+ }
+
+ ///
+ /// Seeds a sealed deployment whose artifact optionally scopes to a
+ /// cluster, and returns the exact bytes stored so a round-trip can be asserted.
+ ///
+ private static (DeploymentId Id, byte[] Artifact) SeedDeployment(
+ IDbContextFactory db, RevisionHash rev, string? clusterId)
+ {
+ object payload = clusterId is null
+ ? new
+ {
+ DriverInstances = Array.Empty