diff --git a/CLAUDE.md b/CLAUDE.md index a424866c..0b0d2588 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -219,6 +219,24 @@ The server supports configurable OPC UA transport security via the `OpcUa:Enable The server supports non-transparent warm/hot redundancy via the `Redundancy` section in `appsettings.json`. Two instances share the same Galaxy DB and the same mxaccessgw (under distinct `MxAccess.ClientName` values) but have unique `ApplicationUri` values. Each exposes `RedundancySupport`, `ServerUriArray`, and a dynamic `ServiceLevel` based on role and runtime health. The primary advertises a higher ServiceLevel than the secondary. See `docs/Redundancy.md` for the full guide. +## LocalDb pair-local config cache (Phase 1) + +Every **driver-role** node keeps a consolidated `ZB.MOM.WW.LocalDb` SQLite database (the retired +LiteDB `LocalCache` subsystem was deleted as superseded). Phase-1 scope: it caches the +**deployed-configuration artifact** (chunked, SHA-256-verified, newest-2 retention per cluster) so a +driver node can **boot from cache when central SQL Server is unreachable** — `DriverHostActor` writes +the cache after each successful apply (`IDeploymentArtifactCache` → `LocalDbDeploymentArtifactCache`, +in `Runtime/Deployment/`) and reads it as a boot fallback, flagging running-from-cache. Wired by +`AddOtOpcUaLocalDb` in the `hasDriver` branch only (`Host/Configuration/LocalDbRegistration.cs` + +`LocalDbSetup.OnReady`, whose DDL→`RegisterReplicated` order is load-bearing); **admin-only nodes +never register it.** The cache can optionally replicate to the node's redundant pair peer over the +library's gRPC sync — **default-OFF, fail-closed bearer auth** (`LocalDbSyncAuthInterceptor`), gated +on `LocalDb:SyncListenPort` (a dedicated Kestrel h2c listener) + `LocalDb:Replication:*`. On the +docker-dev rig the **site-a pair** has replication on; central + site-b stay default-OFF (site-b is +the pin). ApiKey via `${secret:}`/env, never committed (the rig's dev key is the committed-dev-secret +exception). See `docs/Configuration.md` (`LocalDb` section), `docs/Redundancy.md` (pair-local config +cache), and the runbook `docs/operations/2026-07-20-localdb-pair-replication.md`. + ## LDAP Authentication The server uses LDAP-based user authentication via the `Security:Ldap` section in `appsettings.json`. When enabled, credentials are validated by LDAP bind against a GLAuth server, and LDAP group membership maps to OPC UA permissions: `ReadOnly` (browse/read), `WriteOperate` (write FreeAccess/Operate attributes), `WriteTune` (write Tune attributes), `WriteConfigure` (write Configure attributes), `AlarmAck` (alarm acknowledgment). `LdapOpcUaUserAuthenticator` (`src/Server/ZB.MOM.WW.OtOpcUa.Host/OpcUa/LdapOpcUaUserAuthenticator.cs`) implements `IOpcUaUserAuthenticator`, delegating the LDAP bind + group lookup to `OtOpcUaLdapAuthService` (`src/Server/ZB.MOM.WW.OtOpcUa.Security/Ldap/OtOpcUaLdapAuthService.cs`, an `ILdapAuthService`). See `docs/security.md` for the full guide. diff --git a/Directory.Packages.props b/Directory.Packages.props index 155dd762..b62ff772 100644 --- a/Directory.Packages.props +++ b/Directory.Packages.props @@ -29,10 +29,10 @@ + - @@ -122,6 +122,13 @@ + + + diff --git a/NuGet.config b/NuGet.config index 48599f6d..9ae3c0d0 100644 --- a/NuGet.config +++ b/NuGet.config @@ -27,6 +27,8 @@ + + diff --git a/docker-dev/docker-compose.yml b/docker-dev/docker-compose.yml index af4cd50e..2d2d712d 100644 --- a/docker-dev/docker-compose.yml +++ b/docker-dev/docker-compose.yml @@ -224,8 +224,15 @@ services: Serilog__MinimumLevel__Override__Microsoft.EntityFrameworkCore: "Warning" Serilog__MinimumLevel__Override__Microsoft.AspNetCore: "Warning" GALAXY_MXGW_API_KEY: "${GALAXY_MXGW_API_KEY:-mxgw_otopcua2_GI7-tNozYE6cXGUSgEzL3AHDV7bYcYIHdMwKYgyHdX4}" + # Consolidated pair-local config cache (ZB.MOM.WW.LocalDb). Every driver-role node + # (central is admin,driver) gets one, on a per-node named volume so the cache survives + # container recreates. The central pair leaves replication OFF (default) — only the + # site-a pair replicates, as the enablement demo; site-b is the default-OFF pin. + LocalDb__Path: "/app/data/otopcua-localdb.db" ports: - "4840:4840" + volumes: + - otopcua-localdb-central-1:/app/data central-2: <<: *otopcua-host @@ -283,8 +290,12 @@ services: Serilog__MinimumLevel__Override__Microsoft.EntityFrameworkCore: "Warning" Serilog__MinimumLevel__Override__Microsoft.AspNetCore: "Warning" GALAXY_MXGW_API_KEY: "${GALAXY_MXGW_API_KEY:-mxgw_otopcua2_GI7-tNozYE6cXGUSgEzL3AHDV7bYcYIHdMwKYgyHdX4}" + # Pair-local config cache; central pair leaves replication OFF (see central-1). + LocalDb__Path: "/app/data/otopcua-localdb.db" ports: - "4841:4840" + volumes: + - otopcua-localdb-central-2:/app/data # ── Site A cluster (2-node driver-only) ───────────────────────────────────── # Driver-only members of the single mesh, scoped to SITE-A by ClusterId. No UI, @@ -313,8 +324,24 @@ services: # Resolved at runtime by GalaxyDriver.ResolveApiKey when a DriverInstance's # Gateway.ApiKeySecretRef = "env:GALAXY_MXGW_API_KEY". GALAXY_MXGW_API_KEY: "${GALAXY_MXGW_API_KEY:-mxgw_otopcua2_GI7-tNozYE6cXGUSgEzL3AHDV7bYcYIHdMwKYgyHdX4}" + # Pair-local config cache + REPLICATION ON — the site-a pair is the enablement demo. + # site-a-1 is the initiator: it binds the h2c sync listener on 9001 AND dials the peer. + # Setting SyncListenPort makes Program.cs add a dedicated Http2-only Kestrel listener and + # re-bind the primary HTTP port (an explicit Listen* otherwise discards ASPNETCORE_URLS). + LocalDb__Path: "/app/data/otopcua-localdb.db" + LocalDb__SyncListenPort: "9001" + LocalDb__Replication__PeerAddress: "http://site-a-2:9001" + # DEV-ONLY committed key (rig philosophy — like the SQL password + secrets KEK above). It + # MUST be byte-identical on both nodes: the interceptor is fail-closed, so any mismatch + # silently stops the pair converging. NEVER reuse this key outside this local dev rig. + LocalDb__Replication__ApiKey: "dev-site-a-localdb-sync-key" + # Row-count batching against gRPC's 4 MB message cap; artifact chunk rows are ≈171 KB, so + # 16 × 171 KB ≈ 2.7 MB stays under the cap with headroom. + LocalDb__Replication__MaxBatchSize: "16" ports: - "4842:4840" + volumes: + - otopcua-localdb-site-a-1:/app/data site-a-2: <<: *otopcua-host @@ -334,8 +361,18 @@ services: Serilog__MinimumLevel__Override__Microsoft.EntityFrameworkCore: "Warning" Serilog__MinimumLevel__Override__Microsoft.AspNetCore: "Warning" GALAXY_MXGW_API_KEY: "${GALAXY_MXGW_API_KEY:-mxgw_otopcua2_GI7-tNozYE6cXGUSgEzL3AHDV7bYcYIHdMwKYgyHdX4}" + # site-a-2 is the passive half of the replicating pair: it binds the sync listener on 9001 + # but does NOT dial (no PeerAddress). The replication stream is bidirectional, so a-1's + # single dial carries both directions. Same key + MaxBatchSize as a-1 (byte-identical key + # is mandatory — the interceptor fail-closes on a mismatch). + LocalDb__Path: "/app/data/otopcua-localdb.db" + LocalDb__SyncListenPort: "9001" + LocalDb__Replication__ApiKey: "dev-site-a-localdb-sync-key" + LocalDb__Replication__MaxBatchSize: "16" ports: - "4843:4840" + volumes: + - otopcua-localdb-site-a-2:/app/data # ── Site B cluster (2-node driver-only) ───────────────────────────────────── @@ -357,8 +394,14 @@ services: Serilog__MinimumLevel__Override__Microsoft.EntityFrameworkCore: "Warning" Serilog__MinimumLevel__Override__Microsoft.AspNetCore: "Warning" GALAXY_MXGW_API_KEY: "${GALAXY_MXGW_API_KEY:-mxgw_otopcua2_GI7-tNozYE6cXGUSgEzL3AHDV7bYcYIHdMwKYgyHdX4}" + # site-b is the default-OFF pin: it gets the pair-local cache but NO SyncListenPort and NO + # replication config, so no sync listener binds and ISyncStatus stays disconnected/Healthy. + # This is what the live gate's check 8 verifies — the cache works locally with replication off. + LocalDb__Path: "/app/data/otopcua-localdb.db" ports: - "4844:4840" + volumes: + - otopcua-localdb-site-b-1:/app/data site-b-2: <<: *otopcua-host @@ -378,8 +421,12 @@ services: Serilog__MinimumLevel__Override__Microsoft.EntityFrameworkCore: "Warning" Serilog__MinimumLevel__Override__Microsoft.AspNetCore: "Warning" GALAXY_MXGW_API_KEY: "${GALAXY_MXGW_API_KEY:-mxgw_otopcua2_GI7-tNozYE6cXGUSgEzL3AHDV7bYcYIHdMwKYgyHdX4}" + # site-b default-OFF pin (see site-b-1): cache on, replication off. + LocalDb__Path: "/app/data/otopcua-localdb.db" ports: - "4845:4840" + volumes: + - otopcua-localdb-site-b-2:/app/data traefik: image: traefik:v3.1 @@ -400,3 +447,13 @@ services: volumes: # SQL Server data dir — persists the OtOpcUa ConfigDb across container recreates. otopcua-mssql-data: + # Per-node pair-local config-cache (ZB.MOM.WW.LocalDb) files. One volume per node — the + # cache is node-local, not shared; it persists the cached deployment artifact (and, on the + # site-a pair, the replicated oplog state) across container recreates so a node can boot + # from cache after a restart even with central SQL down. + otopcua-localdb-central-1: + otopcua-localdb-central-2: + otopcua-localdb-site-a-1: + otopcua-localdb-site-a-2: + otopcua-localdb-site-b-1: + otopcua-localdb-site-b-2: diff --git a/docs/Configuration.md b/docs/Configuration.md index fdcac78c..5f01bfdc 100644 --- a/docs/Configuration.md +++ b/docs/Configuration.md @@ -146,6 +146,28 @@ key must carry the scopes `historian:read`, `historian:write`, `historian:tags:w [`docs/Historian.md`](Historian.md) for the full key reference, the migration note (old Wonderware keys → gateway keys), and the deployment prerequisites. +### `LocalDb` (pair-local config cache — driver role only) + +Driver-role nodes keep a consolidated **`ZB.MOM.WW.LocalDb`** SQLite database that caches the deployed +configuration artifact (chunked) so the node can **boot from cache when central SQL is unreachable**. The +section is bound by `AddOtOpcUaLocalDb` (`Host/Configuration/LocalDbRegistration.cs`) inside the `hasDriver` +branch — **admin-only nodes never register it, and never require `LocalDb:Path`**. Storage is unconditional; +replication stays inert until the sync port + peer are set. + +| Key | Default | Meaning | +|---|---|---| +| `LocalDb:Path` | `./data/otopcua-localdb.db` | SQLite file path. `ValidateOnStart`-required once `AddZbLocalDb` runs (driver nodes only). In containers put it on a durable volume. | +| `LocalDb:SyncListenPort` | `0` | `0` = replication off (no sync listener). A non-zero port binds a dedicated **h2c** sync listener; set the **same** port on both pair nodes. Setting it makes the Host add an explicit Kestrel `Listen*` and re-bind the primary HTTP port (an explicit `Listen*` otherwise discards `ASPNETCORE_URLS`). | +| `LocalDb:Replication:PeerAddress` | `""` | The peer this node **dials** (`http://:`). Set on the initiator only; leave empty on the passive node. The stream is bidirectional. | +| `LocalDb:Replication:ApiKey` | `""` | Bearer token for the sync stream. **Fail-closed:** no key ⇒ the passive endpoint refuses everything; a mismatch ⇒ the pair silently stops converging. **Must be byte-identical on both nodes.** Supply via `${secret:...}` / env — never a cleartext literal in production. | +| `LocalDb:Replication:MaxBatchSize` | (library default) | Rows per replication batch — **row-count-only** against gRPC's 4 MB cap. `16` for the artifact cache (chunk rows ≈171 KB, so 16 × 171 KB ≈ 2.7 MB). | + +Replication is **default-OFF** across the fleet; it is enabled per-pair as an opt-in. See +[`docs/operations/2026-07-20-localdb-pair-replication.md`](operations/2026-07-20-localdb-pair-replication.md) +for the enablement runbook (ApiKey handling, stop/start-together rule, tombstone-retention window, and the +never-`sqlite3`-a-live-WAL-DB inspection rules) and the "pair-local config cache" section of +[`docs/Redundancy.md`](Redundancy.md) for what boot-from-cache does and does not cover. + --- ## Environment variables diff --git a/docs/Redundancy.md b/docs/Redundancy.md index 19fe48b7..6692bed9 100644 --- a/docs/Redundancy.md +++ b/docs/Redundancy.md @@ -226,6 +226,35 @@ Net effect: each alarm transition appears **once** on `/alerts` and would histor See [ScriptedAlarms.md](ScriptedAlarms.md) and [AlarmTracking.md](AlarmTracking.md) for the scripted-alarm engine internals. +## Pair-local config cache (LocalDb — Phase 1) + +Independently of the ServiceLevel machinery above, every **driver-role** node keeps a consolidated +[`ZB.MOM.WW.LocalDb`](../CLAUDE.md) SQLite database that caches the **deployed-configuration +artifact** (chunked). Its job is a single failure mode redundancy does not otherwise cover: a driver +node restarting into a **central-SQL-Server outage** would come up with no configuration at all. With +the cache, it **boots from the last artifact it applied** instead, and logs a running-from-cache +signal. + +- **What replicates.** The cache can *optionally* replicate to the node's redundant pair peer over + the LocalDb library's gRPC sync (HLC-stamped, last-writer-wins). Two tables replicate: + `deployment_artifacts` and `deployment_pointer`. Replication is **default-OFF and fail-closed** — + it stays inert until `LocalDb:SyncListenPort` + a peer are configured, and a missing/mismatched + `LocalDb:Replication:ApiKey` refuses or silently halts sync. It is enabled per-pair as an opt-in. +- **The replicated-cache payoff.** In a pair with replication on, *either* node can be the one that + applied the latest deploy; the peer holds a byte-identical copy. So a node that restarts during a + central outage can boot the current config **even if it never applied that deploy itself** — its + peer did, and the row replicated. +- **What it does NOT cover.** The cache is a **boot fallback for central-SQL outages only**, not a + standby data path and not a replacement for central. A **new deployment still requires central + SQL**; the cache is read only when the central fetch fails at startup, and the node resumes + fresh-config behavior once central returns. It does not replicate live tag values, alarms, or + historian data — only the config artifact. It is orthogonal to the ServiceLevel/primary-gate + logic: a node running from cache still participates in redundancy normally. + +Full detail: the enablement runbook at +[`docs/operations/2026-07-20-localdb-pair-replication.md`](operations/2026-07-20-localdb-pair-replication.md) +and the `LocalDb` section of [`docs/Configuration.md`](Configuration.md). + ## Client-side failover The OtOpcUa Client CLI at `src/Client/ZB.MOM.WW.OtOpcUa.Client.CLI` supports `-F` / `--failover-urls` for automatic client-side failover; for long-running subscriptions the CLI monitors session KeepAlive and reconnects to the next available server, recreating the subscription on the new endpoint. See [`Client.CLI.md`](Client.CLI.md). diff --git a/docs/operations/2026-07-20-localdb-pair-replication.md b/docs/operations/2026-07-20-localdb-pair-replication.md new file mode 100644 index 00000000..5bc1dea5 --- /dev/null +++ b/docs/operations/2026-07-20-localdb-pair-replication.md @@ -0,0 +1,124 @@ +# LocalDb pair replication — operations runbook + +> **Phase 1 scope.** Every driver-role OtOpcUa node keeps a consolidated +> [`ZB.MOM.WW.LocalDb`](../../CLAUDE.md) SQLite database. Today it caches exactly one thing: the +> **deployed-configuration artifact**, chunked, so a node can **boot from cache when central SQL +> Server is unreachable**. That cache can *optionally* replicate to the node's redundant pair peer +> over the library's gRPC sync. **Replication is default-OFF and fail-closed.** This runbook covers +> enabling it, the operational rules that keep a pair converging, and the DB-inspection safety +> rules. + +## What replicates, and what it buys you + +- **Replicated tables:** `deployment_artifacts` (the artifact, split into base64 chunks) and + `deployment_pointer` (one current-deployment pointer per cluster). Registered — in this order, + after the DDL — by `LocalDbSetup.OnReady`. +- **The payoff:** in a redundant driver pair, *either* node can be the one that applied the latest + deploy and wrote the cache. With replication on, the peer holds a byte-identical copy. So a node + that restarts into a central-SQL outage can boot the current config **even if it never applied + that deploy itself** — its peer did, and the row replicated. +- **Convergence model:** last-writer-wins per primary key, HLC-stamped. Both nodes also store + locally after every apply, so the pair converges to identical content regardless of which node + deployed. Retention (newest-2 deployments per cluster) prunes on one node and the deletes + replicate as tombstones. + +## Enabling replication on a pair + +Replication has two knobs, both under `LocalDb`: + +| Key | Role | Notes | +|---|---|---| +| `LocalDb:SyncListenPort` | binds the dedicated **h2c** sync listener | `0` (default) = no listener, replication off. Set the **same** non-zero port on **both** nodes of the pair. | +| `LocalDb:Replication:PeerAddress` | the address this node **dials** | Set on **one** node (the initiator); leave empty on the other (passive). The stream is bidirectional — one dial carries both directions. | +| `LocalDb:Replication:ApiKey` | bearer token for the sync stream | **Must be byte-identical on both nodes.** See fail-closed rule below. Supply via `${secret:...}` / env — never a cleartext literal in production config. | +| `LocalDb:Replication:MaxBatchSize` | rows per replication batch | `16` for the artifact cache. Batching is **row-count-only** against gRPC's 4 MB message cap; artifact chunk rows are ≈171 KB, so 16 × 171 KB ≈ 2.7 MB stays under the cap. Raising it risks tripping the cap. | + +Example (site-a-1 initiator, site-a-2 passive): + +``` +site-a-1: LocalDb__SyncListenPort=9001 + LocalDb__Replication__PeerAddress=http://site-a-2:9001 + LocalDb__Replication__ApiKey=${secret:site-a-localdb-sync-key} + LocalDb__Replication__MaxBatchSize=16 +site-a-2: LocalDb__SyncListenPort=9001 + LocalDb__Replication__ApiKey=${secret:site-a-localdb-sync-key} # no PeerAddress + LocalDb__Replication__MaxBatchSize=16 +``` + +> **Kestrel note.** Setting `SyncListenPort` makes the Host add an explicit `Listen*` for the h2c +> sync listener. An explicit `Listen*` makes Kestrel **ignore `ASPNETCORE_URLS` entirely**, so the +> Host also re-binds the primary HTTP port in the same block. If you change the primary port, +> verify both `:` and `:` appear in the startup "Now listening on" lines. + +### The ApiKey rule (fail-closed) + +The replication library's passive endpoint verifies **no** authentication — the host +`LocalDbSyncAuthInterceptor` is the only gate, and it is **fail-closed**: + +- **No key configured ⇒ every sync call is refused** (`PermissionDenied`). "No key" is never + "no auth required". +- **A key mismatch ⇒ the pair silently stops converging.** The initiator's stream is rejected at + the peer; nothing errors loudly. A typo in one node's key looks exactly like "replication is + broken". If a pair is not converging, **check the keys match first.** + +## Operational rules + +- **Stop / start the pair together where you can.** Each node keeps working (and caching locally) + while its peer is down; the outage is not a data-loss event — the surviving node accumulates + writes and the peer catches up on rejoin. But a long-lived solo node drifts further from its + peer, so avoid leaving a pair split for extended periods. +- **Tombstone-retention resurrection window.** Retention prunes to the newest 2 deployments and + replicates the prune as tombstones. Tombstones are themselves retained only for a bounded window. + If a node is offline **longer than the tombstone-retention window**, a delete that happened during + its outage may no longer be expressible as a tombstone on rejoin — a pruned deployment could + briefly reappear until the next deploy re-prunes it. Keep pair outages well inside that window. +- **What boot-from-cache does NOT cover.** The cache is a *fallback for central-SQL outages at + boot*, not a replacement for central. A **new deployment still requires central SQL** — the cache + is only read when the central fetch fails at startup. When central returns, the node resumes + fresh-config behavior. Boot-from-cache logs a running-from-cache signal; treat a node that stays + on it as a central-connectivity incident, not steady state. + +## Inspecting the LocalDb file — safety rules + +The database runs in **WAL mode**. These rules exist because violating them corrupted a live DB in +the 2026-07-20 ScadaBridge incident: + +- **Never run `sqlite3` on the live file** (host-side, against a bind-mounted or container path). + Opening a live WAL DB from a second process across virtiofs poisons the WAL. **Always copy the + triplet first** and query the copy: + + ```bash + docker cp :/app/data/otopcua-localdb.db /tmp/localdb.db + docker cp :/app/data/otopcua-localdb.db-wal /tmp/localdb.db-wal + docker cp :/app/data/otopcua-localdb.db-shm /tmp/localdb.db-shm + sqlite3 /tmp/localdb.db 'SELECT cluster_id, deployment_id FROM deployment_pointer;' + ``` + +- **Metrics** come from the container, not a host curl (`aspnet:10.0` has no `curl`): + + ```bash + docker run --rm --network container: curlimages/curl:latest -s localhost:/metrics | grep localdb_ + ``` + +- **If an anomaly appears within seconds of your own measurement, suspect the measurement.** + Restart both nodes and re-observe untouched before blaming the library. + +## Useful queries (on a copied triplet) + +```sql +-- What each node thinks the current deployment is +SELECT cluster_id, deployment_id, revision_hash, applied_at_utc FROM deployment_pointer; + +-- Convergence check: the pointer's origin stamp should match on both nodes +SELECT pk_json, hlc, node_id, is_tombstone +FROM __localdb_row_version WHERE table_name = 'deployment_pointer' ORDER BY pk_json; + +-- Replication backlog (should drain to 0 when a pair is caught up) +SELECT COUNT(*) FROM __localdb_oplog; +``` + +## See also + +- [`docs/Redundancy.md`](../Redundancy.md) — the pair-local config cache section. +- [`docs/Configuration.md`](../Configuration.md) — the `LocalDb` appsettings section. +- The two-node convergence harness: `tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDb/`. diff --git a/docs/plans/2026-07-20-localdb-adoption-phase1.md b/docs/plans/2026-07-20-localdb-adoption-phase1.md new file mode 100644 index 00000000..e70b2a10 --- /dev/null +++ b/docs/plans/2026-07-20-localdb-adoption-phase1.md @@ -0,0 +1,706 @@ +# OtOpcUa LocalDb Adoption — Phase 1 Implementation Plan + +> **For Claude:** REQUIRED SUB-SKILL: Use superpowers-extended-cc:executing-plans to implement this plan task-by-task. +> +> **Execution model:** This plan is optimized for **Claude Opus** agents (`claude --model opus`). +> Dispatch implementer subagents on Opus for every task classified `high-risk`; `small`/`trivial` +> tasks may use Sonnet. Work on branch **`feat/localdb-phase1`** in this repo (`~/Desktop/OtOpcUa`, +> remote `lmxopcua`). Do NOT merge to `master` as part of this plan — stop at the DoD task and +> report. +> +> **Design authority:** `~/Desktop/scadaproj/docs/plans/2026-07-20-otopcua-localdb-design.md`. +> Read it before Task 0. The reference adoption is ScadaBridge (merged PR #23) — when in doubt, +> mirror `~/Desktop/ScadaBridge` (files named per task below). + +**Goal:** Give every driver-role OtOpcUa node a consolidated `ZB.MOM.WW.LocalDb` database that caches +the deployed-configuration artifact (chunked), boots from that cache when central SQL Server is +unreachable, and optionally replicates it to the node's redundant pair peer over the library's gRPC +sync (default-OFF, fail-closed bearer auth). + +**Architecture:** `AddZbLocalDb`/`AddZbLocalDbReplication` wired in a new `LocalDbRegistration` +(mirroring `SecretsRegistration`), DDL + `RegisterReplicated` in `LocalDbSetup.OnReady`, a dedicated +Kestrel h2c listener for `MapZbLocalDbSync` gated on `LocalDb:SyncListenPort`, a consumer-supplied +fail-closed bearer interceptor, and an `IDeploymentArtifactCache` seam written by `DriverHostActor` +after each successful apply and read as a boot fallback. The dormant LiteDB LocalCache subsystem is +deleted as superseded. + +**Tech stack:** .NET 10, `ZB.MOM.WW.LocalDb`/`.Replication`/`.Contracts` **0.1.1** (Gitea feed), +`Grpc.AspNetCore` 2.76.0, Akka.NET (existing), xunit.v3 (Host.IntegrationTests) + xunit2 TestKit +(actor tests). + +**Hard rules (from the ScadaBridge adoption — violating any of these is a defect):** +1. Order inside `OnReady` is load-bearing: **DDL → `RegisterReplicated` → writes**. Rows written + before registration are never captured or snapshotted — silently, forever. +2. No autoincrement PKs, no BLOB columns in replicated tables. +3. The library's passive sync endpoint verifies **no** ApiKey — the host interceptor is the only + auth, and it must be fail-closed (no key configured ⇒ deny everything). +4. `ILocalDb.CreateConnection()` returns an **already-open**, pragma-configured connection with the + `zb_hlc_next()` UDF. Never call `Open()` on it. There is **no in-memory mode** — tests use temp + files + `SqliteConnection.ClearAllPools()` before delete. +5. `Grpc.Core.Testing` does not exist on grpc-dotnet — hand-roll fake `ServerCallContext`s. +6. Explicit Kestrel `Listen*` calls make Kestrel **ignore `ASPNETCORE_URLS`** — when adding the sync + listener you must re-bind the existing HTTP port in the same block, and verify it on the rig. +7. Never run host `sqlite3` against a live bind-mounted WAL DB (copy the `db`/`-wal`/`-shm` triplet + with `cp` instead). `aspnet:10.0` containers have no `curl` — use a + `curlimages/curl` sidecar with `--network container:`. + +--- + +### Task 0: Preflight + recon (produces `docs/plans/2026-07-20-localdb-phase1-recon.md`) + +**Classification:** standard +**Estimated implement time:** ~5 min +**Parallelizable with:** none (everything downstream consumes its findings) + +**Files:** +- Create: `docs/plans/2026-07-20-localdb-phase1-recon.md` +- Read-only: `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Drivers/DriverHostActor.cs`, + `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Drivers/DeploymentArtifact.cs`, + `src/Server/ZB.MOM.WW.OtOpcUa.Host/Program.cs`, + `src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/SecretsRegistration.cs`, + `Directory.Packages.props`, `docker-dev/docker-compose.yml`, + the observability registration (`AddOtOpcUaObservability` implementation), + `src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/*` + +**Step 1: Verify the feed packages restore.** + +Run: +```bash +cd ~/Desktop/OtOpcUa && dotnet package search ZB.MOM.WW.LocalDb --source dohertj2-gitea --format json | head -50 +``` +Expected: `ZB.MOM.WW.LocalDb`, `.Replication`, `.Contracts` at `0.1.1`. If absent → STOP, report. + +**Step 2: Map and record, with `file:line` citations, each of:** +1. **Deploy fetch path:** where `DriverHostActor` loads `Deployment.ArtifactBlob` (the + `_dbFactory.CreateDbContext()` sites, ~L1472/1543), the artifact's runtime type + (`DeploymentArtifact` — how it's deserialized from the blob), the type/format of + `DeploymentId` and `RevisionHash`, and where a successful apply completes (the point after + which the cache write belongs). +2. **Cold-boot path:** what the actor does at startup before any `DispatchDeployment` arrives — + does it query central SQL for the current deployment? Where does the failure path land + (the `Stale` state, `DbHealthProbeActor` interaction, L1345 dispatch-ignore)? Identify the + exact seam where "central fetch failed at boot" is known — that is where the cache fallback goes. +3. **Cluster identity:** how a driver node knows its `ClusterId` (config key / `ClusterNode` row / + `IClusterRoleInfo`) — the cache is keyed by it. +4. **DriverHostActor construction:** how its dependencies are injected + (`WithOtOpcUaRuntimeActors`, `Props` wiring) so `IDeploymentArtifactCache` can be threaded in. + ⚠ `Props.Create` builds an expression tree — adding a parameter silently rebinds + out-of-position named args at call sites; list every construction site. +5. **Telemetry allowlist:** does `AddOtOpcUaObservability` restrict meters + (`ZbTelemetryOptions.Meters` or equivalent)? If yes, record where the list lives. +6. **Kestrel/URLs:** confirm the Host binds via `ASPNETCORE_URLS=http://+:9000` with no + `ConfigureKestrel` calls; record any existing `builder.WebHost` usage. +7. **LiteDB LocalCache:** confirm (grep) that nothing outside + `src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/` and its tests references + `ILocalConfigCache`/`GenerationSealedCache`/`ResilientConfigReader`/`LiteDB`. +8. **docker-dev volumes:** whether nodes have a writable data volume; record what `LocalDb:Path` + should be per node and how env vars are passed (`Cluster__*` style double-underscore). + +**Step 3: STOP conditions — halt the plan and report instead of improvising if:** +- the artifact apply path cannot be given a post-apply hook without restructuring the actor; +- `DeploymentId`/cluster identity are not stable strings/GUIDs; +- LiteDB LocalCache turns out to be referenced by live code. + +**Step 4: Commit.** +```bash +git add docs/plans/2026-07-20-localdb-phase1-recon.md +git commit -m "docs(localdb): phase-1 recon findings" +``` + +--- + +### Task 1: Package references and pins + +**Classification:** small +**Estimated implement time:** ~3 min +**Parallelizable with:** none (all code tasks build on it) + +**Files:** +- Modify: `Directory.Packages.props` +- Modify: `src/Server/ZB.MOM.WW.OtOpcUa.Host/ZB.MOM.WW.OtOpcUa.Host.csproj` +- Modify: `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/ZB.MOM.WW.OtOpcUa.Runtime.csproj` + +**Step 1:** Add to `Directory.Packages.props` (versions exact): +```xml + + + +``` +Check the existing `Grpc.Net.Client`/`Grpc.Core.Api` are already ≥ 2.76.0 and +`Google.Protobuf` ≥ 3.34.1 (the LocalDb.Replication nuspec floors); raise if not. Confirm a +`SQLitePCLRaw` pin ≥ 2.1.12 exists for the transitive `Microsoft.Data.Sqlite` chain (the family +advisory GHSA-2m69-gcr7-jv3q); Core.AlarmHistorian already pins `bundle_e_sqlite3` 2.1.12 — add +`SQLitePCLRaw.lib.e_sqlite3` 2.1.12 as an explicit reference in the Host if the audit flags it. + +**Step 2:** Host csproj: `ZB.MOM.WW.LocalDb`, `ZB.MOM.WW.LocalDb.Replication`, `Grpc.AspNetCore`. +Runtime csproj: `ZB.MOM.WW.LocalDb` only (it needs `ILocalDb`, not the sync engine). + +**Step 3:** `dotnet build ZB.MOM.WW.OtOpcUa.slnx` → 0 warnings/errors (repo builds warnings-as-errors). + +**Step 4: Commit.** `git commit -m "build(localdb): reference ZB.MOM.WW.LocalDb 0.1.1 + Grpc.AspNetCore"` + +--- + +### Task 2: Schema + `LocalDbSetup.OnReady` + registration tests + +**Classification:** standard +**Estimated implement time:** ~5 min +**Parallelizable with:** Task 3 + +**Files:** +- Create: `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/DeploymentCacheSchema.cs` +- Create: `src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbSetup.cs` +- Test: `tests/Server/ZB.MOM.WW.OtOpcUa.Host.Tests/LocalDbSetupTests.cs` (create the test project + reference layout to match neighboring Host unit tests; if Host has no unit-test project, place in + the closest existing one and note the deviation) + +**Step 1: Write the failing tests first** — using a real temp-file `ILocalDb` (build it via +`new ServiceCollection().AddZbLocalDb(config, LocalDbSetup.OnReady)` with `LocalDb:Path` pointed at +a temp file; dispose + `SqliteConnection.ClearAllPools()` in cleanup): +- `OnReady_RegistersExactlyTheTwoDeploymentTables` — `db.ReplicatedTables.Keys` ordinal-sorted + equals `["deployment_artifacts", "deployment_pointer"]` (assert BOTH directions: no fewer, no more). +- `DeploymentArtifacts_PkIsDeploymentIdPlusChunkIndex` and `DeploymentPointer_PkIsClusterId` + (pin `ReplicatedTable.PkColumns`). +- `RowsWrittenAfterOnReady_EnterTheOplog` — insert a row, assert `SELECT COUNT(*) FROM __localdb_oplog` ≥ 1 + (pins the DDL→register ordering; this is the assertion that catches a silently-broken CDC). + +**Step 2: Run tests, watch them fail** (types don't exist yet). + +**Step 3: Implement.** `DeploymentCacheSchema` depends only on `Microsoft.Data.Sqlite` (the +ScadaBridge `*Schema.Apply` pattern — self-sufficient for direct construction in tests): + +```csharp +namespace ZB.MOM.WW.OtOpcUa.Runtime.Deployment; + +public static class DeploymentCacheSchema +{ + public static void Apply(SqliteConnection connection) + { + using var cmd = connection.CreateCommand(); + cmd.CommandText = """ + CREATE TABLE IF NOT EXISTS deployment_artifacts ( + deployment_id TEXT NOT NULL, + chunk_index INTEGER NOT NULL, + cluster_id TEXT NOT NULL, + revision_hash TEXT NOT NULL, + chunk_count INTEGER NOT NULL, + chunk_base64 TEXT NOT NULL, + cached_at_utc TEXT NOT NULL, + PRIMARY KEY (deployment_id, chunk_index) + ); + CREATE TABLE IF NOT EXISTS deployment_pointer ( + cluster_id TEXT NOT NULL PRIMARY KEY, + deployment_id TEXT NOT NULL, + revision_hash TEXT NOT NULL, + artifact_sha256 TEXT NOT NULL, + applied_at_utc TEXT NOT NULL + ); + """; + cmd.ExecuteNonQuery(); + } +} +``` + +`LocalDbSetup` (Host): +```csharp +namespace ZB.MOM.WW.OtOpcUa.Host.Configuration; + +internal static class LocalDbSetup +{ + // ORDER IS LOAD-BEARING: DDL, then RegisterReplicated, then (nothing else in Phase 1). + // Rows written before RegisterReplicated are invisible to the peer forever, silently. + public static void OnReady(ILocalDb db) + { + using var connection = db.CreateConnection(); // already open — do NOT call Open() + DeploymentCacheSchema.Apply(connection); + db.RegisterReplicated("deployment_artifacts"); + db.RegisterReplicated("deployment_pointer"); + } +} +``` + +**Step 4:** `dotnet test --filter "FullyQualifiedName~LocalDbSetupTests"` → PASS. + +**Step 5: Commit.** `git commit -m "feat(localdb): deployment-cache schema + OnReady registration"` + +--- + +### Task 3: Fail-closed sync auth interceptor + tests + +**Classification:** standard +**Estimated implement time:** ~4 min +**Parallelizable with:** Task 2 + +**Files:** +- Create: `src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbSyncAuthInterceptor.cs` +- Test: `tests/Server/ZB.MOM.WW.OtOpcUa.Host.Tests/LocalDbSyncAuthInterceptorTests.cs` + +Port ScadaBridge's `src/ZB.MOM.WW.ScadaBridge.Host/LocalDbSyncAuthInterceptor.cs` (read it first; +keep behavior identical, adjust namespace only). Semantics to preserve exactly: +- `ServicePrefix = "/localdb_sync.v1.LocalDbSync/"`; any other method path passes through untouched. +- Expected token from `IOptions.Value.ApiKey`. +- **Fail-closed:** null/empty configured key ⇒ `RpcException(new Status(StatusCode.PermissionDenied, ...))` + for every sync call. Missing/mismatched bearer ⇒ same. +- `CryptographicOperations.FixedTimeEquals` over UTF-8 bytes; header `authorization`, + case-insensitive, `"Bearer "` prefix. Override all four server handler kinds. + +**Tests** (hand-rolled fake `ServerCallContext` — `Grpc.Core.Testing` does not exist on grpc-dotnet; +copy ScadaBridge's `LocalDbSyncAuthInterceptorTests.cs` fake): non-sync method passes with no key; +sync + no configured key → PermissionDenied; wrong bearer → PermissionDenied; correct bearer → passes. + +Write tests first (fail), implement, `dotnet test --filter "FullyQualifiedName~LocalDbSyncAuthInterceptor"`, +commit `feat(localdb): fail-closed bearer interceptor for the sync endpoint`. + +--- + +### Task 4: `LocalDbRegistration` + Program.cs DI wiring + config defaults + +**Classification:** high-risk (touches Program.cs / role gating) +**Estimated implement time:** ~5 min +**Parallelizable with:** none (Task 5 edits the same file) + +**Files:** +- Create: `src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbRegistration.cs` +- Modify: `src/Server/ZB.MOM.WW.OtOpcUa.Host/Program.cs` (inside the `hasDriver` branch) +- Modify: `src/Server/ZB.MOM.WW.OtOpcUa.Host/appsettings.json` + +**Step 1:** `LocalDbRegistration`, mirroring `SecretsRegistration`'s shape: +```csharp +namespace ZB.MOM.WW.OtOpcUa.Host.Configuration; + +internal static class LocalDbRegistration +{ + /// Driver-role nodes only. Storage is unconditional; replication stays inert until + /// LocalDb:Replication:PeerAddress (initiator) / LocalDb:SyncListenPort (listener) are set. + public static IServiceCollection AddOtOpcUaLocalDb( + this IServiceCollection services, IConfiguration configuration) + { + services.AddZbLocalDb(configuration, LocalDbSetup.OnReady); + services.AddZbLocalDbReplication(configuration); + return services; + } + + public static int SyncListenPort(IConfiguration configuration) => + configuration.GetValue("LocalDb:SyncListenPort"); +} +``` + +**Step 2:** Program.cs, in the `hasDriver` block (near `AddAlarmHistorian`): +`builder.Services.AddOtOpcUaLocalDb(builder.Configuration);` and +`builder.Services.AddGrpc(o => o.Interceptors.Add());` +(AddGrpc only under `hasDriver` — admin-only nodes expose no sync surface). + +**Step 3:** appsettings.json: +```json +"LocalDb": { + "Path": "./data/otopcua-localdb.db", + "SyncListenPort": 0, + "Replication": {} +} +``` +`LocalDb:Path` is `ValidateOnStart`-required once `AddZbLocalDb` runs — since registration is +driver-gated, admin-only configs need nothing. Check the role-overlay files +(`appsettings.driver.json` / `appsettings.admin-driver.json`) and any deploy templates for +conflicting `LocalDb` keys. + +**Step 4:** Build + full Host unit tests. Verify an admin-only graph doesn't require the key: this +is pinned properly in Task 10's integration tests, but do a quick +`OTOPCUA_ROLES=admin dotnet run` smoke only if cheap; otherwise rely on Task 10. + +**Step 5: Commit.** `feat(localdb): wire AddOtOpcUaLocalDb into the driver role` + +--- + +### Task 5: Dedicated h2c sync listener + endpoint mapping (THE Kestrel task) + +**Classification:** high-risk +**Estimated implement time:** ~5 min +**Parallelizable with:** none (Program.cs) + +**Files:** +- Modify: `src/Server/ZB.MOM.WW.OtOpcUa.Host/Program.cs` + +**Why high-risk:** the Host today binds exclusively via `ASPNETCORE_URLS=http://+:9000`. Any +explicit `ConfigureKestrel(... Listen*)` makes Kestrel **ignore URLs entirely** (it logs +"Overriding address(es)"), which would silently kill the AdminUI/deploy API behind Traefik. + +**Step 1:** Add, before `builder.Build()`: +```csharp +var syncPort = LocalDbRegistration.SyncListenPort(builder.Configuration); +if (hasDriver && syncPort > 0) +{ + // Explicit Listen* replaces ASPNETCORE_URLS wholesale, so re-bind the primary HTTP + // endpoint here too. Parse it from the configured URLs rather than hard-coding 9000. + var urls = builder.Configuration["ASPNETCORE_URLS"] ?? builder.Configuration["urls"] ?? "http://+:9000"; + var httpPort = new Uri(urls.Split(';')[0].Replace("+", "localhost").Replace("*", "localhost")).Port; + builder.WebHost.ConfigureKestrel(k => + { + k.ListenAnyIP(httpPort); // HTTP/1.1 (existing surface) + k.ListenAnyIP(syncPort, o => o.Protocols = HttpProtocols.Http2); // h2c, prior-knowledge gRPC + }); +} +``` +When `syncPort == 0` (the default) nothing changes — URLs binding stays untouched. Cleartext +`Http1AndHttp2` cannot serve prior-knowledge h2c, hence the dedicated `Http2`-only listener. + +**Step 2:** After the app pipeline's other `Map*` calls, driver-gated: +```csharp +if (hasDriver && syncPort > 0) +{ + app.MapZbLocalDbSync(); +} +``` +(Mapping is harmless when unauthenticated — the interceptor fail-closes — but gating on the port +keeps admin-only and default-OFF graphs entirely free of the endpoint.) + +**Step 3: Verify both surfaces.** +```bash +dotnet build ZB.MOM.WW.OtOpcUa.slnx +``` +Then a local smoke with the port on: +`ASPNETCORE_URLS=http://+:9000 OTOPCUA_ROLES=driver LocalDb__SyncListenPort=9001 dotnet run --project src/Server/ZB.MOM.WW.OtOpcUa.Host ...` +— expect startup logs to show BOTH `:9000` and `:9001` bound, and `curl -s localhost:9000/healthz` +(or the mapped health route) still answering. If the app needs SQL to boot, defer the smoke to the +Task 12 rig check but say so explicitly in the task notes. + +**Step 4: Commit.** `feat(localdb): dedicated h2c listener + MapZbLocalDbSync gated on LocalDb:SyncListenPort` + +--- + +### Task 6: `IDeploymentArtifactCache` + chunked LocalDb implementation + tests + +**Classification:** standard +**Estimated implement time:** ~5 min +**Parallelizable with:** Task 3, Task 5 + +**Files:** +- Create: `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/IDeploymentArtifactCache.cs` +- Create: `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/LocalDbDeploymentArtifactCache.cs` +- Test: `tests/Server/ZB.MOM.WW.OtOpcUa.Runtime.Tests/Deployment/LocalDbDeploymentArtifactCacheTests.cs` + +**Step 1: Failing tests** (real temp-file `ILocalDb` with `LocalDbSetup.OnReady` applied): +- round-trip: store a 300 KiB random artifact → `GetCurrentAsync` returns byte-identical payload + (forces ≥ 3 chunks at 128 KiB); +- retention: store 3 deployments for one cluster → only the newest 2 remain in + `deployment_artifacts`; pointer names the newest; +- integrity: corrupt one chunk row via SQL → `GetCurrentAsync` returns null (miss), never a + truncated artifact; +- missing pointer → null. + +**Step 2: Implement:** +```csharp +public interface IDeploymentArtifactCache +{ + Task StoreAsync(string clusterId, string deploymentId, string revisionHash, + byte[] artifact, CancellationToken ct = default); + Task GetCurrentAsync(string clusterId, CancellationToken ct = default); +} + +public sealed record CachedDeploymentArtifact( + string DeploymentId, string RevisionHash, byte[] Artifact, DateTimeOffset AppliedAtUtc); +``` +`LocalDbDeploymentArtifactCache(ILocalDb db)`: +- `StoreAsync`: one `ILocalDbTransaction` — delete any existing chunks for this `deployment_id` + (idempotent re-store), insert chunks (raw 128 * 1024 bytes per chunk, `Convert.ToBase64String`), + upsert the pointer (`INSERT ... ON CONFLICT(cluster_id) DO UPDATE`), then prune: delete + `deployment_artifacts` rows whose `deployment_id` is not among the newest 2 `cached_at_utc` for + this `cluster_id`. SHA-256 over the raw artifact into the pointer. ISO-8601 UTC timestamps. +- `GetCurrentAsync`: read pointer; read chunks `ORDER BY chunk_index`; verify count == + `chunk_count` and SHA-256 matches; return null on any mismatch (log a warning naming which check + failed). +- Use `db.ExecuteAsync`/`QueryAsync` with anonymous-object parameters (`@Name` markers). Remember + a `Dictionary` parameter **throws** — use anonymous objects. + +**Step 3:** run tests → PASS. **Step 4: Commit.** +`feat(localdb): chunked deployment-artifact cache over ILocalDb` + +--- + +### Task 7: Cache write path in `DriverHostActor` + +**Classification:** high-risk (actor model, `Props` expression-tree trap) +**Estimated implement time:** ~5 min +**Parallelizable with:** none + +**Files:** (exact edit sites come from the Task 0 recon doc — cite it in the commit) +- Modify: `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Drivers/DriverHostActor.cs` +- Modify: the actor-registration site (`WithOtOpcUaRuntimeActors` / DI extension) to provide + `IDeploymentArtifactCache` +- Modify: `src/Server/ZB.MOM.WW.OtOpcUa.Host/Program.cs` or `LocalDbRegistration` — register + `IDeploymentArtifactCache` → `LocalDbDeploymentArtifactCache` singleton (driver branch) +- Test: the **xunit2** TestKit project used for existing DriverHostActor tests (recon names it) + +**Steps:** +1. **Failing test:** after a successful artifact apply, the cache received `StoreAsync` with the + applied deployment's id/hash/bytes (inject a recording fake `IDeploymentArtifactCache`). + Also: a cache that throws on `StoreAsync` does NOT fail the apply (apply result unchanged, error + logged). +2. Thread the dependency through the actor's constructor. ⚠ `Props.Create` is an expression tree: + after adding the parameter, re-check **every** construction site for positional/named-arg + rebinding (the recon lists them) — do not rely on the compiler. +3. Invoke `StoreAsync` fire-and-forget (`PipeTo`-style or a guarded `Task.Run` per the actor's + existing async conventions — match whatever pattern the actor already uses for side-effect IO) + at the post-apply point identified in recon. Cluster id from the recon-identified source. +4. Run the actor test suite for this project. Commit: + `feat(localdb): DriverHostActor stores applied artifacts in the pair-local cache` + +--- + +### Task 8: Boot-from-cache read path + running-from-cache signal + +**Classification:** high-risk +**Estimated implement time:** ~5 min +**Parallelizable with:** none + +**Files:** +- Modify: `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Drivers/DriverHostActor.cs` +- Test: same xunit2 TestKit project as Task 7 + +**Steps:** +1. **Failing tests:** + - central fetch fails at startup AND cache has an artifact → the actor applies the cached + artifact (assert via whatever observable the apply already exposes) and logs/flags + running-from-cache; + - central fetch fails AND cache empty → today's behavior exactly (Stale, no apply); + - central fetch **succeeds** → cache is NOT consulted (fresh config wins; assert the fake + cache's `GetCurrentAsync` was never called on the happy path). +2. Implement at the recon-identified boot-failure seam. The signal: a log warning at minimum plus, + if the actor already publishes health/status (recon says how), a `RunningFromCache` marker on it. + Do NOT touch `DispatchDeployment` handling — a new deployment still requires central. +3. Run tests; commit `feat(localdb): boot from the pair-local artifact cache when central SQL is unreachable`. + +--- + +### Task 9: Delete the dormant LiteDB LocalCache subsystem + +**Classification:** small +**Estimated implement time:** ~3 min +**Parallelizable with:** Task 6, Task 10, Task 11 + +**Files:** +- Delete: `src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/` (entire directory: + `ILocalConfigCache`, `LiteDbConfigCache`, `GenerationSealedCache`, `ResilientConfigReader`, + `GenerationSnapshot`, `StaleConfigFlag`, exceptions) +- Delete: `tests/Core/ZB.MOM.WW.OtOpcUa.Configuration.Tests/GenerationSealedCacheTests.cs`, + `ResilientConfigReaderTests.cs` +- Modify: `Directory.Packages.props` + the Configuration csproj — remove the `LiteDB` package if + nothing else references it (grep first) + +**Steps:** Confirm the Task 0 recon's "nothing references it" finding still holds +(`grep -rn "ILocalConfigCache\|GenerationSealedCache\|ResilientConfigReader\|LiteDB" src/ tests/`), +delete, build the full solution, run the Configuration test project. DoD phrasing: **no references +from code** — leave any explanatory prose/comments that point readers at LocalDb instead. Add one +line to the recon doc noting LiteDB's removal. Commit: +`refactor(localdb): delete the dormant LiteDB LocalCache (superseded by ZB.MOM.WW.LocalDb)` + +--- + +### Task 10: Health check + telemetry meter allowlist + +**Classification:** small +**Estimated implement time:** ~4 min +**Parallelizable with:** Task 9, Task 11 + +**Files:** +- Create: `src/Server/ZB.MOM.WW.OtOpcUa.Host/Health/LocalDbReplicationHealthCheck.cs` (or the + directory `AddOtOpcUaHealth` uses — recon names it) +- Modify: the `AddOtOpcUaHealth` registration (driver branch) +- Modify: the observability meter list IF Task 0 found an allowlist + +**Steps:** +1. Failing unit tests: replication unconfigured (no peer, no listener) → `Healthy` (default-OFF + must not degrade a plain node); peer configured + `ISyncStatus.Connected == false` → `Degraded`; + connected + backlog `null` → `Degraded` (unknown backlog is not healthy); connected + backlog + small → `Healthy`. +2. Implement against `ISyncStatus` + `IOptions` (peer-configured = non-empty + `PeerAddress` OR `SyncListenPort > 0` — pass the latter in via options/config). +3. **Meter allowlist:** if `AddOtOpcUaObservability` restricts meters, add + `"ZB.MOM.WW.LocalDb.Replication"` (use `LocalDbMetrics.MeterName`) — this is a silent allowlist; + the ScadaBridge live gate is the proof it bites. If there is no allowlist, record that in the + recon doc and skip. +4. Commit `feat(localdb): replication health check + meter export`. + +--- + +### Task 11: DI-pin integration tests over the real Program.cs + +**Classification:** standard +**Estimated implement time:** ~5 min +**Parallelizable with:** Task 9, Task 10 + +**Files:** +- Test: `tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbWiringTests.cs` (xunit.v3, + `WebApplicationFactory` — the project already builds the real Program.cs; set + `OTOPCUA_ROLES` per test case the way neighboring tests do) + +**Pins (each its own test):** +1. Driver graph: `ILocalDb` resolves, is a singleton, `ReplicatedTables` ordinal-sorted == + `["deployment_artifacts", "deployment_pointer"]` — exact set, both directions load-bearing. +2. Driver graph: `IDeploymentArtifactCache` resolves to `LocalDbDeploymentArtifactCache`. +3. Default-OFF pin: `ISyncStatus` resolves with `Connected == false`, `PeerNodeId == null`. +4. Admin-only graph: `ILocalDb` is NOT registered (`GetService()` is null) and boot does + not demand `LocalDb:Path`. +5. Health: the LocalDb health check is registered in the driver graph. + +Point `LocalDb:Path` at a per-test temp file via the factory's config overrides; +`ClearAllPools` + delete in cleanup. These tests exist because DI extensions without a +container-built test have shipped inert three times in this family (Secrets 0.2.0/0.2.2, +ScadaBridge#22). Commit `test(localdb): DI pins over the real host graph`. + +--- + +### Task 12: Two-node convergence harness + tests + +**Classification:** high-risk +**Estimated implement time:** ~5 min (harness) + ~4 min (scenarios) — split the commit if needed +**Parallelizable with:** none (depends on Tasks 2, 3, 5, 6) + +**Files:** +- Create: `tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDb/LocalDbPairHarness.cs` +- Test: `tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDb/LocalDbPairConvergenceTests.cs` + +**Harness:** copy the pattern from +`~/Desktop/ScadaBridge/tests/ZB.MOM.WW.ScadaBridge.IntegrationTests/LocalDbSitePairHarness.cs` and +the library's `ConvergenceFixture` (`~/Desktop/scadaproj/ZB.MOM.WW.LocalDb/tests/.../Convergence/ConvergenceFixture.cs`): +- Two full stacks over a real loopback Kestrel h2c socket (port 0), **through the real + `LocalDbSyncAuthInterceptor`** with a shared test key; node A initiator, node B passive. +- Both initialized via the **production `LocalDbSetup.OnReady`** — a hand-written schema proves + only that the test agrees with itself. +- DBs owned by the harness, registered as pre-constructed singletons (so MS.DI won't dispose them + on host teardown); `KillTransportAsync`/`RestartTransportAsync`; tight `FlushInterval` (50 ms), + bounded backoff (2 s); temp files, `ClearAllPools` cleanup; a collection definition to serialize + these tests. + +**Scenarios:** +1. `ArtifactStoredOnA_ConvergesToB_ByteIdentical` — store a multi-chunk artifact via + `LocalDbDeploymentArtifactCache` on A; B's cache returns byte-identical bytes; both nodes' + `__localdb_row_version` rows for the pointer carry the **same HLC and origin node id** (B holds + A's row, not a re-derived one). +2. `RetentionPruneOnA_TombstonesReachB` — third deployment on A prunes the first; B's chunk rows + for the pruned deployment disappear and `__localdb_row_version WHERE is_tombstone=1` rows exist + on B. +3. `WritesWhileTransportDown_SurviveRejoin` — kill transport, store on A, restart, converge. +4. `WrongApiKey_NeverConverges` — harness with mismatched keys: assert **no** convergence after a + bounded wait AND (positive control) that the same scenario with matching keys converges — an + absence assertion without a positive control passed vacuously in ScadaBridge. + +**DoD within this task:** temporarily comment out the two `RegisterReplicated` calls and confirm +scenarios 1–3 go red (run locally, do not commit the red state); restore. Record the red/green +evidence in the task notes. Commit `test(localdb): 2-node convergence harness + scenarios`. + +--- + +### Task 13: docker-dev rig configuration + +**Classification:** small +**Estimated implement time:** ~4 min +**Parallelizable with:** Task 14 + +**Files:** +- Modify: `docker-dev/docker-compose.yml` + +**Steps** (adjust to the recon's findings on volumes/env style): +1. Every driver node: a named volume (or existing data mount) backing `/app/data`, and + `LocalDb__Path=/app/data/otopcua-localdb.db`. +2. **site-a pair only** (mirror the ScadaBridge default-OFF posture — site-b stays unreplicated as + the pin): + - site-a-1: `LocalDb__SyncListenPort=9001`, `LocalDb__Replication__PeerAddress=http://site-a-2:9001`, + `LocalDb__Replication__ApiKey=dev-site-a-localdb-sync-key`, `LocalDb__Replication__MaxBatchSize=16` + - site-a-2: `LocalDb__SyncListenPort=9001`, same `ApiKey`, same `MaxBatchSize`, **no PeerAddress** + (passive; the stream is bidirectional). + The key must be byte-identical on both — the interceptor fail-closes on any mismatch and the + pair silently stops converging. +3. `MaxBatchSize=16` because batching is row-count-only against gRPC's 4 MB cap and artifact chunk + rows are ≈ 171 KB (16 × 171 KB ≈ 2.7 MB worst case). +4. `docker compose config` to validate; do NOT bring the rig up in this task (the live gate task + owns rig runs). +5. Commit `chore(localdb): rig config — consolidated path everywhere, replication on the site-a pair`. + +--- + +### Task 14: Documentation + +**Classification:** small +**Estimated implement time:** ~4 min +**Parallelizable with:** Task 13 + +**Files:** +- Create: `docs/operations/2026-07-20-localdb-pair-replication.md` (runbook: enable/disable, + ApiKey handling via `${secret:}`, stop/start-together rule, tombstone-retention resurrection + window, MaxBatchSize row-count-vs-4MB, never host-`sqlite3` a live WAL DB / cp-triplet recipe) +- Modify: `docs/Redundancy.md` (a short "pair-local config cache" section: what replicates, what + boot-from-cache does and does not cover) +- Modify: `CLAUDE.md` (this repo): LocalDb adoption row/paragraph — state Phase 1 scope, default-OFF, + rig-only enablement +- Modify: `docs/Configuration.md` if it documents config keys (add the `LocalDb` section) + +Commit `docs(localdb): phase-1 runbook + redundancy/config docs`. + +--- + +### Task 15: DoD sweep (offline) + +**Classification:** standard +**Estimated implement time:** ~5 min +**Parallelizable with:** none (final offline task) + +**Steps:** +1. `dotnet build ZB.MOM.WW.OtOpcUa.slnx` → **0 warnings** (warnings-as-errors). +2. `dotnet test ZB.MOM.WW.OtOpcUa.slnx` → full suite green; record counts. Baseline any + pre-existing failures BEFORE this branch (`git stash` → run → unstash) rather than assuming; + report deltas only. +3. Greps (phrase as "no references from code"; explanatory comments may remain): + `grep -rn "LiteDB\|ILocalConfigCache\|GenerationSealedCache" src/ tests/` → no code references. +4. Confirm the positive-control evidence from Task 12 is recorded. +5. Update `docs/plans/2026-07-20-localdb-adoption-phase1.md.tasks.json` statuses. +6. Commit `chore(localdb): phase-1 DoD sweep` and STOP. Report: branch name, commit list, test + counts, and that the live gate (Task 16) needs the rig + operator go-ahead. + +--- + +### Task 16: Live gate on the docker-dev rig (run only with explicit user go-ahead) + +**Classification:** high-risk (rig) +**Estimated implement time:** ~30 min wall-clock (mostly waiting) +**Parallelizable with:** none + +**Preamble:** All DB inspection via `cp` of the `db`/`-wal`/`-shm` triplet out of the container +(`docker cp`) and querying the copy — never host `sqlite3` on the live file (it poisons the WAL +across virtiofs; root cause of the 2026-07-20 ScadaBridge incident). Metrics via +`docker run --rm --network container: curlimages/curl:latest -s localhost:/metrics`. +If an anomaly appears within seconds of one of your own measurements, suspect the measurement +first — restart both nodes and re-run untouched before blaming the library. + +**Checks (all must PASS; record evidence in `docs/plans/2026-07-20-localdb-phase1-live-gate.md`):** +1. Rig up, site-a pair healthy, `/metrics` on both shows `localdb_` series (proves the meter + allowlist work). +2. Deploy a config through the normal deploy API → both site-a nodes' `deployment_pointer` + + `deployment_artifacts` rows are byte-identical with **identical HLC + origin node id** (one + node's row replicated, not two independent derivations — both nodes also locally store after + apply, so expect LWW-converged identical content either way; assert convergence, and note which + origin won). +3. **Boot-from-cache:** stop the SQL container; restart site-a-2; it comes up serving the cached + config with the running-from-cache signal in its log; start SQL again; node recovers fresh + behavior. +4. **Replicated-cache payoff:** wipe site-a-2's local DB file (container stopped), restart it with + SQL **up**, let replication repopulate the cache from site-a-1, then repeat check 3's SQL-down + restart — it boots from a cache it never wrote itself. +5. Transport kill (pause site-a-2 container): store/deploy on a-1, oplog depth rises; unpause; + drains to 0; converged. +6. Retention: deploy a 3rd config; pruned deployment's chunks gone on BOTH nodes with tombstone + rows present on both. +7. Both-nodes-together restart: clean rejoin, identical counts, zero + `disk I/O error`/`SQLITE_IOERR`/`corrupt` in either log. +8. site-b pair (default-OFF pin): LocalDb file exists and works locally, `ISyncStatus` health + Healthy, no sync connections, no listener on 9001. + +Record PASS/FAIL per check with the actual evidence (row counts, md5s, log lines). Commit the gate +doc. Do not merge — report. + +--- + +## Task persistence + +Tasks file: `docs/plans/2026-07-20-localdb-adoption-phase1.md.tasks.json` (same directory). +Update statuses as tasks complete; any deviation from this plan gets a `deviation` note on the task +(the ScadaBridge tasks.json convention — the record of *why* is what makes the plan auditable). diff --git a/docs/plans/2026-07-20-localdb-adoption-phase1.md.tasks.json b/docs/plans/2026-07-20-localdb-adoption-phase1.md.tasks.json new file mode 100644 index 00000000..8136c608 --- /dev/null +++ b/docs/plans/2026-07-20-localdb-adoption-phase1.md.tasks.json @@ -0,0 +1,153 @@ +{ + "planPath": "docs/plans/2026-07-20-localdb-adoption-phase1.md", + "tasks": [ + { + "id": 0, + "subject": "Task 0: Preflight + recon (findings doc, STOP conditions)", + "status": "completed", + "notes": "Feed verified: LocalDb/.Contracts/.Replication all 0.1.1. All 3 STOP conditions clear. Findings: docs/plans/2026-07-20-localdb-phase1-recon.md. 8 deviations recorded (D-1..D-8); D-1/D-3/D-5/D-6 change downstream work materially.", + "deviation": "D-1 cache read cannot be keyed by ClusterId at the boot seam (unkeyed newest-pointer read instead); D-3 a third LiteDB test file must be deleted; D-5 WebApplicationFactory is deliberately unused in this repo (use TwoNodeClusterHarness); D-6 driver-only nodes have no ASPNETCORE_URLS so the Kestrel re-bind fallback is wrong. See recon doc \u00a79." + }, + { + "id": 1, + "subject": "Task 1: Package references and pins (LocalDb 0.1.1, Grpc.AspNetCore, SQLitePCLRaw)", + "status": "completed", + "blockedBy": [ + 0 + ] + }, + { + "id": 2, + "subject": "Task 2: Schema + LocalDbSetup.OnReady + registration tests", + "status": "completed", + "blockedBy": [ + 1 + ] + }, + { + "id": 3, + "subject": "Task 3: Fail-closed sync auth interceptor + tests", + "status": "completed", + "blockedBy": [ + 1 + ] + }, + { + "id": 4, + "subject": "Task 4: LocalDbRegistration + Program.cs DI wiring + config defaults", + "status": "completed", + "blockedBy": [ + 2, + 3 + ] + }, + { + "id": 5, + "subject": "Task 5: Dedicated h2c sync listener + MapZbLocalDbSync (Kestrel URLs-override risk)", + "status": "completed", + "blockedBy": [ + 4 + ] + }, + { + "id": 6, + "subject": "Task 6: IDeploymentArtifactCache + chunked LocalDb implementation + tests", + "status": "completed", + "blockedBy": [ + 2 + ] + }, + { + "id": 7, + "subject": "Task 7: Cache write path in DriverHostActor (Props expression-tree trap)", + "status": "completed", + "blockedBy": [ + 0, + 6 + ] + }, + { + "id": 8, + "subject": "Task 8: Boot-from-cache read path + running-from-cache signal", + "status": "completed", + "blockedBy": [ + 7 + ] + }, + { + "id": 9, + "subject": "Task 9: Delete the dormant LiteDB LocalCache subsystem", + "status": "completed", + "blockedBy": [ + 0 + ] + }, + { + "id": 10, + "subject": "Task 10: Health check + telemetry meter allowlist", + "status": "completed", + "blockedBy": [ + 4 + ] + }, + { + "id": 11, + "subject": "Task 11: DI-pin integration tests over the real Program.cs", + "status": "completed", + "blockedBy": [ + 5, + 10 + ] + }, + { + "id": 12, + "subject": "Task 12: Two-node convergence harness + scenarios (+ positive control)", + "status": "completed", + "blockedBy": [ + 5, + 6 + ] + }, + { + "id": 13, + "subject": "Task 13: docker-dev rig configuration (site-a pair on, site-b pin off)", + "status": "completed", + "blockedBy": [ + 5 + ] + }, + { + "id": 14, + "subject": "Task 14: Documentation (runbook, Redundancy.md, CLAUDE.md)", + "status": "completed", + "blockedBy": [ + 8 + ] + }, + { + "id": 15, + "subject": "Task 15: DoD sweep (offline) \u2014 STOP and report after this", + "status": "completed", + "blockedBy": [ + 7, + 8, + 9, + 11, + 12, + 13, + 14 + ], + "notes": "Offline DoD sweep: full-solution build 0 errors (824 pre-existing OTOPCUA0001 analyzer warnings in driver *test* projects; none reference any LocalDb/DeploymentCache file). Grep: no LiteDB/ILocalConfigCache/GenerationSealedCache code refs (2 explanatory doc-comment lines in ILdapGroupRoleMappingService remain, allowed). Runtime.Tests 407/0/31 twice (the previously-flagged intermittent did NOT reproduce). Host.IntegrationTests LocalDb subset 40/0/0. Full-solution `dotnet test` with stash-baseline NOT run: many suites are infra-gated (driver fixtures, full Akka mesh, LDAP real-bind, shared-SQL DB tests) and cannot run offline on macOS \u2014 deferred to Task 16 / CI. Task 12 positive-control evidence recorded in commit afa5be71. Did NOT merge to master \u2014 stopped here per plan." + }, + { + "id": 16, + "subject": "Task 16: Live gate on the docker-dev rig (needs explicit user go-ahead)", + "status": "completed", + "blockedBy": [ + 15 + ], + "notes": "Live gate RAN on the docker-dev rig with explicit user go-ahead. 8/8 checks pass (check 4 with a documented limitation). Caught + fixed FOUR real defects offline tests missed (4b2f0e6e NU1101 packageSourceMapping, ce9fa07f ASPNETCORE_HTTP_PORTS re-bind, 9137cb41 empty-address-space-on-cache-boot, c6a9f93a cache-not-repopulated-on-RestoreApplied), each with a regression test. Two documented limitations (follow-ups): no replication back-fill of a fully-wiped node; oplog growth on default-OFF nodes. Post-fix: full build 0 errors, Runtime.Tests 409/0, all LocalDb Host.IntegrationTests green; only consistent failure is the infra-gated AbCip_Green_AgainstSim (sim not up). Evidence: 2026-07-20-localdb-phase1-live-gate.md. NOT merged." + } + ], + "lastUpdated": "2026-07-20T00:00:00Z" +} \ No newline at end of file diff --git a/docs/plans/2026-07-20-localdb-adoption-phase2.md b/docs/plans/2026-07-20-localdb-adoption-phase2.md new file mode 100644 index 00000000..943bcc48 --- /dev/null +++ b/docs/plans/2026-07-20-localdb-adoption-phase2.md @@ -0,0 +1,293 @@ +# OtOpcUa LocalDb Adoption — Phase 2 Implementation Plan (alarm-historian store-and-forward) + +> **For Claude:** REQUIRED SUB-SKILL: Use superpowers-extended-cc:executing-plans to implement this plan task-by-task. +> +> **Execution model:** Optimized for **Claude Opus** agents (`claude --model opus`); dispatch +> `high-risk` tasks on Opus. Branch **`feat/localdb-phase2`** in `~/Desktop/OtOpcUa` (remote +> `lmxopcua`). **Prerequisite: Phase 1 (`docs/plans/2026-07-20-localdb-adoption-phase1.md`) is +> merged (or this branch is stacked on it) and its live gate passed.** Do not merge as part of this +> plan; stop at the DoD task. +> +> **Design authority:** `~/Desktop/scadaproj/docs/plans/2026-07-20-otopcua-localdb-design.md` §3.4, +> D8, D9. Reference implementation for "replace a bespoke store" phasing: ScadaBridge Phase 2 +> (`~/Desktop/ScadaBridge/docs/plans/2026-07-19-localdb-adoption-phase2.md` + its live gate doc). + +**Goal:** Move the alarm-historian store-and-forward buffer (today the standalone +`alarm-historian.db` owned by `SqliteStoreAndForwardSink`) into the consolidated LocalDb file as a +replicated table with a primary-gated drain, so a redundant pair no longer loses buffered alarm +history when a node dies — and delete the sink's bespoke file/connection management outright. + +**Architecture:** `alarm_sf_events` (TEXT GUID PK) registered in `LocalDbSetup.OnReady`; the sink +rewired onto `ILocalDb` behind its unchanged public seam; drain gated on the delivered-snapshot +Primary role via `PrimaryGatePolicy` (at-least-once across failover, accepted and documented); a +one-time idempotent migrator from the legacy file, running **after** registration. + +**Risk framing (from ScadaBridge):** Phase 2 replaces a *working* mechanism — a harder risk class +than Phase 1's "add where none existed." Cutover (delete + rewire) lands in **one commit**; there is +no dual-mechanism period, and the cutover is the test. + +**Hard rules:** identical to Phase 1's list (OnReady ordering, no autoincrement/BLOB, fail-closed +auth already in place, no in-memory SQLite in tests, cp-triplet-only rig inspection), plus: +- **Legacy-copy column lists must INTERSECT** with what the legacy file actually has + (`pragma_table_info` probe) — a missing column throws and readers silently discard every row. +- Absence assertions need a positive control. +- DoD greps phrased as "no references from code" (explanatory comments may survive). + +--- + +### Task 0: Recon (produces `docs/plans/2026-07-20-localdb-phase2-recon.md`) + +**Classification:** standard +**Estimated implement time:** ~5 min +**Parallelizable with:** none + +**Files:** +- Create: `docs/plans/2026-07-20-localdb-phase2-recon.md` +- Read-only: `src/Core/ZB.MOM.WW.OtOpcUa.Core.AlarmHistorian/SqliteStoreAndForwardSink.cs` (and the + whole `Core.AlarmHistorian` project), `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Historian/AlarmHistorianOptions.cs`, + the drain worker (whatever forwards batches to `GatewayHistorian`/`SendEvent`), + `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Drivers/PrimaryGatePolicy.cs`, + `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Drivers/DriverHostActor.cs` (how `_localRole` + + driver-member-count reach the gate today) + +**Record, with `file:line` citations:** +1. The sink's exact table schema: name, columns, PK type (**autoincrement?** — decides whether the + migrator needs deterministic `mig-{node}-{legacyId}` ids), indices, and any BLOB columns + (**STOP condition:** a BLOB payload column cannot be registered — it must become base64 TEXT in + the new schema, and the recon must size the largest realistic payload against the 171 KB-ish + chunk guidance; alarm events are small JSON, so expect this to be fine, but verify). +2. The public seam: the interface the drain worker and producers use (e.g. `IAlarmHistorianSink` / + enqueue+dequeue+markDelivered+deadLetter methods), so the rewire can keep it byte-compatible. +3. Semantics to preserve: `Capacity` (1,000,000) enforcement, `MaxAttempts` (10), dead-letter + retention (30 d), `BatchSize` (100), `DrainIntervalSeconds` (5) — where each lives. +4. The drain worker's lifecycle: hosted service or actor? Where a Primary-role check can be + injected, and how the delivered-snapshot role (`RedundancyStateChanged` cache) is accessible + from it (via `DriverHostActor`, a shared status service, or a message). If the role is only + available inside `DriverHostActor`, note the cleanest bridge (e.g. an `IRedundancyRoleView` + singleton the actor updates) — that becomes Task 4's shape. +5. Whether the sink is constructed per-node config path (`AlarmHistorian:DatabasePath`) anywhere + else (tests, tooling). +6. How `AlarmHistorian:Enabled=false` short-circuits (NullAlarmHistorianSink) — the rewire must + keep the disabled path allocating no LocalDb tables? No: tables are created unconditionally in + `OnReady` (cheap, empty); only the sink/drain stay Null. Note this in the doc. + +Commit: `docs(localdb): phase-2 recon findings`. + +--- + +### Task 1: `alarm_sf_events` schema + registration (+ tests) + +**Classification:** standard +**Estimated implement time:** ~4 min +**Parallelizable with:** none (Task 2 depends on it) + +**Files:** +- Create: `src/Core/ZB.MOM.WW.OtOpcUa.Core.AlarmHistorian/AlarmSfSchema.cs` (depends only on + `Microsoft.Data.Sqlite`) +- Modify: `src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbSetup.cs` +- Test: extend `LocalDbSetupTests` + +Schema shape (adjust column names to the recon's findings — preserve today's semantics): +```sql +CREATE TABLE IF NOT EXISTS alarm_sf_events ( + id TEXT NOT NULL PRIMARY KEY, -- app-minted GUID (never autoincrement) + payload_json TEXT NOT NULL, + enqueued_at_utc TEXT NOT NULL, + attempts INTEGER NOT NULL DEFAULT 0, + status TEXT NOT NULL DEFAULT 'pending', -- pending | delivered | dead + last_attempt_utc TEXT NULL, + dead_at_utc TEXT NULL +); +CREATE INDEX IF NOT EXISTS ix_alarm_sf_events_status ON alarm_sf_events(status, enqueued_at_utc); +``` +`OnReady` order becomes: Phase-1 DDL → `AlarmSfSchema.Apply` → the two Phase-1 +`RegisterReplicated` calls → `RegisterReplicated("alarm_sf_events")` → **migrator (Task 5) last**. +(All DDL may run before all registrations; the invariant is registration-before-writes.) + +TDD: failing test first — exact replicated set becomes +`["alarm_sf_events", "deployment_artifacts", "deployment_pointer"]` (ordinal-sorted; update the +Phase-1 exact-set pins in the same commit — they are *supposed* to go red here, that's them +working). Oplog-capture test for an `alarm_sf_events` insert. Commit +`feat(localdb): alarm_sf_events replicated table`. + +--- + +### Task 2: Rewire the sink onto `ILocalDb` + delete bespoke file management (the cutover commit, part 1 of 2 — see Task 3) + +**Classification:** high-risk +**Estimated implement time:** ~5 min +**Parallelizable with:** none + +**Files:** +- Modify: `src/Core/ZB.MOM.WW.OtOpcUa.Core.AlarmHistorian/SqliteStoreAndForwardSink.cs` (or replace + with `LocalDbStoreAndForwardSink.cs` — keep the public seam identical either way) +- Modify: its registration (`AddAlarmHistorian`) to inject `ILocalDb` +- Modify: `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Historian/AlarmHistorianOptions.cs` — remove + `DatabasePath` (a breaking config key removal: note it in the runbook/CHANGELOG task) +- Test: the sink's existing unit tests, rewired to a temp-file `ILocalDb` via `TestLocalDb`-style + helper (create `tests/.../TestSupport` helper if none exists — real DB, never a stub: a stubbed + bare `SqliteConnection` lacks `zb_hlc_next()` and fails closed on registered tables) + +**Steps:** +1. Write/port failing tests for the seam's semantics: enqueue, drain batch of `BatchSize`, + `MaxAttempts` → dead-letter, capacity enforcement, dead-letter retention purge. +2. Implement over `ILocalDb.ExecuteAsync/QueryAsync` (anonymous-object params). Delete the private + connection/pragma/file-open code and any `PRAGMA journal_mode` calls (LocalDb owns pragmas). + GUIDs minted at enqueue (`Guid.NewGuid().ToString("N")`). +3. `Capacity` enforcement: count-based insert guard (preserve today's overflow behavior per recon). +4. Core.AlarmHistorian gains a package ref on core `ZB.MOM.WW.LocalDb` (interface only). +5. Build + project tests green. **Commit together with Task 3** if the drain gate can't compile + separately (the ScadaBridge tasks-14/15/16 circular-dependency landmine — check before assuming + they're independent commits). + +--- + +### Task 3: Primary-gated drain + +**Classification:** high-risk +**Estimated implement time:** ~5 min +**Parallelizable with:** none + +**Files:** (exact shape from recon item 4) +- Modify: the drain worker +- Create (if recon says so): `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Redundancy/IRedundancyRoleView.cs` + — a singleton snapshot (`RedundancyRole? LocalRole`, `int DriverMemberCount`) updated by + `DriverHostActor` where it already caches `_localRole` +- Test: drain-worker tests + an actor test pinning that `DriverHostActor` publishes role changes to + the view + +**Semantics:** +- Drain runs only when `PrimaryGatePolicy.ShouldServiceAsPrimary(localRole, driverMemberCount)` is + true — same policy, same boot-window posture (unknown role drains only when the node is alone). +- Delivered/dead-letter marks are row UPDATEs → they replicate, so the standby's copy tracks drain + progress and does not re-deliver already-marked rows after failover. +- **At-least-once across failover is accepted:** rows delivered on the old primary whose + `delivered` mark hadn't replicated yet will be re-sent by the new primary. Document in the + runbook (Task 7); do NOT build dedup. +- When replication is OFF (default), the gate still applies but `driverMemberCount` for a solo + node keeps today's behavior — verify with a test: single-node, role unknown → drains (no + regression for unpaired deployments). + +TDD: failing tests — secondary role does not drain; primary drains; unknown+alone drains; +unknown+paired does not. Commit (with Task 2 if coupled): +`feat(localdb): alarm S&F on LocalDb with primary-gated drain (cutover)`. + +--- + +### Task 4: One-time legacy migrator (`alarm-historian.db` → consolidated) + +**Classification:** standard +**Estimated implement time:** ~5 min +**Parallelizable with:** Task 6 + +**Files:** +- Create: `src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/AlarmSfLegacyMigrator.cs` +- Modify: `LocalDbSetup.OnReady` — call it **last**, after all `RegisterReplicated` calls +- Test: `tests/Server/ZB.MOM.WW.OtOpcUa.Host.Tests/AlarmSfLegacyMigratorTests.cs` + +Mirror `~/Desktop/ScadaBridge/src/ZB.MOM.WW.ScadaBridge.Host/SiteLocalDbLegacyMigrator.cs`: +- Source path from the pre-removal `AlarmHistorian:DatabasePath` default (`alarm-historian.db`) — + read the raw config key even though the option property is gone. +- `ShouldMigrate`: skip if file missing or `.migrated` exists. `:memory:`/`file:` sources → no-op. +- **If the legacy PK is autoincrement (recon):** deterministic ids `mig-{NodeName}-{legacyId}` + (node name from the recon-identified config key) — rerunnable without duplicates under + `INSERT OR IGNORE`, and no cross-node collision. If already GUIDs, copy as-is. +- One transaction for the whole copy; `File.Move(path, path + ".migrated")` only after commit; + failure throws out of `OnReady` → boot fails, legacy untouched. +- **Column intersection** via `pragma_table_info` on the legacy table; required-PK guard. +- Both pair nodes migrate independently; their rows have distinct ids (node-prefixed), so the + merged buffer is the union — expected; note that the new primary will drain the standby's + migrated rows too. + +TDD: failing tests — happy path row counts; idempotent re-run; failure leaves legacy file +untouched; migrated rows **enter the oplog** (assert `__localdb_oplog` — the registration-order +pin); older legacy file missing a column still migrates (intersection). Commit +`feat(localdb): one-time alarm-historian.db migrator`. + +--- + +### Task 5: Convergence + failover scenarios in the pair harness + +**Classification:** high-risk +**Estimated implement time:** ~5 min +**Parallelizable with:** Task 4 + +**Files:** +- Test: `tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDb/AlarmSfConvergenceTests.cs` + (reuses the Phase-1 `LocalDbPairHarness`) + +**Scenarios:** +1. `AlarmBurstOnA_ConvergesToB_AndOplogDrains` — enqueue N events on A; identical rowset on B; + oplog → 0. +2. `DeliveredMarksReplicate` — mark rows delivered on A; B's copies show delivered (the + no-redeliver-after-failover property, asserted at the data layer). +3. `WritesWhileTransportDown_SurviveRejoin`. +4. Positive control: with `RegisterReplicated("alarm_sf_events")` commented out, scenarios 1–3 go + red (run locally, record, restore — do not commit red). + +Commit `test(localdb): alarm S&F convergence scenarios`. + +--- + +### Task 6: Rig config + docs + +**Classification:** small +**Estimated implement time:** ~4 min +**Parallelizable with:** Task 4 + +**Files:** +- Modify: `docker-dev/docker-compose.yml` — enable `AlarmHistorian__Enabled=true` on the site-a + pair if not already (the gate needs a live buffer); remove any `AlarmHistorian__DatabasePath` + env vars (key deleted). +- Modify: `docs/operations/2026-07-20-localdb-pair-replication.md` — add: alarm S&F replication + semantics, at-least-once-across-failover statement, migrator behavior (`.migrated` sidecar), + `DatabasePath` key removal. +- Modify: `docs/AlarmHistorian.md` + `CLAUDE.md` — sink now lives in the consolidated LocalDb; + drain is primary-gated. + +Commit `docs+chore(localdb): phase-2 rig config + docs`. + +--- + +### Task 7: DoD sweep (offline) + +**Classification:** standard +**Estimated implement time:** ~5 min +**Parallelizable with:** none + +1. Full solution build → 0 warnings; full test suite green (deltas vs pre-branch baseline only). +2. Greps, phrased as "no references from **code**": the old bespoke connection management + (`AlarmHistorian:DatabasePath`, direct `new SqliteConnection` inside Core.AlarmHistorian except + via schema helpers/tests) — explanatory comments may remain. +3. Positive-control evidence from Task 5 recorded. +4. Exact-set replicated-tables pin = 3 tables, both directions. +5. Update `…phase2.md.tasks.json`; commit `chore(localdb): phase-2 DoD sweep`; STOP and report. + +--- + +### Task 8: Live gate on the docker-dev rig (run only with explicit user go-ahead) + +**Classification:** high-risk (rig) +**Estimated implement time:** ~30 min wall-clock +**Parallelizable with:** none + +Same inspection rules as Phase 1's gate (cp-triplet, curl sidecar, observer-suspicion rule). +Record in `docs/plans/2026-07-20-localdb-phase2-live-gate.md`: +1. Migration ran: `.migrated` sidecar present on both site-a nodes; row counts match legacy. +2. Alarm burst (drive a real driver alarm or the historian-gateway-unreachable path) converges: + identical rowsets, oplog drains to 0, dead letters 0. +3. Only the primary drains: stop the historian gateway egress, buffer builds on BOTH nodes' + tables (replicated), but only the primary's drain worker logs attempts. +4. Failover: stop the primary; the standby (new primary) resumes draining the shared buffer; + count of double-delivered events observed and recorded (at-least-once evidence, not a failure). +5. Both-nodes-together restart clean; zero `disk I/O error`/`SQLITE_IOERR` in logs. +6. site-b (default-OFF pin): sink works locally, no sync traffic. + +Commit the gate doc; report; do not merge. + +--- + +## Task persistence + +Tasks file: `docs/plans/2026-07-20-localdb-adoption-phase2.md.tasks.json`. Record deviations per +task — especially if Tasks 2/3 had to land as one commit (the expected outcome). diff --git a/docs/plans/2026-07-20-localdb-adoption-phase2.md.tasks.json b/docs/plans/2026-07-20-localdb-adoption-phase2.md.tasks.json new file mode 100644 index 00000000..7cbfdc9a --- /dev/null +++ b/docs/plans/2026-07-20-localdb-adoption-phase2.md.tasks.json @@ -0,0 +1,15 @@ +{ + "planPath": "docs/plans/2026-07-20-localdb-adoption-phase2.md", + "tasks": [ + { "id": 0, "subject": "Task 0: Recon — sink schema, seam, drain lifecycle, role-view bridge (STOP conditions)", "status": "pending" }, + { "id": 1, "subject": "Task 1: alarm_sf_events schema + registration (+ exact-set pin update)", "status": "pending", "blockedBy": [0] }, + { "id": 2, "subject": "Task 2: Rewire sink onto ILocalDb + delete bespoke file management (cutover 1/2)", "status": "pending", "blockedBy": [1] }, + { "id": 3, "subject": "Task 3: Primary-gated drain via PrimaryGatePolicy (cutover 2/2 — may co-commit with Task 2)", "status": "pending", "blockedBy": [2] }, + { "id": 4, "subject": "Task 4: One-time alarm-historian.db legacy migrator", "status": "pending", "blockedBy": [3] }, + { "id": 5, "subject": "Task 5: Convergence + failover scenarios in the pair harness (+ positive control)", "status": "pending", "blockedBy": [3] }, + { "id": 6, "subject": "Task 6: Rig config + docs", "status": "pending", "blockedBy": [3] }, + { "id": 7, "subject": "Task 7: DoD sweep (offline) — STOP and report after this", "status": "pending", "blockedBy": [4, 5, 6] }, + { "id": 8, "subject": "Task 8: Live gate on the docker-dev rig (needs explicit user go-ahead)", "status": "pending", "blockedBy": [7] } + ], + "lastUpdated": "2026-07-20T00:00:00Z" +} diff --git a/docs/plans/2026-07-20-localdb-phase1-live-gate.md b/docs/plans/2026-07-20-localdb-phase1-live-gate.md new file mode 100644 index 00000000..379ca4be --- /dev/null +++ b/docs/plans/2026-07-20-localdb-phase1-live-gate.md @@ -0,0 +1,68 @@ +# LocalDb Phase 1 — live gate (docker-dev rig) + +> Task 16 of `2026-07-20-localdb-adoption-phase1.md`. Evidence log for the 8 live-rig checks. +> Rig: local `docker-dev/docker-compose.yml` (6 host nodes, one shared SQL). site-a pair replicates; +> central + site-b are default-OFF. +> +> **Safety rules followed:** all DB inspection via `docker cp` of the `db`/`-wal`/`-shm` triplet out +> of the container, querying the copy — never host `sqlite3` on the live WAL file. Metrics via a +> `curlimages/curl` sidecar with `--network container:` (`aspnet:10.0` has no curl). + +## Rig build/up + +Two real defects surfaced by the rebuild (both fixed on-branch, re-verified): + +1. **`NU1101` — packageSourceMapping gap** (`4b2f0e6e`). `ZB.MOM.WW.LocalDb*` was not mapped to the + `dohertj2-gitea` feed, so any *clean* restore (Docker image build, CI, fresh clone) failed. Local + dev builds masked it via the warm global NuGet cache. Fixed by adding the two patterns; verified + with a temp-packages-dir restore. +2. **`ASPNETCORE_HTTP_PORTS` re-bind gap** (`ce9fa07f`). The Kestrel re-bind read only + `urls`/`ASPNETCORE_URLS`; the aspnet:8.0+ base image sets `ASPNETCORE_HTTP_PORTS=8080` (all + interfaces) as the container default instead. A driver node that never sets `URLS` fell through to + `localhost:5000`, silently moving its health/metrics surface to loopback. Central nodes were fine + (they set `ASPNETCORE_URLS=9000` explicitly). Fixed to fall through URLS → HTTP_PORTS/HTTPS_PORTS → + 5000; 4 new unit tests. After the fix, driver nodes bind `[::]:8080` + `[::]:9001` (site-a) / + `[::]:8080` only (site-b). + +Rig: 6 nodes rebuilt from branch `feat/localdb-phase1` @ `ce9fa07f`, all up. + +## Checks + +| # | Check | Result | Evidence | +|---|---|---|---| +| 1 | Rig up, site-a pair healthy, `/metrics` shows `localdb_` series | ✅ PASS | Both site-a nodes `/healthz` 200 Healthy; `/metrics` shows `localdb_oplog_depth` + `localdb_sync_reconnects_total` on meter `ZB.MOM.WW.LocalDb.Replication` (allowlist works); site-a-2 served `/localdb_sync.v1.LocalDbSync/Sync` with routing `match_status=success` (interceptor passed, session established); both oplogs depth 0 | +| 2 | Deploy → both site-a nodes' pointer + artifacts byte-identical, same HLC + origin | ✅ PASS | Deploy `6e687451…` (rev `efb04c79…`) via `POST :9200/api/deployments` → both site-a-1 and site-a-2 `deployment_pointer` = `SITE-A / 6e687451… / efb04c79… / sha EFB04C79…` (identical); artifact 1 chunk each; **pointer `__localdb_row_version` identical on both: hlc `116955620225646592`, origin node_id `a0576df6-…`, tombstone 0** — one row replicated, not two derivations; both oplogs depth 0 | +| 3 | Boot-from-cache: SQL down, restart site-a-2, serves cached config w/ signal | ✅ PASS (found + fixed a defect) | With SQL stopped, site-a-2 restart logged `RUNNING FROM CACHE — … booted deployment 6e687451…`; **address space materialised from the cached blob**: `AddressSpaceApplier: applied plan (added=18)` / `OpcUaPublish: applied rebuild (added=18)`. Browse confirms **16 nodes on site-a-2 = 16 on site-a-1**. **Defect the gate caught (`9137cb41`):** pre-fix the rebuild re-read the artifact from the down ConfigDb and no-op'd, so the cache-booted node served **1** node vs the healthy peer's 16 — it logged success but browsed empty. Fixed by passing the in-hand cached blob to `RebuildAddressSpace`; regression test added | +| 4 | Replicated-cache payoff: wipe a-2 DB, repopulate, then SQL-down boot | ⚠️ PARTIAL — 1 limitation documented, 1 defect fixed | Wiped a-2's LocalDb volume, restarted with **SQL down**: sync session re-established (a-2 served a `Sync` call) but a-2's cache stayed **0/0/0** and a-1's oplog was **0** — the deploy's rows had been acked by the old a-2 and pruned, so there was no delta to send and **no snapshot-resync fired**. ⇒ **Limitation: pure replication does not back-fill a fully-wiped, already-converged node** (library delta-replication; snapshot-resync is gated on the oplog cap, which the running engine never hits). Separately found a **defect (`c6a9f93a`)**: recovery via `RestoreApplied` never wrote the cache (only fresh `ApplyAndAck` did), so a wiped node stayed cache-less. Fixed: `RestoreApplied` now re-caches. Verified — after the fix, wiped a-2 restarted with SQL up: `restored served state … on bootstrap` → cache repopulated `SITE-A / 6e687451`, chunks=1. **Net: a wiped node self-heals its cache from central on next boot (regaining boot-from-cache for future outages); it is not back-filled by its peer during the same outage.** | +| 5 | Transport kill: pause a-2, oplog rises; unpause, drains to 0 | ✅ PASS | `docker pause` on a-2 severed the sync transport; during the window a-2 accumulated **oplog=2** unacked local writes (its re-cache) while a-1 held 0. `docker unpause` → a-2's oplog **drained to 0 within 5s**, both converged on the same pointer. Note: the rig's redeploys carry an identical revision, so a fresh cache write on a-1 short-circuits (no-op) — the store-during-partition variant is covered by the automated harness scenario `WritesWhileTransportDown_SurviveRejoin` (proven red without `RegisterReplicated`) | +| 6 | Retention: 3rd deploy prunes oldest; chunks gone + tombstones on BOTH nodes | ✅ PASS | Cached 3 distinct deployment-ids (via deploy + restart re-cache). Both nodes retain exactly the newest 2 (`475df87a`, `962962dc`); the oldest `6e687451`'s chunks are **0 on both** (pruned); **1 artifact tombstone on both** — the prune replicated as a tombstone, not left as a live chunk | +| 7 | Both-nodes-together restart: clean rejoin, identical counts, no I/O errors | ✅ PASS | Restarted site-a-1 + site-a-2 together: **0** `disk I/O error`/`SQLITE_IOERR`/`corrupt` lines in either log; both re-cached and converged to **identical** counts (ptr `962962dc…`, chunks=2, rowver=3). Byte-identical convergence preserved across a simultaneous restart | +| 8 | site-b default-OFF pin: cache works locally, Healthy, no sync listener on 9001 | ✅ PASS (+1 finding) | site-b-1 cache works locally (`SITE-B / 6e687451`, chunks=1, from its own apply); listeners **`[::]:8080` only — no 9001**; port 9001 **not bound** (`/proc/net/tcp`); `/healthz` **200**; no sync/reconnect metric series. **Finding (follow-up, not a blocker):** `localdb_oplog_depth=5` — because `OnReady` registers replication unconditionally, a default-OFF node captures cache writes into its oplog but has no peer to ack/prune them, so the oplog grows slowly (≈2 rows/deploy, amplified by the re-cache-on-restore fix). Bounded by the library `MaxOplogRows` cap (1M); worth a periodic pruner or skip-if-unchanged in a follow-up | + +## Summary + +**8/8 checks pass** (check 4 with a documented limitation). The gate exercised the real image built +from the branch, the real interceptor over a real loopback+cross-container h2c transport, and real +`docker cp`-triplet DB inspection — and caught **four real defects that every offline test had +missed**, each now fixed on-branch with a regression test: + +1. **`NU1101` packageSourceMapping gap** (`4b2f0e6e`) — `ZB.MOM.WW.LocalDb*` unmapped to the gitea + feed; every clean restore (Docker, CI, fresh clone) failed. Masked locally by the warm NuGet cache. +2. **`ASPNETCORE_HTTP_PORTS` re-bind gap** (`ce9fa07f`) — the sync-listener re-bind ignored the + aspnet:8.0+ container-default port var, silently moving driver health/metrics to loopback:5000. +3. **Empty address space on boot-from-cache** (`9137cb41`) — the rebuild re-read the artifact from the + down ConfigDb and no-op'd, so a cache-booted node logged success but served clients **1 node vs the + healthy peer's 16**. The single most important find: boot-from-cache was half-working. +4. **Cache not repopulated on `RestoreApplied`** (`c6a9f93a`) — a wiped/fresh-volume node recovered its + served state from central but stayed cache-less, unable to boot-from-cache on the next outage. + +**Two documented limitations (follow-ups, not blockers):** + +- **No replication back-fill of a fully-wiped node** (check 4): a peer's already-acked cache rows are + pruned from its oplog and snapshot-resync is gated on the (never-hit) oplog cap, so a wiped node is + not healed by its peer — it self-heals from central on next boot instead. +- **Oplog growth on default-OFF nodes** (check 8): `OnReady` registers replication unconditionally, so + a node with no peer accumulates unacked/unpruned oplog rows (~2/deploy). Slow, bounded by the 1M cap; + a periodic pruner or a skip-if-unchanged cache write would close it. + +Not merged, per the plan. The two limitations are recorded as Phase 1 follow-ups. diff --git a/docs/plans/2026-07-20-localdb-phase1-progress.md b/docs/plans/2026-07-20-localdb-phase1-progress.md new file mode 100644 index 00000000..cafc7621 --- /dev/null +++ b/docs/plans/2026-07-20-localdb-phase1-progress.md @@ -0,0 +1,72 @@ +# LocalDb Phase 1 — Execution Progress (resume state) + +**Branch:** `feat/localdb-phase1` (off `master`). Working tree CLEAN, all work committed, nothing pushed. +**Plan:** `docs/plans/2026-07-20-localdb-adoption-phase1.md` +**Recon:** `docs/plans/2026-07-20-localdb-phase1-recon.md` (READ THIS — it has the 8 deviations D-1..D-8 that override the plan text). +**Skill in use:** `superpowers-extended-cc:executing-plans` — batch, verify, commit per task, report between batches. +**Do NOT merge to master** — plan stops at Task 15 (DoD) and reports; Task 16 is a rig live-gate needing explicit user go-ahead. + +## Status: Tasks 0–9 DONE (10 commits). Tasks 10–16 REMAIN. + +| # | Task | State | Commit | +|---|---|---|---| +| 0 | Preflight + recon | ✅ | `42f85507` | +| 1 | Package refs (LocalDb 0.1.1, Grpc.AspNetCore 2.76.0) | ✅ | `44255fcb` | +| 2 | Schema + LocalDbSetup.OnReady + tests | ✅ | `3b65c24b` | +| 3 | Fail-closed sync auth interceptor + tests | ✅ | `3d29be83` | +| 4 | LocalDbRegistration + Program.cs DI + appsettings | ✅ | `6aff9a83` | +| 5 | h2c sync listener + MapZbLocalDbSync (Kestrel) | ✅ | `b9ddf20e` | +| (fix) | rename Runtime.Deployment → Runtime.DeploymentCache | ✅ | `e771a11a` | +| 6 | IDeploymentArtifactCache + chunked impl + tests | ✅ | `a38a52b8` | +| 9 | Delete LiteDB LocalCache | ✅ | `2bae1b4c` | +| 7 | Cache write path in DriverHostActor | ✅ | `1becf591` | +| 8 | Boot-from-cache + RunningFromCache signal | ✅ | `a27eff32` | +| 10 | Health check + telemetry meter allowlist | ⬜ TODO | | +| 11 | DI-pin integration tests over real host graph | ⬜ TODO | | +| 12 | 2-node convergence harness + scenarios | ⬜ TODO | | +| 13 | docker-dev rig config | ⬜ TODO | | +| 14 | Documentation | ⬜ TODO | | +| 15 | DoD sweep (offline) — STOP + report | ⬜ TODO | | +| 16 | Live gate on docker-dev rig — NEEDS USER GO-AHEAD | ⬜ TODO | | + +Native task list (TaskCreate ids 1–17 map to plan tasks 0–16) tracks the same; tasks.json in plan dir mirrors it. + +## What exists now (key files) + +- `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/DeploymentCacheSchema.cs` — DDL, namespace `Runtime.DeploymentCache`. +- `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/IDeploymentArtifactCache.cs` — `StoreAsync` / `GetCurrentAsync` / **`GetCurrentUnkeyedAsync`** (D-1 addition) + `CachedDeploymentArtifact` record. +- `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/LocalDbDeploymentArtifactCache.cs` — chunked impl. NOTE the prune guard `AND deployment_id <> @DeploymentId` (bug I found: pointer-target could be pruned under same-tick GUIDs → permanent silent miss). +- `src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbSetup.cs` — `OnReady` (DDL→RegisterReplicated). PUBLIC (Host has no InternalsVisibleTo). +- `src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbRegistration.cs` — `AddOtOpcUaLocalDb`, `SyncListenPort(config)`. Registers `IDeploymentArtifactCache → LocalDbDeploymentArtifactCache` singleton. +- `src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbSyncAuthInterceptor.cs` — fail-closed bearer. +- `src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/KestrelHttpBinding.cs` — URL parse/re-bind helper (D-6). +- `src/Server/ZB.MOM.WW.OtOpcUa.Host/Program.cs` — LocalDb wired in `hasDriver` branch (~L176 after AddAlarmHistorian: `AddOtOpcUaLocalDb` + `AddGrpc(interceptor)`); Kestrel re-bind block before `builder.Build()`; `MapZbLocalDbSync()` gated `hasDriver && syncListenPort>0` before `MapOtOpcUaHealth()`. +- `src/Server/ZB.MOM.WW.OtOpcUa.Host/appsettings.json` — `LocalDb` section (Path, SyncListenPort:0, Replication{}), between AlarmHistorian and Deployment. +- `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Drivers/DriverHostActor.cs` — cache dep threaded LAST in Props/ctor/forwarding; `CacheAppliedArtifact` (write, after PushDesiredSubscriptions, empty-blob skip, own try/catch); `TryBootFromCache` + `ApplyCachedArtifact` (in Bootstrap catch); `PushDesiredSubscriptionsFromArtifact` split out; `_isRunningFromCache` field; `SingleClusterCacheKey="__single"`; `ReconcileDrivers` now returns `byte[]?`. +- `src/Server/ZB.MOM.WW.OtOpcUa.Runtime/ServiceCollectionExtensions.cs` — resolves `IDeploymentArtifactCache` (optional, nullable) and passes into DriverHostActor.Props. +- `src/Core/ZB.MOM.WW.OtOpcUa.Commons/Interfaces/NodeDiagnosticsSnapshot.cs` — added `bool RunningFromCache = false`. +- Tests (all green): `Host.IntegrationTests/LocalDbSetupTests.cs`, `LocalDbSyncAuthInterceptorTests.cs`, `LocalDbSyncListenerTests.cs`; `Runtime.Tests/Deployment/LocalDbDeploymentArtifactCacheTests.cs`, `Runtime.Tests/Drivers/DriverHostActorArtifactCacheTests.cs`, `DriverHostActorBootFromCacheTests.cs`. + +## Load-bearing facts for the remaining tasks + +- **Namespace trap (recurs!):** any `Deployment` child namespace under `Runtime` OR `Runtime.Tests` shadows the `Deployment` EF entity → CS0118 across the assembly. Use `...DeploymentCache`. Only a FULL-SOLUTION build catches it (per-project builds pass). +- **xUnit1051 is an error** in these test projects: every `ILocalDb`/DB call in a test must pass `TestContext.Current.CancellationToken` (xunit.v3 Host.IntegrationTests) — but `Runtime.Tests` is **xunit v2** (no `TestContext.Current`). +- **Test project reality (D-5):** NO `Host.Tests` project; NO `WebApplicationFactory` (deliberately — `TwoNodeClusterHarness.cs:48` explains why). Host unit tests live in `Host.IntegrationTests` (xunit.v3). Actor tests in `Runtime.Tests` (xunit v2 + Akka.TestKit.Xunit2). +- **Task 10 (D-7):** `AddOtOpcUaHealth()` takes NO args, called unconditionally at `Program.cs:365`. DON'T change its signature — register the LocalDb check unconditionally and return `Healthy` when LocalDb absent (ActiveNodeHealthCheck precedent). Health check classes: there are ZERO in-repo today; `Host/Health/` has only `HealthEndpoints.cs`. Meter allowlist is REAL and REQUIRED: `Host/Observability/ObservabilityExtensions.cs:30` `o.Meters = [OtOpcUaTelemetry.MeterName]` — add `LocalDbMetrics.MeterName` (from `ZB.MOM.WW.LocalDb.Replication`). Health check reads `ISyncStatus` + `IOptions`. +- **Task 11:** copy the `TwoNodeClusterHarness` direct-host-build pattern, NOT WebApplicationFactory. Pins: `ILocalDb` singleton + `ReplicatedTables` == `["deployment_artifacts","deployment_pointer"]`; `IDeploymentArtifactCache`→concrete; `ISyncStatus` default-OFF (`Connected==false`); admin-only graph has NO `ILocalDb` + doesn't demand `LocalDb:Path`; health check registered. Set `OTOPCUA_ROLES` per case; `LocalDb:Path`→temp file, `ClearAllPools`+delete triplet in cleanup. +- **Task 12:** harness copied from `~/Desktop/ScadaBridge/tests/.../LocalDbSitePairHarness.cs` + `~/Desktop/scadaproj/ZB.MOM.WW.LocalDb/tests/.../Convergence/ConvergenceFixture.cs`. Both nodes init via PRODUCTION `LocalDbSetup.OnReady`. Real loopback Kestrel h2c through the REAL interceptor. DBs as pre-constructed singletons. Positive control: comment out both `RegisterReplicated`, confirm scenarios 1–3 red, restore (don't commit red). +- **Task 13 (D-8):** NO host node has a writable volume today — must ADD named volumes. `docker-dev/docker-compose.yml` services: `central-1/2` (admin,driver, MAIN), `site-a-1/2` (driver, SITE-A), `site-b-1/2` (driver, SITE-B). Env style `Section__Key`. `9001` free container-internal. Enable replication on **site-a pair only** (a-1 initiator w/ PeerAddress, a-2 passive, byte-identical ApiKey, MaxBatchSize=16). site-b stays OFF = default-OFF pin. Validate `docker compose config`; DON'T bring rig up. +- **Task 15:** baseline pre-existing test failures FIRST (`git stash` → run full `dotnet test` → unstash), report deltas only. **KNOWN intermittent:** one `Runtime.Tests` failure seen ~3× across runs (mine + subagent's), never reproduced under trx logger, NOT one of the new tests — investigate/baseline here. +- **Warnings-as-errors** across src + most test projects. Pre-existing CS0618/xUnit1051 warnings exist ONLY in `Client.Shared*` (which don't set TreatWarningsAsErrors) — not mine, ignore. +- **SQLitePCLRaw:** no pin needed — LocalDb 0.1.1 nuspec floors `lib.e_sqlite3` at 2.1.12 (CVE-fixed). Confirmed. + +## Follow-ups logged (NOT in scope, don't action without asking) + +- `Polly.Core` in `Configuration.csproj` now orphaned (ResilientConfigReader was its only real consumer). Left in place — recorded in recon §7.3b. +- Pre-existing: `TryRecoverFromStale` marks Applied w/o reconciling drivers (recon §3.5); `Install-Services.ps1` gives driver-only nodes no `ASPNETCORE_URLS` (recon §6.4, worked around by D-6). + +## Verify-anything-quick + +Full build: `dotnet build ZB.MOM.WW.OtOpcUa.slnx` (expect 0 errors). LocalDb tests: +`dotnet test tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests --filter "FullyQualifiedName~LocalDb"` and +`dotnet test tests/Server/ZB.MOM.WW.OtOpcUa.Runtime.Tests --filter "FullyQualifiedName~ArtifactCache|FullyQualifiedName~BootFromCache"`. diff --git a/docs/plans/2026-07-20-localdb-phase1-recon.md b/docs/plans/2026-07-20-localdb-phase1-recon.md new file mode 100644 index 00000000..b2f90df4 --- /dev/null +++ b/docs/plans/2026-07-20-localdb-phase1-recon.md @@ -0,0 +1,617 @@ +# LocalDb Phase 1 — Task 0 Recon Findings + +**Date:** 2026-07-20 +**Branch:** `feat/localdb-phase1` (based on `master`) +**Plan:** `docs/plans/2026-07-20-localdb-adoption-phase1.md` +**Design authority:** `~/Desktop/scadaproj/docs/plans/2026-07-20-otopcua-localdb-design.md` + +All `file:line` citations are against the branch base commit. + +--- + +## Step 1 — Feed verification: PASS + +``` +$ dotnet package search ZB.MOM.WW.LocalDb --source dohertj2-gitea --format json +ZB.MOM.WW.LocalDb 0.1.1 +ZB.MOM.WW.LocalDb.Contracts 0.1.1 +ZB.MOM.WW.LocalDb.Replication 0.1.1 +``` + +All three present at **0.1.1**. No STOP. (Note: the local NuGet cache holds both `0.1.0` and `0.1.1` +— pin exactly `0.1.1` so a stale cache cannot win.) + +--- + +## Step 3 — STOP conditions: NONE TRIGGERED + +| STOP condition | Verdict | +|---|---| +| Artifact apply path cannot take a post-apply hook without restructuring the actor | **Clear** — `ApplyAndAck` has a clean insertion point (§2.2) | +| `DeploymentId` / cluster identity are not stable strings/GUIDs | **Clear** — both are stable strings, but cluster identity is **not resolvable at boot**; see D-1 | +| LiteDB LocalCache is referenced by live code | **Clear** — zero live-code references (§7) | + +Three STOP conditions clear. **Eight deviations** from the plan's stated assumptions are recorded in +§9 — four of them (D-1, D-3, D-5, D-6) change the work materially. + +--- + +## 1. Deploy fetch path + +### 1.1 Blob-load sites — there are **five** `CreateDbContext()` calls, not two + +| Line | Method | Purpose | +|---|---|---| +| `DriverHostActor.cs:513` | `Bootstrap()` | `NodeDeploymentStates` + `Deployments` at PreStart | +| `DriverHostActor.cs:1472` | `ReconcileDrivers(DeploymentId)` | **loads `ArtifactBlob`** | +| `DriverHostActor.cs:1543` | `PushDesiredSubscriptions(DeploymentId)` | **loads `ArtifactBlob` again** | +| `DriverHostActor.cs:2000` | `TryRecoverFromStale()` | latest `Sealed` deployment (id + hash only, no blob) | +| `DriverHostActor.cs:2029` | `UpsertNodeDeploymentState(...)` | writes `NodeDeploymentState` | + +`DriverHostActor.cs:1472-1476`: +```csharp +using var db = _dbFactory.CreateDbContext(); +blob = db.Deployments.AsNoTracking() + .Where(d => d.DeploymentId == deploymentId.Value) + .Select(d => d.ArtifactBlob) + .FirstOrDefault() ?? Array.Empty(); +``` +`:1543-1547` is byte-identical apart from the local. **The same blob is read twice per apply.** + +### 1.2 Artifact runtime type — there is no instance type + +`DeploymentArtifact` (`DeploymentArtifact.cs:48`) is a **static parser class** over +`ReadOnlySpan`. The blob is never materialised into an object graph. + +- Storage: `byte[]` — `Deployment.cs:29` `public byte[] ArtifactBlob { get; init; } = Array.Empty();` +- Format: **UTF-8 JSON**, `JsonDocument.Parse` (`DeploymentArtifact.cs:66`, `:490`). Not MessagePack. +- Leniency contract: empty/malformed blob → empty result, **never throws** (`:62`, `:537-541`). + +**Consequence for the cache:** we cache raw `byte[]`. No serializer coupling. Good. + +### 1.3 Identity types + +| Type | Definition | Shape | +|---|---|---| +| `DeploymentId` | `Commons/Types/DeploymentId.cs:3` — `readonly record struct DeploymentId(Guid Value)` | `ToString()` → `Value.ToString("N")`, 32 hex chars, **no hyphens** (`:10`) | +| `RevisionHash` | `Commons/Types/RevisionHash.cs:8` — `readonly record struct RevisionHash(string Value)` | SHA-256 hex, **lowercase 64 chars**, no `0x`; empty invalid (`:3-6`) | + +Both are stable, filesystem/SQL-safe strings. Use `DeploymentId.ToString()` (the "N" form) as the +`deployment_id` TEXT column value. + +--- + +## 2. Post-apply point + +### 2.1 `ApplyAndAck` control flow — `DriverHostActor.cs:1413-1458` + +``` +1415 _applyingDeploymentId = deploymentId; +1416 Become(Applying); +1425 UpsertNodeDeploymentState(..., Applying, null) +1427 try { +1429 ReconcileDrivers(deploymentId); // blob fetch #1 — SWALLOWS DB errors +1430 _currentRevision = revision; +1431 UpsertNodeDeploymentState(..., Applied, null); +1432 SendAck(..., ApplyAckOutcome.Applied, null, correlation); +1436 _opcUaPublishActor?.Tell(RebuildAddressSpace(...)); +1439 PushDesiredSubscriptions(deploymentId); // blob fetch #2 — SWALLOWS DB errors +1440 OtOpcUaTelemetry.DeploymentApplied.Add(1, "outcome"="ack"); +1441 _log.Info("applied deployment ..."); +1443 } +1444 catch (Exception ex) { ...Failed ACK... } +1452 finally { _applyingDeploymentId = null; Become(Steady); } +``` + +### 2.2 Where the cache write belongs + +**Immediately after `DriverHostActor.cs:1439`**, before `:1440` — but **wrapped in its own +try/catch**. Reason: `:1432` already sent an `Applied` ACK to the coordinator. An unguarded cache +write that throws would fall into the `catch` at `:1444` and send a *second*, contradictory +`Failed` ACK for a deployment the fleet already believes is applied. + +### 2.3 Paths that must NOT write to the cache + +1. **`catch` at `:1444`** — any apply failure. +2. **Idempotent short-circuit at `:1402-1409`** — `HandleDispatchFromSteady` returns early with an + `Applied` ACK when `_currentRevision == msg.RevisionHash`. `ApplyAndAck` is never entered and no + blob is read. +3. **⚠ THE SILENT-DEGRADATION TRAP.** `ReconcileDrivers:1478-1483` and + `PushDesiredSubscriptions:1549-1553` catch DB failures, log a **warning**, and `return` **without + rethrowing**. `ApplyAndAck` therefore reaches `:1440` and declares success having applied **zero + drivers** when the blob load failed. Caching "whatever we read" would **persist an empty artifact + as a successful apply** — poisoning the boot-from-cache path with a config that starts no drivers. + + **Mitigation (load-bearing):** the cache write must be gated on the blob having actually loaded + non-empty. Have `ReconcileDrivers` surface the loaded blob upward (a field or an out-param) + rather than re-reading it a third time, and skip the cache write when it is + `Array.Empty()`. This also removes the duplicate read noted in §1.1. + +4. **`RestoreApplied` (`:1514-1528`)** — the bootstrap replay path. Restoring from central; caching + here is harmless and arguably desirable, but it is a distinct seam. **Phase 1 scope: skip it.** +5. **`TryRecoverFromStale` (`:1996-2022`)** — never reads the artifact at all (see §3.4). + +--- + +## 3. Cold-boot path + +### 3.1 PreStart — `DriverHostActor.cs:415-436` + +Ctor already did `Become(Steady)` at `:411`. PreStart subscribes to topics, spawns the virtual-tag +and scripted-alarm hosts, then calls `Bootstrap()` at `:435`. + +### 3.2 `Bootstrap()` — `DriverHostActor.cs:506-565` — **yes, it queries central SQL at boot** + +Two queries inside one `try`: latest `NodeDeploymentState` for self (`:514-518`), then the matching +`Deployment` row (`:527-532`). Four outcomes: no prior deployments → Steady; `Applied` → +`RestoreApplied` (`:544`); `Applying` orphan → `ApplyAndAck` replay (`:549`); `Failed` → Steady. + +### 3.3 **THE SEAM** — `DriverHostActor.cs:560-564` + +```csharp +catch (Exception ex) +{ + _log.Warning(ex, "DriverHost {Node}: ConfigDb unreachable on bootstrap; entering Stale", _localNode); + Become(Stale); +} +``` + +**This is the exact point where "central fetch failed at boot" is known.** The cache fallback goes +between `:562` and `:563`: attempt a local-cache load, and only `Become(Stale)` if the cache is also +a miss. + +⚠ The catch also fires if the *second* query throws after the first succeeded, so `latest` is not in +scope-valid state from the catch. **The cache must be self-describing** — it must carry its own +`DeploymentId` + `RevisionHash`, because at this seam the actor has no idea what it should be +restoring. The planned `deployment_pointer` shape already satisfies this. + +### 3.4 The `Stale` state — `DriverHostActor.cs:1343-1379` + +`ReconnectInterval = 30s` (`:55`), retry timer started at `:1378`. + +The dispatch-ignore at **`:1345-1348`** discards `DispatchDeployment` entirely — no ACK, no +buffering (contrast `Applying:599` which does `Self.Forward(msg)`). A Stale node looks like a +timeout to the coordinator. Phase 1 does **not** change this (per design §3.3: a *new* deployment +still requires central). + +`RouteNodeWrite` fast-fails at `:1359-1360` with `"driver host stale (config DB unreachable)"`. + +### 3.5 Pre-existing gap worth recording (not Phase 1 scope) + +`TryRecoverFromStale` (`:1996-2022`) sets `_currentRevision` and marks `Applied` **without calling +`ReconcileDrivers` or `PushDesiredSubscriptions`** — documented at `:1692-1695` and `:1706`. A +Stale-recovered node reports Applied while running **zero drivers**, and +`HandleDispatchFromSteady:1402` then short-circuits on revision match. Boot-from-cache largely +neutralises this in practice, but the bookkeeping remains optimistic. **Logged as a follow-up, not +fixed here.** + +### 3.6 `DbHealthProbeActor` + +**No interaction with `DriverHostActor` whatsoever.** Spawned as a sibling +(`ServiceCollectionExtensions.cs:243-246`), consumed only by `OpcUaPublishActor` +(`OpcUaPublishActor.cs:105`, `:256`, `:528`, `:543-544`). `DriverHostActor` runs its own independent +30s `retry-db` timer. If running-from-cache should later influence OPC UA ServiceLevel, +`OpcUaPublishActor`'s DB-health seam is the precedent to mirror. + +--- + +## 4. Cluster identity — **the one real design problem (see D-1)** + +**There is no `ClusterId` field on `DriverHostActor` and no accessor for it.** The actor knows only +`_localNode` (`DriverHostActor.cs:61`, type `Commons.Types.NodeId`, aliased at `:28`). + +`ClusterId` is resolved **per-artifact**, by `DeploymentArtifact.ResolveClusterScope(blob, nodeId)` +(`DeploymentArtifact.cs:485-516`) — it scans the artifact's `Nodes[]` array for a matching `NodeId` +and reads that element's `ClusterId`. Types: `ClusterScope` (`:46`), `ClusterFilterMode` (`:28`). + +`IClusterRoleInfo` (`Commons/Interfaces/IClusterRoleInfo.cs:12`) exposes `LocalNode`, `LocalRoles`, +`HasRole`, `MembersWithRole`, `RoleLeader`, `RoleLeaderChanged` — **no `ClusterId`**. +`docker-dev/docker-compose.yml` has `Cluster__Hostname`/`Port`/`Roles`/`SeedNodes` — **no +`Cluster__ClusterId`**. The `ClusterNode` entity +(`Configuration/Entities/ClusterNode.cs:7,10`) carries both, but lives in central SQL — exactly +what is unreachable at the seam that needs it. + +> **The bind:** the design keys the cache by `ClusterId` so a replicated pair shares one cache +> entry. But `ClusterId` is only discoverable *from a blob you already have* or *from central SQL*. +> At the §3.3 boot-failure seam we have neither. + +**Resolution adopted (D-1):** keep `cluster_id` as the `deployment_pointer` PK — pair sharing is the +whole point of replication and must not be given up. Resolve it at each end as follows: + +- **Write path** (`ApplyAndAck`, §2.2): call `DeploymentArtifact.ResolveClusterScope(blob, nodeId)` + on the blob just applied. `ClusterFilterMode.ScopeTo` → use its `ClusterId`. `None` (single-cluster + artifact) or `Suppress` → fall back to the literal `"__single"` sentinel, matching the actor's own + existing degenerate-case convention at `:1827-1830`. +- **Read path** (boot seam, §3.3): we cannot compute the key, so **do not key the read**. Select the + single `deployment_pointer` row, ordered by `applied_at_utc DESC`, `LIMIT 1`. A node belongs to + exactly one cluster and its pair peer replicates the same cluster's row, so the table holds one + row in every real topology. If more than one row is present (a node was re-homed between + clusters), take the newest and **log a warning naming both cluster ids** — an operator-visible + signal rather than a silent wrong-config boot. +- Optional escape hatch: honour a `Cluster:ClusterId` config key when present. **Deferred** — not + needed for the rig, and adding a required key contradicts the design's "storage ships + unconditionally" posture. + +This preserves the replication payoff (design §3.3) and the schema exactly as specified, and +confines the change to the read query's `WHERE` clause. + +--- + +## 5. `DriverHostActor` construction — the `Props` expression tree + +### 5.1 The factory — `DriverHostActor.cs:326-346` + +`static Props(...)` takes **16 parameters**, 14 optional, and forwards them **fully positionally** +into `Akka.Actor.Props.Create(() => new DriverHostActor(...))` at **`:343-346`**. Ctor at +`:375-412`, assignments `:393-408`. + +> ⚠ Six `IActorRef?` parameters and three interface-typed parameters mean many mis-bindings are +> **type-compatible and therefore compile clean**. Callers use *named* arguments (safe); the +> internal forwarding at `:343-346` is *positional* (unsafe). +> +> **Rule for Task 7: append the new parameter LAST in the `Props` signature, LAST in the ctor +> signature, and LAST in the positional forwarding list at `:346`. Change nothing else.** + +### 5.2 Construction sites — **27 total** (1 production + 26 test) + +**Production (1):** `Runtime/ServiceCollectionExtensions.cs:332`, inside `WithOtOpcUaRuntimeActors` +(`:193`); registered as `DriverHostActorKey` (`:344`). All args named except the first three. Does +not pass `virtualTagHostOverride`, `scriptedAlarmHostOverride`, or `driverMemberCountProvider`. + +`WithOtOpcUaRuntimeActors` callers: `Host/Program.cs:331`, +`Host.IntegrationTests/TwoNodeClusterHarness.cs:398`, `ServiceCollectionExtensionsTests.cs:42,144`. + +**Tests (26)**, all in `tests/Server/ZB.MOM.WW.OtOpcUa.Runtime.Tests/`: + +| File | Lines | +|---|---| +| `Drivers/DriverHostActorTests.cs` | 28, 60, 81, 118, 143, 189 | +| `Drivers/DriverHostActorReconcileTests.cs` | 37, 59, 90, 119, 156, 193 | +| `Drivers/DriverHostActorDiscoveryTests.cs` | 385, 671, 748, 1023 | +| `Drivers/DriverHostActorWriteRoutingTests.cs` | 148, 208 | +| `Drivers/DriverHostActorPrimaryGateTests.cs` | 228, 247 | +| `Drivers/DriverHostActorVirtualTagTests.cs` | 44, 76 | +| `Drivers/DriverHostActorNativeAlarmTests.cs` | 350 | +| `Drivers/DriverHostActorNativeAlarmAckRoutingTests.cs` | 126 | +| `Drivers/DriverHostActorProbeResultDropTests.cs` | 66 | +| `Drivers/DriverHostActorHistoryWriterTests.cs` | 59 | +| `Drivers/DiscoveryInjectionEndToEndTests.cs` | 219 | +| `VirtualTags/DependencyMuxActorTests.cs` | 139 | + +No site calls `Props.Create()` or `new DriverHostActor(...)` directly — the only +`new DriverHostActor` in the tree is inside the factory at `:343`. Because the new parameter is +optional and appended last, **no test site should need editing.** + +### 5.3 Status object for the running-from-cache signal + +`NodeDiagnosticsSnapshot` (`Commons/Interfaces/NodeDiagnosticsSnapshot.cs:17-22`) — built in +`HandleGetDiagnostics` (`DriverHostActor.cs:1381-1398`), a positional record of +`(NodeId, RevisionHash? CurrentRevision, IReadOnlyList Drivers, DateTime AsOfUtc)`. +`GetDiagnostics` is handled in **all three states** (`Steady:570`, `Applying:601`, `Stale:1349`), so +a flag here is observable regardless of state. **This is the natural home for `RunningFromCache`** +(append last to the record). Consumer: `IFleetDiagnosticsClient` → AdminUI. + +`IDriverHealthPublisher` (`:73`, defaulted `:402`) is **not** a fit — the host never calls it; it +only passes it down to `DriverInstanceActor` children (`:1845`, `:1859`). Per-driver, not per-node. + +### 5.4 Async / side-effect IO convention + +The actor is **overwhelmingly synchronous-blocking on DB IO** — all five `CreateDbContext()` sites +use `using var db = ...` with synchronous LINQ on the actor thread. No `async`/`await` in the file, +no `Task.Run`. + +The single async pattern is `ContinueWith(..., TaskScheduler.Default).PipeTo(replyTo)` in +`HandleRouteNodeWrite` (`:1188-1200`), with an explicit `Sender` capture before the continuation. + +**Recommendation for Task 7:** a **synchronous** cache write at the §2.2 point is *consistent with +the file's existing style* and is correct there — the actor is mid-apply, holds `Applying`, and +every other DB call on that path already blocks. A fire-and-forget `PipeTo(Self)` would need a +handler registered in `Steady`, `Applying` **and** `Stale`, or it dead-letters after the +`Become(Steady)` at `:1456`. **Prefer synchronous + try/catch.** (This deviates from the plan's +"fire-and-forget"; see D-4.) + +`IWithTimers` is already implemented (`:50`, `:282`); `"retry-db"` key in use at `:1378`/`:2009`. + +--- + +## 6. Host composition root + +### 6.1 Role flag — `Program.cs:47-50` + +```csharp +var roles = RoleParser.Parse(Environment.GetEnvironmentVariable("OTOPCUA_ROLES")); +var hasAdmin = roles.Contains("admin"); +var hasDriver = roles.Contains("driver"); +``` +Allowed roles: `admin`, `driver`, `dev` (`Cluster/RoleParser.cs:5-8`); unknown throws (`:25-27`). + +### 6.2 `hasDriver` branch — `Program.cs:131-310` + +Anchor for the new call: **`Program.cs:173-175`** `builder.Services.AddAlarmHistorian(...)`, inside +the config-gated cluster at `:154-195`. Must land **before** `AddAkka` at `:314`. + +### 6.3 Pipeline — `Program.cs:368-404` + +`:383` `MapStaticAssets` · `:385-396` `if (hasAdmin) { ... }` · **`:398` `MapOtOpcUaHealth()`** · +`:399` `MapOtOpcUaMetrics()` · `:404` `RunAsync()`. + +⚠ **There is no `hasDriver` block in the pipeline half** — only `hasAdmin` is branched on. A +driver-gated `app.Map*` is a new construct; it goes between `:396` and `:398`. + +`Program.cs:406-410` re-exports `public partial class Program`. + +### 6.4 Kestrel / URLs — **CONFIRMED: no Kestrel configuration exists** + +- `Program.cs:57` `builder.WebHost.UseStaticWebAssets();` — the **only** `builder.WebHost` call. +- **Zero** `ConfigureKestrel` / `UseUrls` / `ListenAnyIP` / `UseKestrel` anywhere in `src/`. +- Binding is exclusively via `ASPNETCORE_URLS`: `docker-compose.yml:173` (central-1) and `:240` + (central-2) set `http://+:9000`; `launchSettings.json:7` sets `http://localhost:9000`. + +Two traps for Task 5: + +1. ⚠ **`scripts/install/Install-Services.ps1:186`** (`$hostEnv += "ASPNETCORE_URLS=..."`) is inside + `if ($hasAdmin)` at `:185`. **A driver-only Windows-service node gets no `ASPNETCORE_URLS` at + all** and binds the ASP.NET default. The plan's parse-URLs-then-rebind snippet would re-bind + `http://+:9000` there — a port that node never listened on. The `?? "http://+:9000"` fallback in + the plan's snippet is therefore **wrong for driver-only nodes**; see D-6. +2. ⚠ **`Host.IntegrationTests/TwoNodeClusterHarness.cs:331`** already calls + `builder.WebHost.UseKestrel(o => o.Listen(IPAddress.Parse(LoopbackHost), 0));`. Adding a + production `ConfigureKestrel` gives two competing explicit configurations in integration tests — + last-one-wins. The `syncPort > 0` gate keeps this dormant by default (harness sets no sync port), + but any harness that *does* set one will fight the port-0 binding. + +**Health route for smoke tests:** `/healthz` — liveness, runs no checks, always 200 while the +process is up, `AllowAnonymous`. (`Health/HealthEndpoints.cs:48` → `MapZbHealth()`; routes from the +`ZB.MOM.WW.Health` package.) Also `/health/ready`, `/health/active`, and `/metrics` (`Program.cs:399`). + +### 6.5 `SecretsRegistration` template — `Host/Configuration/SecretsRegistration.cs` + +Shape for `LocalDbRegistration` to mirror: `public static class` in +`ZB.MOM.WW.OtOpcUa.Host.Configuration`; `public const string` section/key path constants; a +`public static bool IsXEnabled(IConfiguration)` predicate using +`GetValue(key, defaultValue: false)` (**default-deny**) that `Program.cs` can also call; +`AddOtOpcUaX(this IServiceCollection, IConfiguration)` returning `IServiceCollection`, both args +`ArgumentNullException.ThrowIfNull`-guarded; XML docs with `` explaining *why* the +gate is default-deny. + +Call-site precedent: `Program.cs:363` `AddOtOpcUaSecrets(...)`; predicate reused inside the `AddAkka` +lambda at `Program.cs:323`. + +### 6.6 Telemetry meter allowlist — **EXISTS. Task 10 step 3 is REQUIRED, not conditional.** + +`Host/Observability/ObservabilityExtensions.cs:25-38`: +```csharp +return services.AddZbTelemetry(o => +{ + o.ServiceName = "otopcua"; + o.Meters = [OtOpcUaTelemetry.MeterName]; // <-- ObservabilityExtensions.cs:30 + o.ActivitySources = [OtOpcUaTelemetry.ActivitySourceName]; +``` + +`o.Meters` is a **strict allowlist** — `ZbTelemetryExtensions.cs:51-54` iterates it calling +`metrics.AddMeter(name)`; **no wildcard support**. Anything unlisted is silently not exported. +Single element today: `OtOpcUaTelemetry.MeterName` = `"ZB.MOM.WW.OtOpcUa"` +(`Commons/Observability/OtOpcUaTelemetry.cs:19`). + +**Action:** add `LocalDbMetrics.MeterName` at `ObservabilityExtensions.cs:30`. This is exactly the +omission the ScadaBridge live gate caught. + +### 6.7 Health registration — `Host/Health/HealthEndpoints.cs:19-41` + +`AddOtOpcUaHealth()` **takes no parameters** and is called **unconditionally** at `Program.cs:365`, +outside both role blocks. Three `.AddTypeActivatedCheck<>` calls (`configdb`, `akka`, +`admin-leader`); **no registration-time role gating exists** — `ActiveNodeHealthCheck` does runtime +role scoping instead (returns `Healthy` when the node lacks the role). + +⚠ **There are zero `IHealthCheck` implementations in this repo** — all three come from the +`ZB.MOM.WW.Health.*` packages (`Directory.Packages.props:124-126`, v0.1.0). A LocalDb check would be +the **first health-check class in the tree**; `Host/Health/` holds only `HealthEndpoints.cs` today. + +**Consequence for Task 10:** driver-gating the new check requires changing +`AddOtOpcUaHealth()`'s signature (add `bool hasDriver` or `IConfiguration`) and its `Program.cs:365` +call site. **Preferred alternative:** follow the `ActiveNodeHealthCheck` precedent — register +unconditionally and have the check return `Healthy` when LocalDb is not registered. That keeps the +signature and matches the "default-OFF must not degrade a plain node" requirement for free. See D-7. + +### 6.8 Config files + +| File | Top-level sections | +|---|---| +| `appsettings.json` | `Serilog`, `Security`, `Secrets`, `ServerHistorian`, `ContinuousHistorization`, `AlarmHistorian`, `Deployment` | +| `appsettings.driver.json` / `.admin.json` / `.admin-driver.json` / `.Development.json` | `Serilog`, `Security` only | + +Overlay selection: `Program.cs:67` sorts roles ordinally and joins with `-` → `admin,driver` +becomes `appsettings.admin-driver.json`, added `optional: true` at `:70`; env vars (`:77`) and CLI +args (`:78`) are re-appended after, so deployment overrides outrank the overlay. + +**`LocalDb` key search: CONFIRMED ABSENT.** Case-insensitive `localdb` across all `*.json`, +`*.yml`, `*.ps1`, `*.csproj`, `*.props` matches only the two Phase-1/Phase-2 planning artifacts. No +conflicting keys in any overlay or in `Install-Services.ps1`. + +**Placement convention:** each feature section opens with a `"_comment"` string then `"Enabled": +false`; per-field caveats use sibling `"_Comment"` keys (`appsettings.json:34`, `:45`). Place +`LocalDb` between `AlarmHistorian` (ends `:61`) and `Deployment` (`:62`). + +--- + +## 7. LiteDB LocalCache — dormant, safe to delete + +### 7.1 Files (6) under `src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/` + +`ILocalConfigCache.cs`, `GenerationSnapshot.cs`, `StaleConfigFlag.cs`, `LiteDbConfigCache.cs` +(also declares `LocalConfigCacheCorruptException` at `:128`), `GenerationSealedCache.cs`, +`ResilientConfigReader.cs`. Single namespace `...Configuration.LocalCache`. The only two +`using LiteDB;` statements in `src/` are `LiteDbConfigCache.cs:1` and `GenerationSealedCache.cs:1`. + +### 7.2 Reference classification — **no live-code references. NO STOP.** + +- **(c) live code elsewhere: ZERO.** Also confirms full dormancy: `SealedBootstrap` (referenced by + `docs/v2/v2-release-readiness.md:46`) has **zero hits** in `src/` and `tests/` — it no longer exists. +- Two benign near-misses: + - `Configuration/Services/ILdapGroupRoleMappingService.cs:14` — the string `ResilientConfigReader` + inside an **XML doc comment**. Prose only. Needs a comment edit or the doc goes stale. + - `Configuration/ZB.MOM.WW.OtOpcUa.Configuration.csproj:29` — ``, + the only one in the repo. +- Dozens of `LocalConfigCacheCorruptException` hits under `src/**/bin/**/*.xml` are **generated doc + build artifacts** (`GenerateDocumentationFile=true`), not source. + +### 7.3 ⚠ Test-file drift — the plan's delete list is incomplete + +Plan `…phase1.md:484-485` names only `GenerationSealedCacheTests.cs` and +`ResilientConfigReaderTests.cs`. Actual test files in +`tests/Core/ZB.MOM.WW.OtOpcUa.Configuration.Tests/`: + +| File | Status | +|---|---| +| `GenerationSealedCacheTests.cs` | in plan ✅ | +| `ResilientConfigReaderTests.cs` | in plan ✅ — note `StaleConfigFlagTests` is co-located here at `:323` | +| **`LiteDbConfigCacheTests.cs`** | **MISSING from the plan** — will fail to compile if the source is deleted (`:125` calls `new LiteDB.LiteDatabase(...)` directly) | + +See D-3. + +### 7.3b Outcome (Task 9, executed 2026-07-20) + +Deleted: the 6 `LocalCache/` sources and **all three** test files (including +`LiteDbConfigCacheTests.cs`, which the plan omitted — D-3). `LiteDB` removed from both +`Configuration.csproj` and `Directory.Packages.props`. The stale XML-doc reference at +`ILdapGroupRoleMappingService.cs:14` was rewritten to state the present truth (no fallback exists; +reviving it means an admin-side cache on LocalDb, not restoring the old pipeline) rather than +deleted, per the DoD's "explanatory prose may remain". Configuration tests: 92/92 green. + +**Follow-up found while executing:** `Polly.Core` in `Configuration.csproj` is now orphaned too — +`ResilientConfigReader` was its only real consumer in that project (remaining `Polly` mentions are +comments). Left in place deliberately: removing it is outside Task 9's scope and could break a +consumer relying on the transitive flow. Worth a separate cleanup. + +### 7.4 Package removal + +- `PackageReference`: `Configuration.csproj:29` — only consumer. +- `PackageVersion`: `Directory.Packages.props:35` — `LiteDB` `5.0.21`. + +Both safe to delete; no transitive consumer. + +--- + +## 8. Test projects, rig, packages + +### 8.1 Test project layout + +| Question | Answer | +|---|---| +| `tests/Server/ZB.MOM.WW.OtOpcUa.Host.Tests/` | **DOES NOT EXIST** — there is no Host *unit*-test project at all | +| Closest Host project | `Host.IntegrationTests` — despite the name it carries plenty of pure unit tests (`LdapOptionsValidatorTests`, `ServerHistorianOptionsValidatorTests`, `ResilienceInvokerFactoryRegistrationTests`, …) | +| `tests/Server/ZB.MOM.WW.OtOpcUa.Runtime.Tests/` | **EXISTS** | +| DriverHostActor tests | `Runtime.Tests/Drivers/` — 11 `DriverHostActor*Tests.cs` files. **xunit v2** (`xunit` 2.9.3), references **`Akka.TestKit.Xunit2`** ✅ | +| `Host.IntegrationTests` xunit version | **xunit.v3** 1.1.0; SDK `Microsoft.NET.Sdk.Web` | + +`Directory.Packages.props` documents that AdminUI.Tests / ControlPlane.Tests / **Runtime.Tests** are +the three xunit-v2 holdouts, held there by `Akka.TestKit.Xunit2` (verified 2026-07-13: no xunit.v3 +Akka TestKit exists). This matches the design's "actor tests live in an xunit2 TestKit project". + +### 8.2 ⚠ `WebApplicationFactory` is **deliberately not used** — Task 11 must be rewritten + +The only mention of the type in the entire `tests/` tree is a comment explaining why it is avoided: + +`Host.IntegrationTests/TwoNodeClusterHarness.cs:48` +``` +/// Why not WebApplicationFactory<Program>? Program.cs reads OTOPCUA_ROLES ... +``` + +The actual pattern builds the host directly — `TwoNodeClusterHarness.cs:329`: +```csharp +var builder = WebApplication.CreateBuilder(new WebApplicationOptions { Args = [] }); +``` +`Microsoft.AspNetCore.Mvc.Testing` is still referenced (csproj `:15`) but only for transitive +test-host plumbing. **Copy `TwoNodeClusterHarness`, not `WebApplicationFactory`.** See D-5. + +`Host.IntegrationTests` also has its own `docker-compose.yml` (sql `14331`, ldap `3894`) gated by +`DockerFixtureAvailability.cs`. + +### 8.3 docker-dev rig topology — plan's site-a/site-b assumption is **correct** + +Single Akka mesh (`otopcua`), seeded by `central-1`. Tenant separation is by `ServerCluster.ClusterId` +rows (MAIN / SITE-A / SITE-B) inside **one shared ConfigDb** — not separate meshes. + +| Service | Roles | Cluster | +|---|---|---| +| `central-1`, `central-2` | `admin,driver` | MAIN | +| `site-a-1`, `site-a-2` | `driver` | SITE-A | +| `site-b-1`, `site-b-2` | `driver` | SITE-B | +| `sql`, `migrator`, `cluster-seed`, `traefik` | — | infra | + +**There is no admin-only service** — the "admin" nodes are fused `admin,driver`. All six host nodes +share the `&otopcua-host` YAML anchor. + +⚠ **NO HOST NODE HAS A WRITABLE VOLUME.** The only `volumes:` in the file are `sql` → +`otopcua-mssql-data`, plus two read-only binds (`./seed:ro`, `./traefik-dynamic.yml:ro`). All six +nodes write to the container's ephemeral layer, destroyed on recreate. This is already an accepted +pattern (per-container `Secrets:SqlitePath=otopcua-secrets.db`, replicated over Akka rather than +persisted). **Task 13 must add six named volumes** — there is nothing to piggyback on. See D-8. + +**Env style: confirmed double-underscore `Section__Key`**, with array indices and dictionary keys as +a third segment (`Cluster__SeedNodes__0`, `Security__Ldap__GroupToRole__ReadOnly`). Exceptions that +are plain SCREAMING_SNAKE (read directly, not config-bound): `OTOPCUA_ROLES`, +`ZB_SECRETS_MASTER_KEY`, `GALAXY_MXGW_API_KEY`, `OTOPCUA_CONFIG_CONNECTION`, `ASPNETCORE_URLS`. + +**Ports.** Host-published: `14330`→sql, `4840`-`4845`→the six nodes' OPC UA, `9200`→traefik web, +`8089`→traefik dashboard. Container-internal on every node: `4053` (Akka remoting), `9000` +(Kestrel), `4840` (OPC UA). Also reserved on this machine: `14331`/`3894` (Host.IntegrationTests +compose), `80`/`8080` (sibling scadabridge/scadalink stack), `10.100.0.35:3893` (shared GLAuth). + +**`9001` (the plan's choice) is free** container-internal and need not be published. ✅ + +### 8.4 Package versions — `Directory.Packages.props` + +| Package | Version | Line | +|---|---|---| +| `Grpc.Core.Api` | 2.76.0 | 32 | +| `Grpc.Net.Client` | 2.76.0 | 33 | +| **`Grpc.AspNetCore`** | **NOT PRESENT** | — | +| `Google.Protobuf` | 3.34.1 | 31 | +| `Microsoft.Data.Sqlite` | 10.0.7 | ~55 | +| `SQLitePCLRaw.bundle_e_sqlite3` | 2.1.12 | ~108 | +| `LiteDB` | 5.0.21 | 35 | + +Both Grpc floors and the Protobuf floor are already satisfied — **no bumps needed**, only the new +`Grpc.AspNetCore` entry (use 2.76.0 to match the train; watch for NU1605/NU1608 if it wants a +Protobuf newer than the pinned 3.34.1). + +⚠ **`SQLitePCLRaw` is a surgical direct pin, not a transitive one.** +`CentralPackageTransitivePinningEnabled` is deliberately **off** (it breaks the Roslyn version +split). The 2.1.12 pin is promoted by a **direct `PackageReference` in `Core.AlarmHistorian`** +(`Core.AlarmHistorian.csproj:15` + `:21`) — today the only Sqlite consumer in `src/`. + +> **Any new project that pulls `Microsoft.Data.Sqlite` MUST also take a direct +> ``,** or it silently resolves the +> vulnerable 2.1.11 native bundle. This applies to **Host** and **Runtime** (both gain `ZB.MOM.WW.LocalDb`, +> which depends on `Microsoft.Data.Sqlite`) and to every test project that builds a real `ILocalDb`. +> Copy the `Core.AlarmHistorian.csproj:15,21` pair exactly. The plan's Task 1 mentions +> `SQLitePCLRaw.lib.e_sqlite3` "if the audit flags it" — the correct package is +> **`bundle_e_sqlite3`**, and it is **required, not conditional**. + +--- + +## 9. Deviations from the plan + +Recorded per the tasks.json `deviation` convention. **D-1, D-3, D-5, D-6 change the work +materially.** + +| # | Plan says | Reality | Resolution | +|---|---|---|---| +| **D-1** | Cache keyed by `ClusterId`; "cluster identity" is a recon item with a STOP condition | `ClusterId` is stable, but **not resolvable at the boot seam** — it lives inside the artifact or in the unreachable central DB (§4) | Keep `cluster_id` as PK (pair sharing is the point). **Write:** resolve via `DeploymentArtifact.ResolveClusterScope`, `"__single"` sentinel for the degenerate case. **Read:** unkeyed `ORDER BY applied_at_utc DESC LIMIT 1`, warn if >1 row. No STOP. | +| **D-2** | Cache write "fire-and-forget (`PipeTo`-style or guarded `Task.Run`)" | The actor has **no async DB IO at all**; a `PipeTo(Self)` needs handlers in all three states or it dead-letters after `Become(Steady)` (§5.4) | Synchronous write in its own try/catch at `:1439`. Matches the file's style; failure still cannot fail the apply. | +| **D-3** | Task 9 deletes 2 test files | **3** test files reference the subsystem — `LiteDbConfigCacheTests.cs` is missing from the list (§7.3) | Delete all three. Also fix the stale XML-doc mention at `ILdapGroupRoleMappingService.cs:14`. | +| **D-4** | Cache the artifact after a successful apply | `ReconcileDrivers`/`PushDesiredSubscriptions` **swallow** DB errors, so a "successful" apply can have loaded an **empty blob** and started zero drivers (§2.3) | Gate the write on a non-empty blob; surface the blob from `ReconcileDrivers` instead of a third read. **Load-bearing — without it the cache can be poisoned with an empty config.** | +| **D-5** | Task 11 uses `WebApplicationFactory` | Deliberately **not used** in this repo; `TwoNodeClusterHarness.cs:48` documents why (Program.cs reads `OTOPCUA_ROLES` from the environment) (§8.2) | Rewrite Task 11 against the `TwoNodeClusterHarness` direct-host-build pattern. | +| **D-6** | Task 5 re-binds `ASPNETCORE_URLS ?? "http://+:9000"` | `Install-Services.ps1:185-186` sets `ASPNETCORE_URLS` **only when `hasAdmin`** — the fallback would bind a driver-only node to a port it never served (§6.4) | Re-bind only when a URL is actually configured; when absent, add **only** the sync listener. Also note `TwoNodeClusterHarness.cs:331` already calls `UseKestrel(Listen(…, 0))`. | +| **D-7** | Task 10 driver-gates the health check | `AddOtOpcUaHealth()` takes no args and is called unconditionally at `Program.cs:365`; no registration-time gating exists anywhere (§6.7) | Register unconditionally; return `Healthy` when LocalDb is absent — the `ActiveNodeHealthCheck` precedent. Avoids a signature change and satisfies "default-OFF must not degrade a plain node". | +| **D-8** | Task 13 "existing data mount" | **No host node has any writable volume** (§8.3) | Add six named volumes. Note: `9001` is free container-internal. Meter allowlist (§6.6) is **required**, not conditional. | + +### Follow-ups logged, not fixed here + +1. **`TryRecoverFromStale` marks `Applied` without reconciling drivers** (§3.5) — pre-existing, + documented in-code at `:1692-1695`. Out of Phase 1 scope. +2. **Duplicate blob read per apply** (`:1472` and `:1543`, §1.1) — D-4's fix removes one of them as a + side effect. +3. **`Install-Services.ps1` gives driver-only nodes no `ASPNETCORE_URLS`** (§6.4) — pre-existing; + D-6 works around it rather than fixing it. diff --git a/src/Core/ZB.MOM.WW.OtOpcUa.Commons/Interfaces/NodeDiagnosticsSnapshot.cs b/src/Core/ZB.MOM.WW.OtOpcUa.Commons/Interfaces/NodeDiagnosticsSnapshot.cs index e31b62ff..a2fafebd 100644 --- a/src/Core/ZB.MOM.WW.OtOpcUa.Commons/Interfaces/NodeDiagnosticsSnapshot.cs +++ b/src/Core/ZB.MOM.WW.OtOpcUa.Commons/Interfaces/NodeDiagnosticsSnapshot.cs @@ -14,8 +14,16 @@ public sealed record DriverInstanceDiagnostics( /// Per-node diagnostics returned by IFleetDiagnosticsClient. Populated by the node's /// local DriverHostActor via a request/response over Akka. /// +/// +/// True when the node booted its configuration from the node-local artifact cache because the +/// central ConfigDb was unreachable. Such a node looks entirely healthy — full address space, +/// live values — but its configuration is frozen and no deployment can reach it, so the state +/// needs to be explicitly visible rather than inferred. Defaults to false, which is also what +/// every node that booted normally reports. +/// public sealed record NodeDiagnosticsSnapshot( NodeId NodeId, RevisionHash? CurrentRevision, IReadOnlyList Drivers, - DateTime AsOfUtc); + DateTime AsOfUtc, + bool RunningFromCache = false); diff --git a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/GenerationSealedCache.cs b/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/GenerationSealedCache.cs deleted file mode 100644 index 220b9f9b..00000000 --- a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/GenerationSealedCache.cs +++ /dev/null @@ -1,197 +0,0 @@ -using LiteDB; - -namespace ZB.MOM.WW.OtOpcUa.Configuration.LocalCache; - -/// -/// Generation-sealed LiteDB cache per docs/v2/plan.md and Phase 6.1 -/// Stream D.1. Each published generation writes one read-only LiteDB file under -/// <cache-root>/<clusterId>/<generationId>.db. A per-cluster -/// CURRENT text file holds the currently-active generation id; it is updated -/// atomically (temp file + ) only after -/// the sealed file is fully written. -/// -/// -/// Mixed-generation reads are impossible: any read opens the single file pointed to -/// by CURRENT, which is a coherent snapshot. Corruption of the CURRENT file or the -/// sealed file surfaces as — the reader -/// fails closed rather than silently falling back to an older generation. Recovery path -/// is to re-fetch from the central DB (and the Phase 6.1 Stream C UsingStaleConfig -/// flag goes true until that succeeds). -/// -/// This cache is the read-path fallback when the central DB is unreachable. The -/// write path (draft edits, publish) bypasses the cache and fails hard on DB outage per -/// Stream D.2 — inconsistent writes are worse than a temporary inability to edit. -/// -public sealed class GenerationSealedCache -{ - private const string CollectionName = "generation"; - private const string CurrentPointerFileName = "CURRENT"; - private readonly string _cacheRoot; - - // Private per-database BsonMapper with the entity pre-registered. LiteDB's default - // BsonMapper.Global is a process-wide singleton whose lazy per-type member registration is - // not thread-safe across concurrently-constructed LiteDatabase instances; a seal racing a - // read (or this cache racing LiteDbConfigCache) corrupts the global mapper, surfacing as - // "Member … not found on BsonMapper" or a bogus duplicate-_id insert. - private static BsonMapper BuildMapper() - { - var mapper = new BsonMapper(); - mapper.Entity(); - return mapper; - } - - /// Root directory for all clusters' sealed caches. - public string CacheRoot => _cacheRoot; - - /// Initializes a new instance of the GenerationSealedCache class. - /// The root directory for the cache. - public GenerationSealedCache(string cacheRoot) - { - ArgumentException.ThrowIfNullOrWhiteSpace(cacheRoot); - _cacheRoot = cacheRoot; - Directory.CreateDirectory(_cacheRoot); - } - - /// - /// Seal a generation: write the snapshot to <cluster>/<generationId>.db, - /// mark the file read-only, then atomically publish the CURRENT pointer. Existing - /// sealed files for prior generations are preserved (prune separately). - /// - /// The generation snapshot to seal. - /// The cancellation token. - /// A task representing the asynchronous operation. - public async Task SealAsync(GenerationSnapshot snapshot, CancellationToken ct = default) - { - ArgumentNullException.ThrowIfNull(snapshot); - ct.ThrowIfCancellationRequested(); - - var clusterDir = Path.Combine(_cacheRoot, snapshot.ClusterId); - Directory.CreateDirectory(clusterDir); - var sealedPath = Path.Combine(clusterDir, $"{snapshot.GenerationId}.db"); - - if (File.Exists(sealedPath)) - { - // Already sealed — idempotent. Treat as no-op + update pointer in case an earlier - // seal succeeded but the pointer update failed (crash recovery). - WritePointerAtomically(clusterDir, snapshot.GenerationId); - return; - } - - var tmpPath = sealedPath + ".tmp"; - try - { - using (var db = new LiteDatabase(new ConnectionString { Filename = tmpPath, Upgrade = false }, BuildMapper())) - { - var col = db.GetCollection(CollectionName); - col.Insert(snapshot); - } - - File.Move(tmpPath, sealedPath); - File.SetAttributes(sealedPath, File.GetAttributes(sealedPath) | FileAttributes.ReadOnly); - WritePointerAtomically(clusterDir, snapshot.GenerationId); - } - catch - { - try { if (File.Exists(tmpPath)) File.Delete(tmpPath); } catch { /* best-effort */ } - throw; - } - - await Task.CompletedTask; - } - - /// - /// Read the current sealed snapshot for . Throws - /// when the pointer is missing - /// (first-boot-no-snapshot case) or when the sealed file is corrupt. Never silently - /// falls back to a prior generation. - /// - /// The cluster ID to read the snapshot for. - /// The cancellation token. - /// A task representing the asynchronous operation containing the generation snapshot. - public Task ReadCurrentAsync(string clusterId, CancellationToken ct = default) - { - ArgumentException.ThrowIfNullOrWhiteSpace(clusterId); - ct.ThrowIfCancellationRequested(); - - var clusterDir = Path.Combine(_cacheRoot, clusterId); - var pointerPath = Path.Combine(clusterDir, CurrentPointerFileName); - if (!File.Exists(pointerPath)) - throw new GenerationCacheUnavailableException( - $"No sealed generation for cluster '{clusterId}' at '{clusterDir}'. First-boot case: the central DB must be reachable at least once before cache fallback is possible."); - - long generationId; - try - { - var text = File.ReadAllText(pointerPath).Trim(); - generationId = long.Parse(text, System.Globalization.CultureInfo.InvariantCulture); - } - catch (Exception ex) - { - throw new GenerationCacheUnavailableException( - $"CURRENT pointer at '{pointerPath}' is corrupt or unreadable.", ex); - } - - var sealedPath = Path.Combine(clusterDir, $"{generationId}.db"); - if (!File.Exists(sealedPath)) - throw new GenerationCacheUnavailableException( - $"CURRENT points at generation {generationId} but '{sealedPath}' is missing — fails closed rather than serving an older generation."); - - try - { - using var db = new LiteDatabase(new ConnectionString { Filename = sealedPath, ReadOnly = true }, BuildMapper()); - var col = db.GetCollection(CollectionName); - var snapshot = col.FindAll().FirstOrDefault() - ?? throw new GenerationCacheUnavailableException( - $"Sealed file '{sealedPath}' contains no snapshot row — file is corrupt."); - return Task.FromResult(snapshot); - } - catch (GenerationCacheUnavailableException) { throw; } - catch (Exception ex) when (ex is LiteException or InvalidDataException or IOException - or NotSupportedException or FormatException) - { - throw new GenerationCacheUnavailableException( - $"Sealed file '{sealedPath}' is corrupt or unreadable — fails closed rather than falling back to an older generation.", ex); - } - } - - /// Return the generation id the CURRENT pointer points at, or null if no pointer exists. - /// The cluster ID to get the current generation ID for. - /// The generation ID, or null if no pointer exists. - public long? TryGetCurrentGenerationId(string clusterId) - { - ArgumentException.ThrowIfNullOrWhiteSpace(clusterId); - var pointerPath = Path.Combine(_cacheRoot, clusterId, CurrentPointerFileName); - if (!File.Exists(pointerPath)) return null; - try - { - return long.Parse(File.ReadAllText(pointerPath).Trim(), System.Globalization.CultureInfo.InvariantCulture); - } - catch - { - return null; - } - } - - private static void WritePointerAtomically(string clusterDir, long generationId) - { - var pointerPath = Path.Combine(clusterDir, CurrentPointerFileName); - var tmpPath = pointerPath + ".tmp"; - File.WriteAllText(tmpPath, generationId.ToString(System.Globalization.CultureInfo.InvariantCulture)); - if (File.Exists(pointerPath)) - File.Replace(tmpPath, pointerPath, destinationBackupFileName: null); - else - File.Move(tmpPath, pointerPath); - } -} - -/// Sealed cache is unreachable — caller must fail closed. -public sealed class GenerationCacheUnavailableException : Exception -{ - /// Initializes a new instance of the GenerationCacheUnavailableException class. - /// The error message. - public GenerationCacheUnavailableException(string message) : base(message) { } - /// Initializes a new instance of the GenerationCacheUnavailableException class with an inner exception. - /// The error message. - /// The inner exception. - public GenerationCacheUnavailableException(string message, Exception inner) : base(message, inner) { } -} diff --git a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/GenerationSnapshot.cs b/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/GenerationSnapshot.cs deleted file mode 100644 index b883719d..00000000 --- a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/GenerationSnapshot.cs +++ /dev/null @@ -1,20 +0,0 @@ -namespace ZB.MOM.WW.OtOpcUa.Configuration.LocalCache; - -/// -/// A self-contained snapshot of one generation — enough to rebuild the address space on a node -/// that has lost DB connectivity. The payload is the JSON-serialized sp_GetGenerationContent -/// result; the local cache doesn't inspect the shape, it just round-trips bytes. -/// -public sealed class GenerationSnapshot -{ - /// Gets or sets the auto-generated LiteDB ID. - public int Id { get; set; } // LiteDB auto-ID - /// Gets or sets the cluster identifier. - public required string ClusterId { get; set; } - /// Gets or sets the generation identifier. - public required long GenerationId { get; set; } - /// Gets or sets the time this snapshot was cached. - public required DateTime CachedAt { get; set; } - /// Gets or sets the JSON-serialized payload content. - public required string PayloadJson { get; set; } -} diff --git a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/ILocalConfigCache.cs b/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/ILocalConfigCache.cs deleted file mode 100644 index 9483de8d..00000000 --- a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/ILocalConfigCache.cs +++ /dev/null @@ -1,32 +0,0 @@ -namespace ZB.MOM.WW.OtOpcUa.Configuration.LocalCache; - -/// -/// Per-node local cache of the most-recently-applied generation(s). Used to bootstrap the -/// address space when the central DB is unreachable (degraded-but-running). -/// -/// -/// Concurrency contract: implementations must serialize writes — specifically, -/// for the same (ClusterId, GenerationId) from concurrent -/// callers must not produce duplicate rows. Reads may run concurrently with reads and writes. -/// The implementation enforces this via an instance-level -/// around the find-then-insert/update window. -/// -public interface ILocalConfigCache -{ - /// Retrieves the most recent generation snapshot for the specified cluster. - /// The cluster identifier. - /// The cancellation token. - /// The most recent generation snapshot, or null if none exists. - Task GetMostRecentAsync(string clusterId, CancellationToken ct = default); - /// Stores a generation snapshot in the local cache. - /// The generation snapshot to store. - /// The cancellation token. - /// A task that represents the asynchronous operation. - Task PutAsync(GenerationSnapshot snapshot, CancellationToken ct = default); - /// Removes old generations, keeping only the most recent N. - /// The cluster identifier. - /// The number of latest generations to keep. - /// The cancellation token. - /// A task that represents the asynchronous operation. - Task PruneOldGenerationsAsync(string clusterId, int keepLatest = 10, CancellationToken ct = default); -} diff --git a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/LiteDbConfigCache.cs b/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/LiteDbConfigCache.cs deleted file mode 100644 index 4a597b12..00000000 --- a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/LiteDbConfigCache.cs +++ /dev/null @@ -1,129 +0,0 @@ -using LiteDB; - -namespace ZB.MOM.WW.OtOpcUa.Configuration.LocalCache; - -/// -/// LiteDB-backed . One file per node (default -/// config_cache.db), one collection per snapshot. Corruption surfaces as -/// on construction or read — callers should -/// delete and re-fetch from the central DB. -/// -public sealed class LiteDbConfigCache : ILocalConfigCache, IDisposable -{ - private const string CollectionName = "generations"; - - // LiteDB's default BsonMapper.Global is a process-wide singleton whose per-type member - // registration is lazy and NOT thread-safe across concurrently-constructed LiteDatabase - // instances. When several caches (this one + GenerationSealedCache) initialise in parallel - // the global mapper races, surfacing as "Member ClusterId not found on BsonMapper" or a - // bogus "duplicate key _id = 0" (the int auto-id mapping was lost so Insert writes a literal - // 0 twice). Give each database a private, pre-registered mapper so member - // resolution happens once, single-threaded, at construction and never touches the global. - private static BsonMapper BuildMapper() - { - var mapper = new BsonMapper(); - mapper.Entity(); - return mapper; - } - - private readonly LiteDatabase _db; - private readonly ILiteCollection _col; - // PutAsync is a find-then-insert/update; without serialization, two concurrent puts for the - // same (ClusterId, GenerationId) can both observe `existing is null` and both Insert, - // producing duplicate rows. Serialize writes through this semaphore so - // the read-modify-write block is atomic for a given instance. LiteDB itself only locks the - // page-level write, not the find-then-insert window. - private readonly SemaphoreSlim _writeGate = new(initialCount: 1, maxCount: 1); - - /// Initializes a new instance of the class. - /// Path to the LiteDB database file. - public LiteDbConfigCache(string dbPath) - { - // LiteDB can be tolerant of header-only corruption at construction time (it may overwrite - // the header and "recover"), so we force a write + read probe to fail fast on real corruption. - try - { - _db = new LiteDatabase(new ConnectionString { Filename = dbPath, Upgrade = true }, BuildMapper()); - _col = _db.GetCollection(CollectionName); - _col.EnsureIndex(s => s.ClusterId); - _col.EnsureIndex(s => s.GenerationId); - _ = _col.Count(); - } - catch (Exception ex) when (ex is LiteException or InvalidDataException or IOException - or NotSupportedException or UnauthorizedAccessException - or ArgumentOutOfRangeException or FormatException) - { - _db?.Dispose(); - throw new LocalConfigCacheCorruptException( - $"LiteDB cache at '{dbPath}' is corrupt or unreadable — delete the file and refetch from the central DB.", - ex); - } - } - - /// - public Task GetMostRecentAsync(string clusterId, CancellationToken ct = default) - { - ct.ThrowIfCancellationRequested(); - var snapshot = _col - .Find(s => s.ClusterId == clusterId) - .OrderByDescending(s => s.GenerationId) - .FirstOrDefault(); - return Task.FromResult(snapshot); - } - - /// - public async Task PutAsync(GenerationSnapshot snapshot, CancellationToken ct = default) - { - ct.ThrowIfCancellationRequested(); - // Serialize the find-then-insert/update so concurrent callers do not observe a stale - // `existing is null` and both Insert. LiteDB's per-call lock is not enough — the - // read and the write are independent calls. - await _writeGate.WaitAsync(ct).ConfigureAwait(false); - try - { - // upsert by (ClusterId, GenerationId) — replace in place if already cached - var existing = _col - .Find(s => s.ClusterId == snapshot.ClusterId && s.GenerationId == snapshot.GenerationId) - .FirstOrDefault(); - - if (existing is null) - _col.Insert(snapshot); - else - { - snapshot.Id = existing.Id; - _col.Update(snapshot); - } - } - finally - { - _writeGate.Release(); - } - } - - /// - public Task PruneOldGenerationsAsync(string clusterId, int keepLatest = 10, CancellationToken ct = default) - { - ct.ThrowIfCancellationRequested(); - var doomed = _col - .Find(s => s.ClusterId == clusterId) - .OrderByDescending(s => s.GenerationId) - .Skip(keepLatest) - .Select(s => s.Id) - .ToList(); - - foreach (var id in doomed) - _col.Delete(id); - - return Task.CompletedTask; - } - - /// Releases all resources used by the cache. - public void Dispose() - { - _writeGate.Dispose(); - _db.Dispose(); - } -} - -public sealed class LocalConfigCacheCorruptException(string message, Exception inner) - : Exception(message, inner); diff --git a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/ResilientConfigReader.cs b/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/ResilientConfigReader.cs deleted file mode 100644 index 2c55b199..00000000 --- a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/ResilientConfigReader.cs +++ /dev/null @@ -1,140 +0,0 @@ -using System.Text.RegularExpressions; -using Microsoft.Extensions.Logging; -using Polly; -using Polly.Retry; -using Polly.Timeout; - -namespace ZB.MOM.WW.OtOpcUa.Configuration.LocalCache; - -/// -/// Wraps a central-DB fetch function with Phase 6.1 Stream D.2 resilience: -/// timeout 2 s → retry 3× jittered → fallback to sealed cache. Maintains the -/// — fresh on central-DB success, stale on cache fallback. -/// -/// -/// Read-path only per plan. The write path (draft save, publish) bypasses this -/// wrapper entirely and fails hard on DB outage so inconsistent writes never land. -/// -/// Fallback is triggered by any exception the fetch raises (central-DB -/// unreachable, SqlException, timeout). If the sealed cache also fails (no pointer, -/// corrupt file, etc.), surfaces — caller -/// must fail the current request (InitializeAsync for a driver, etc.). -/// -public sealed class ResilientConfigReader -{ - private readonly GenerationSealedCache _cache; - private readonly StaleConfigFlag _staleFlag; - private readonly ResiliencePipeline _pipeline; - private readonly ILogger _logger; - - /// Initializes a resilient config reader with the given cache and options. - /// The sealed cache for fallback. - /// The stale config flag to manage. - /// The logger instance. - /// The timeout for central fetch (default 2s). - /// The number of retries (default 3). - public ResilientConfigReader( - GenerationSealedCache cache, - StaleConfigFlag staleFlag, - ILogger logger, - TimeSpan? timeout = null, - int retryCount = 3) - { - _cache = cache; - _staleFlag = staleFlag; - _logger = logger; - var builder = new ResiliencePipelineBuilder() - .AddTimeout(new TimeoutStrategyOptions { Timeout = timeout ?? TimeSpan.FromSeconds(2) }); - - if (retryCount > 0) - { - builder.AddRetry(new RetryStrategyOptions - { - MaxRetryAttempts = retryCount, - BackoffType = DelayBackoffType.Exponential, - UseJitter = true, - Delay = TimeSpan.FromMilliseconds(100), - MaxDelay = TimeSpan.FromSeconds(1), - // Handle ALL exceptions including OperationCanceledException. A SQL command-level - // timeout surfaces as TaskCanceledException (derives from OperationCanceledException) - // when the caller's token is NOT cancelled, and must be retried just like any other - // transient error. Polly itself checks the cancellation token between retries and - // stops with OperationCanceledException on genuine caller cancellation regardless of - // this predicate. - ShouldHandle = new PredicateBuilder().Handle(), - }); - } - - _pipeline = builder.Build(); - } - - /// - /// Redacts connection-string fragments (Password, User Id, Pwd, etc.) - /// that a caller's exception message could carry. Conservative regex pass — anything - /// matching Key=Value with a known credential key gets its value replaced. - /// - private static readonly Regex SecretsRegex = new( - @"(?ix)\b(Password|Pwd|User\s*Id|Uid|AccessToken|Authorization|Api[-_]?Key)\s*=\s*[^;,)\s]*", - RegexOptions.Compiled); - - /// Redacts sensitive credential information from a message. - /// The message to scrub. - /// The message with redacted credentials. - internal static string ScrubSecrets(string? message) - { - if (string.IsNullOrEmpty(message)) return message ?? string.Empty; - // Replace the entire matched fragment (key + value) with a redaction marker so the - // key name itself doesn't leak — log scrapers grep for "Password=" too. - return SecretsRegex.Replace(message, "[redacted credential]"); - } - - /// - /// Executes a central fetch through the resilience pipeline. On full failure - /// (post-retry), reads the sealed cache and extracts the requested shape. - /// - /// The type of configuration to read. - /// The cluster ID to fetch for. - /// Function to fetch from central DB. - /// Function to extract the config from a snapshot. - /// Cancellation token. - /// The configuration of type T. - public async ValueTask ReadAsync( - string clusterId, - Func> centralFetch, - Func fromSnapshot, - CancellationToken cancellationToken) - { - ArgumentException.ThrowIfNullOrWhiteSpace(clusterId); - ArgumentNullException.ThrowIfNull(centralFetch); - ArgumentNullException.ThrowIfNull(fromSnapshot); - - try - { - var result = await _pipeline.ExecuteAsync(centralFetch, cancellationToken).ConfigureAwait(false); - _staleFlag.MarkFresh(); - return result; - } - // Catch all exceptions that are NOT genuine caller cancellations. A SQL command-level - // timeout surfaces as TaskCanceledException (derives from OperationCanceledException) - // but the caller's token is NOT cancelled — we must fall back to the sealed cache for - // that case, not propagate. Only rethrow if the caller actually requested cancellation. - catch (Exception ex) when (ex is not OperationCanceledException || !cancellationToken.IsCancellationRequested) - { - // Do NOT pass the raw exception object — it carries the stack - // and inner-exception chain, and SqlException/wrapping delegates can surface - // connection-string fragments (Password=…, User Id=…) embedded in messages. - // Log only the exception type and a scrubbed message so secrets stay out of logs. - _logger.LogWarning( - "Central-DB read failed after retries ({ExceptionType}: {SanitizedMessage}); falling back to sealed cache for cluster {ClusterId}", - ex.GetType().Name, - ScrubSecrets(ex.Message), - clusterId); - // GenerationCacheUnavailableException surfaces intentionally — fails the caller's - // operation. StaleConfigFlag stays unchanged; the flag only flips when we actually - // served a cache snapshot. - var snapshot = await _cache.ReadCurrentAsync(clusterId, cancellationToken).ConfigureAwait(false); - _staleFlag.MarkStale(); - return fromSnapshot(snapshot); - } - } -} diff --git a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/StaleConfigFlag.cs b/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/StaleConfigFlag.cs deleted file mode 100644 index 35e7e237..00000000 --- a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/LocalCache/StaleConfigFlag.cs +++ /dev/null @@ -1,20 +0,0 @@ -namespace ZB.MOM.WW.OtOpcUa.Configuration.LocalCache; - -/// -/// Thread-safe UsingStaleConfig signal per Phase 6.1 Stream D.3. Flips true whenever -/// a read falls back to a sealed cache snapshot; flips false on the next successful central-DB -/// round-trip. Surfaced on /healthz body and on the Admin /hosts page. -/// -public sealed class StaleConfigFlag -{ - private int _stale; - - /// True when the last config read was served from the sealed cache, not the central DB. - public bool IsStale => Volatile.Read(ref _stale) != 0; - - /// Mark the current config as stale (a read fell back to the cache). - public void MarkStale() => Volatile.Write(ref _stale, 1); - - /// Mark the current config as fresh (a central-DB read succeeded). - public void MarkFresh() => Volatile.Write(ref _stale, 0); -} diff --git a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/Services/ILdapGroupRoleMappingService.cs b/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/Services/ILdapGroupRoleMappingService.cs index 5f0193dd..1a21d2a5 100644 --- a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/Services/ILdapGroupRoleMappingService.cs +++ b/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/Services/ILdapGroupRoleMappingService.cs @@ -10,10 +10,18 @@ namespace ZB.MOM.WW.OtOpcUa.Configuration.Services; /// Phase 6.2 compliance check on control/data-plane separation). /// /// -/// Per Phase 6.2 Stream A.2 this service is expected to run behind the Phase 6.1 -/// ResilientConfigReader pipeline (timeout → retry → fallback-to-cache) so a -/// transient DB outage during sign-in falls back to the sealed snapshot rather than -/// denying every login. +/// +/// This service has no local-cache fallback: a DB outage during sign-in denies logins. +/// +/// +/// The Phase 6.1 ResilientConfigReader pipeline (timeout → retry → +/// fallback-to-sealed-snapshot) this was once expected to run behind was never wired to +/// anything and was deleted along with the rest of the dormant LiteDB local cache, which +/// ZB.MOM.WW.LocalDb supersedes. LocalDb caches the deployed-configuration artifact +/// for driver-role nodes; it does not currently cover admin-plane reads like this one. +/// Reviving the fallback means adding an admin-side cache on LocalDb, not restoring the +/// old pipeline. +/// /// public interface ILdapGroupRoleMappingService { diff --git a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/ZB.MOM.WW.OtOpcUa.Configuration.csproj b/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/ZB.MOM.WW.OtOpcUa.Configuration.csproj index 4c8b187f..ce45b41c 100644 --- a/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/ZB.MOM.WW.OtOpcUa.Configuration.csproj +++ b/src/Core/ZB.MOM.WW.OtOpcUa.Configuration/ZB.MOM.WW.OtOpcUa.Configuration.csproj @@ -26,7 +26,6 @@ - diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/KestrelHttpBinding.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/KestrelHttpBinding.cs new file mode 100644 index 00000000..dcf343d2 --- /dev/null +++ b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/KestrelHttpBinding.cs @@ -0,0 +1,102 @@ +using System.Net; +using Microsoft.AspNetCore.Server.Kestrel.Core; + +namespace ZB.MOM.WW.OtOpcUa.Host.Configuration; + +/// +/// One HTTP endpoint the host was asked to serve, parsed out of the configured URL list so it +/// can be re-bound explicitly. +/// +/// The host component as written — +, *, a hostname, or an IP. +/// The port. +/// True when the URL used the https scheme. +public sealed record KestrelHttpBinding(string Host, int Port, bool IsSecure) +{ + /// + /// Parses a semicolon-separated URL list (the ASPNETCORE_URLS / urls format) + /// into bindings. Unparseable entries are skipped rather than throwing — a malformed URL + /// must not take the process down at startup. + /// + /// The configured URL list, or null/empty when none is configured. + /// The parsed bindings, in the order given; empty when nothing is configured. + public static IReadOnlyList Parse(string? urls) + { + if (string.IsNullOrWhiteSpace(urls)) + return []; + + var result = new List(); + foreach (var raw in urls.Split(';', StringSplitOptions.RemoveEmptyEntries | StringSplitOptions.TrimEntries)) + { + // Uri cannot parse the "+"/"*" wildcard hosts Kestrel accepts, so substitute a + // placeholder host purely to get scheme/port out, then keep the original host token. + var isWildcard = raw.Contains("://+", StringComparison.Ordinal) + || raw.Contains("://*", StringComparison.Ordinal); + var probe = isWildcard + ? raw.Replace("://+", "://placeholder", StringComparison.Ordinal) + .Replace("://*", "://placeholder", StringComparison.Ordinal) + : raw; + + if (!Uri.TryCreate(probe, UriKind.Absolute, out var uri)) + continue; + + var host = isWildcard + ? raw.Contains("://+", StringComparison.Ordinal) ? "+" : "*" + : uri.Host; + + result.Add(new KestrelHttpBinding(host, uri.Port, uri.Scheme == Uri.UriSchemeHttps)); + } + + return result; + } + + /// + /// Parses the bare-port list format of ASPNETCORE_HTTP_PORTS / HTTP_PORTS (and + /// the HTTPS variants) into wildcard (+, all-interfaces) bindings — the shape + /// Kestrel itself gives those variables. Modern .NET base images (aspnet:8.0+) set + /// ASPNETCORE_HTTP_PORTS=8080 as the container default instead of + /// ASPNETCORE_URLS, so a node that never sets URLS explicitly still has a real + /// configured surface here that must be re-bound, not replaced with Kestrel's localhost:5000. + /// + /// A ;- or ,-separated list of bare ports, or null/empty. + /// True to mark the parsed bindings https (the HTTPS_PORTS vars). + /// One all-interfaces binding per parseable port; empty when nothing is configured. + public static IReadOnlyList ParsePorts(string? ports, bool isSecure) + { + if (string.IsNullOrWhiteSpace(ports)) + return []; + + var result = new List(); + foreach (var token in ports.Split([';', ','], StringSplitOptions.RemoveEmptyEntries | StringSplitOptions.TrimEntries)) + { + // Skip a malformed port rather than throwing — a bad env var must not down startup. + if (int.TryParse(token, out var port) && port is > 0 and <= 65535) + result.Add(new KestrelHttpBinding("+", port, isSecure)); + } + + return result; + } + + /// + /// Applies this binding to Kestrel, choosing the narrowest listen call the host component + /// allows. + /// + /// + /// A hostname that is neither a literal IP nor localhost cannot be reliably mapped to + /// a local interface, so it falls back to — + /// the safe superset. Narrowing it and guessing wrong would silently stop serving. + /// + /// The Kestrel options being configured. + public void Apply(KestrelServerOptions options) + { + ArgumentNullException.ThrowIfNull(options); + + if (Host is "+" or "*") + options.ListenAnyIP(Port); + else if (IPAddress.TryParse(Host, out var address)) + options.Listen(address, Port); + else if (string.Equals(Host, "localhost", StringComparison.OrdinalIgnoreCase)) + options.ListenLocalhost(Port); + else + options.ListenAnyIP(Port); + } +} diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbRegistration.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbRegistration.cs new file mode 100644 index 00000000..c23fa6bc --- /dev/null +++ b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbRegistration.cs @@ -0,0 +1,101 @@ +using Microsoft.Extensions.Configuration; +using Microsoft.Extensions.DependencyInjection; +using ZB.MOM.WW.LocalDb; +using ZB.MOM.WW.LocalDb.Replication; +using ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache; + +namespace ZB.MOM.WW.OtOpcUa.Host.Configuration; + +/// +/// Single registration point for the node-local LocalDb subsystem: the embedded SQLite store +/// that caches the deployed-configuration artifact, plus the optional gRPC replication that +/// mirrors it to the node's redundant pair peer. +/// +/// +/// +/// Exists as a named extension rather than inline AddZbLocalDb calls in +/// Program.cs so the wiring is covered by a DI resolution test — Program.cs is +/// top-level statements and cannot be exercised directly, which is exactly how a +/// "registered but never resolvable" defect ships unnoticed. This family has shipped that +/// defect three times (Secrets 0.2.0, Secrets 0.2.2, ScadaBridge #22). +/// +/// +/// Driver-role nodes only. Admin-only nodes have no deployed configuration to cache, +/// and registering here would impose the LocalDb:Path-required-or-no-boot constraint +/// on them for nothing. +/// +/// +/// Storage is unconditional; replication is default-OFF. A node with no +/// PeerAddress and no SyncListenPort is simply a fast local SQLite file — +/// AddZbLocalDbReplication registers the engine but it stays idle. Replication is +/// opt-in per pair because there is no production distribution story for the sync +/// ApiKey yet (the same open question the Secrets KEK has). +/// +/// +public static class LocalDbRegistration +{ + /// Configuration section holding LocalDbOptions. + public const string LocalDbSectionPath = "LocalDb"; + + /// Configuration section holding ReplicationOptions. + public const string ReplicationSectionPath = "LocalDb:Replication"; + + /// + /// Configuration key carrying the dedicated h2c sync listener's port. Zero or absent means + /// no listener is bound and the passive sync endpoint is not mapped. + /// + public const string SyncListenPortKey = "LocalDb:SyncListenPort"; + + /// + /// The port the dedicated cleartext-HTTP/2 sync listener should bind, or 0 when the + /// node should not listen at all. Defaults to 0 — absent configuration must mean + /// "off". + /// + /// The application configuration. + /// The configured sync port, or 0. + public static int SyncListenPort(IConfiguration configuration) + { + ArgumentNullException.ThrowIfNull(configuration); + return configuration.GetValue(SyncListenPortKey, defaultValue: 0); + } + + /// + /// Registers the local database (with the deployment-cache schema applied and registered for + /// replication) and the replication engine. + /// + /// + /// + /// runs inside AddZbLocalDb's singleton factory, + /// before the first caller receives the ILocalDb — so every consumer is guaranteed + /// to see the tables created and their capture triggers installed. + /// + /// + /// AddZbLocalDbReplication is called unconditionally rather than behind an + /// Enabled flag because the library already models "off" as "no + /// PeerAddress": the initiator background service idles and the passive endpoint + /// is only reachable if Program.cs maps it, which it does only when + /// is set. Adding a second gate on top would give two + /// ways to express the same thing and a way for them to disagree. + /// + /// + /// The service collection to add to. + /// The application configuration. + /// The same instance, for chaining. + public static IServiceCollection AddOtOpcUaLocalDb( + this IServiceCollection services, + IConfiguration configuration) + { + ArgumentNullException.ThrowIfNull(services); + ArgumentNullException.ThrowIfNull(configuration); + + services.AddZbLocalDb(configuration, LocalDbSetup.OnReady); + services.AddZbLocalDbReplication(configuration); + + // The consumer-facing seam. WithOtOpcUaRuntimeActors resolves this optionally and threads it + // into DriverHostActor; registering the store without this leaves the cache tables present, + // replicating, and never written to. + services.AddSingleton(); + + return services; + } +} diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbSetup.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbSetup.cs new file mode 100644 index 00000000..0ce2d3be --- /dev/null +++ b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbSetup.cs @@ -0,0 +1,48 @@ +using ZB.MOM.WW.LocalDb; +using ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache; + +namespace ZB.MOM.WW.OtOpcUa.Host.Configuration; + +/// +/// The onReady callback handed to AddZbLocalDb: creates the deployment-cache +/// tables and opts them into replication. +/// +/// +/// +/// Public rather than internal only so LocalDbSetupTests can drive the production +/// callback directly. Initialising a test database from a hand-written copy of this schema +/// would prove only that the test agrees with itself — the whole value of those tests is +/// that they exercise this method. +/// +/// +public static class LocalDbSetup +{ + /// + /// Initialises the local database: DDL first, then replication registration. + /// + /// + /// + /// THE ORDER IS LOAD-BEARING: DDL → RegisterReplicated → writes. + /// RegisterReplicated is what installs the three AFTER triggers that capture + /// changes into the oplog. Any row written before that call is never captured, so it + /// never reaches the peer — silently, and permanently, because nothing ever revisits + /// history. Phase 1 writes nothing here; when Phase 2 adds its store-and-forward + /// migrator, the migrator must run after both registrations for the same reason. + /// + /// + /// The freshly constructed local database. + public static void OnReady(ILocalDb db) + { + ArgumentNullException.ThrowIfNull(db); + + // CreateConnection() hands back an already-open, pragma-configured connection carrying the + // zb_hlc_next() UDF the capture triggers need. Calling Open() on it would throw. + using (var connection = db.CreateConnection()) + { + DeploymentCacheSchema.Apply(connection); + } + + db.RegisterReplicated(DeploymentCacheSchema.ArtifactsTable); + db.RegisterReplicated(DeploymentCacheSchema.PointerTable); + } +} diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbSyncAuthInterceptor.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbSyncAuthInterceptor.cs new file mode 100644 index 00000000..a16acd61 --- /dev/null +++ b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Configuration/LocalDbSyncAuthInterceptor.cs @@ -0,0 +1,162 @@ +using System.Security.Cryptography; +using System.Text; +using Grpc.Core; +using Grpc.Core.Interceptors; +using Microsoft.Extensions.Logging; +using Microsoft.Extensions.Options; +using ZB.MOM.WW.LocalDb.Replication; + +namespace ZB.MOM.WW.OtOpcUa.Host.Configuration; + +/// +/// Gates the LocalDb passive sync endpoint. The replication library deliberately leaves inbound +/// authentication to the host — its LocalDbSyncService verifies nothing — so without this +/// interceptor anything that can reach a driver node's sync port could stream arbitrary rows +/// into that node's deployment-artifact cache. +/// +/// +/// +/// Why that matters here specifically. The cached artifact is what a driver node +/// boots from when central SQL is unreachable. An attacker able to write it could choose +/// the configuration a node comes up on during exactly the outage nobody is watching. +/// +/// +/// Scoped by method path. Only calls under /localdb_sync.v1.LocalDbSync/ are +/// gated; every other method on the shared gRPC pipeline passes through untouched. +/// +/// +/// Fail-closed. With no LocalDb:Replication:ApiKey configured, NO sync stream +/// is accepted, authenticated or not. That is the deliberate choice: the alternative — +/// treating "no key" as "no auth required" — would silently expose the endpoint on exactly +/// the default configuration every node ships with. An operator enabling replication must +/// set the same key on both nodes of the pair, which is already required for the initiator +/// to dial out (SyncBackgroundService sends Authorization: Bearer <key>). +/// A key typo therefore stops convergence outright rather than degrading to unauthenticated. +/// +/// +/// Comparison is over UTF-8 bytes, so +/// a wrong key cannot be recovered byte-by-byte from response timing. Length differences are +/// unavoidably observable and are not sensitive. +/// +/// +public sealed class LocalDbSyncAuthInterceptor : Interceptor +{ + private const string ServicePrefix = "/localdb_sync.v1.LocalDbSync/"; + private const string AuthorizationHeader = "authorization"; + private const string BearerPrefix = "Bearer "; + + private readonly IOptions _options; + private readonly ILogger _logger; + + /// Creates the interceptor. + /// Replication options; ApiKey is the expected bearer token. + /// Logger for denial diagnostics. + public LocalDbSyncAuthInterceptor( + IOptions options, + ILogger logger) + { + ArgumentNullException.ThrowIfNull(options); + ArgumentNullException.ThrowIfNull(logger); + + _options = options; + _logger = logger; + } + + /// + public override Task UnaryServerHandler( + TRequest request, + ServerCallContext context, + UnaryServerMethod continuation) + { + Authorize(context); + return continuation(request, context); + } + + /// + public override Task DuplexStreamingServerHandler( + IAsyncStreamReader requestStream, + IServerStreamWriter responseStream, + ServerCallContext context, + DuplexStreamingServerMethod continuation) + { + Authorize(context); + return continuation(requestStream, responseStream, context); + } + + /// + public override Task ClientStreamingServerHandler( + IAsyncStreamReader requestStream, + ServerCallContext context, + ClientStreamingServerMethod continuation) + { + Authorize(context); + return continuation(requestStream, context); + } + + /// + public override Task ServerStreamingServerHandler( + TRequest request, + IServerStreamWriter responseStream, + ServerCallContext context, + ServerStreamingServerMethod continuation) + { + Authorize(context); + return continuation(request, responseStream, context); + } + + /// + /// Throws with if this + /// is a sync call that does not carry the configured bearer token. Non-sync calls return + /// immediately. + /// + private void Authorize(ServerCallContext context) + { + if (!context.Method.StartsWith(ServicePrefix, StringComparison.Ordinal)) + return; + + var expected = _options.Value.ApiKey; + if (string.IsNullOrEmpty(expected)) + { + _logger.LogWarning( + "Rejected a LocalDb sync call to {Method}: no LocalDb:Replication:ApiKey is configured, " + + "so the passive sync endpoint is closed. Configure the same key on both nodes of the pair.", + context.Method); + throw new RpcException(new Status( + StatusCode.PermissionDenied, + "LocalDb sync is not accepting connections: no API key is configured on this node.")); + } + + var presented = ExtractBearerToken(context.RequestHeaders); + if (presented is null || !FixedTimeEquals(presented, expected)) + { + _logger.LogWarning( + "Rejected a LocalDb sync call to {Method}: {Reason}.", + context.Method, + presented is null ? "no bearer token presented" : "bearer token did not match"); + throw new RpcException(new Status( + StatusCode.PermissionDenied, + "LocalDb sync authentication failed.")); + } + } + + private static string? ExtractBearerToken(Metadata headers) + { + // gRPC lowercases header keys on the wire; compare case-insensitively anyway so a + // hand-built Metadata in a test behaves the same as a real request. + foreach (var entry in headers) + { + if (!string.Equals(entry.Key, AuthorizationHeader, StringComparison.OrdinalIgnoreCase)) + continue; + + var value = entry.Value; + if (value is not null && value.StartsWith(BearerPrefix, StringComparison.OrdinalIgnoreCase)) + return value[BearerPrefix.Length..]; + } + + return null; + } + + private static bool FixedTimeEquals(string presented, string expected) + => CryptographicOperations.FixedTimeEquals( + Encoding.UTF8.GetBytes(presented), Encoding.UTF8.GetBytes(expected)); +} diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Host/Health/HealthEndpoints.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Health/HealthEndpoints.cs index a2abe173..4ad7385e 100644 --- a/src/Server/ZB.MOM.WW.OtOpcUa.Host/Health/HealthEndpoints.cs +++ b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Health/HealthEndpoints.cs @@ -1,10 +1,15 @@ using Microsoft.AspNetCore.Routing; using Microsoft.EntityFrameworkCore; +using Microsoft.Extensions.Configuration; using Microsoft.Extensions.DependencyInjection; +using Microsoft.Extensions.Diagnostics.HealthChecks; +using Microsoft.Extensions.Options; using ZB.MOM.WW.Health; using ZB.MOM.WW.Health.Akka; using ZB.MOM.WW.Health.EntityFrameworkCore; +using ZB.MOM.WW.LocalDb.Replication; using ZB.MOM.WW.OtOpcUa.Configuration; +using ZB.MOM.WW.OtOpcUa.Host.Configuration; namespace ZB.MOM.WW.OtOpcUa.Host.Health; @@ -36,7 +41,20 @@ public static class HealthEndpoints "admin-leader", failureStatus: null, tags: new[] { ZbHealthTags.Active }, - args: "admin"); + args: "admin") + // Registered unconditionally, not driver-gated. AddOtOpcUaHealth takes no role argument + // and runs on every node; the check itself resolves ISyncStatus optionally and reports + // Healthy when LocalDb is absent (admin-only graphs) or replication is default-OFF, so a + // plain node is never degraded by it. A factory registration keeps this no-arg signature + // while still reading ISyncStatus + options + the sync port from the container. + .Add(new HealthCheckRegistration( + "localdb-replication", + sp => new LocalDbReplicationHealthCheck( + sp.GetService(), + sp.GetRequiredService>(), + LocalDbRegistration.SyncListenPort(sp.GetRequiredService())), + failureStatus: null, + tags: new[] { ZbHealthTags.Active })); return services; } diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Host/Health/LocalDbReplicationHealthCheck.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Health/LocalDbReplicationHealthCheck.cs new file mode 100644 index 00000000..8919d179 --- /dev/null +++ b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Health/LocalDbReplicationHealthCheck.cs @@ -0,0 +1,126 @@ +using Microsoft.Extensions.Diagnostics.HealthChecks; +using Microsoft.Extensions.Options; +using ZB.MOM.WW.LocalDb.Replication; + +namespace ZB.MOM.WW.OtOpcUa.Host.Health; + +/// +/// Pure decision function for the LocalDb replication probe, factored out of +/// so the whole matrix is table-testable without +/// standing up a replication engine. +/// +public static class LocalDbReplicationDecision +{ + /// + /// Maps the resolved replication facts to a health status. + /// + /// + /// True when this node participates in replication at all — it has a peer address to dial, or + /// a sync listener bound. False means default-OFF, and default-OFF must never degrade a node. + /// + /// Whether a sync session is currently running. + /// + /// The unacked oplog backlog, or when it could not be read. Null is + /// deliberately not treated as zero: a failed poll is "unknown", not "caught up". + /// + /// + /// Backlog at or above which a connected pair is judged to be falling behind. + /// + /// The status plus a human-readable reason. + public static (HealthStatus Status, string Description) Evaluate( + bool peerConfigured, bool connected, long? oplogBacklog, long degradedThreshold) + { + if (!peerConfigured) + return (HealthStatus.Healthy, "LocalDb replication is not configured on this node (default-OFF)."); + + if (!connected) + return (HealthStatus.Degraded, "LocalDb replication is configured but no sync session is connected."); + + if (oplogBacklog is null) + return (HealthStatus.Degraded, "LocalDb replication is connected but its oplog backlog is unknown (the poll failed)."); + + if (oplogBacklog.Value >= degradedThreshold) + return (HealthStatus.Degraded, + $"LocalDb replication is connected but its oplog backlog ({oplogBacklog.Value}) is at or above the degraded threshold ({degradedThreshold})."); + + return (HealthStatus.Healthy, $"LocalDb replication is connected; oplog backlog {oplogBacklog.Value}."); + } +} + +/// +/// Reports the health of this node's LocalDb replication link. Registered unconditionally with +/// the shared health pipeline; when LocalDb is not present at all (admin-only graphs) it reports +/// rather than degrading — matching the +/// ActiveNodeHealthCheck precedent of "not applicable ⇒ Healthy". +/// +/// +/// A node running LocalDb with replication switched off (no peer, no listener) is also Healthy: +/// that is the default posture for most of the fleet, and it must not look degraded. Only a node +/// that is meant to be replicating and is not — or is connected but cannot confirm it is +/// draining — is surfaced as Degraded. It never reports Unhealthy: a replication problem does not +/// stop the node serving its address space, so it must not fail a readiness gate. +/// +public sealed class LocalDbReplicationHealthCheck : IHealthCheck +{ + /// + /// Backlog at or above which a connected pair is reported Degraded. Well below the library's + /// MaxOplogRows default (1,000,000, where a snapshot resync is forced), so the probe + /// flags a pair that is falling behind long before the engine itself intervenes. + /// + public const long DefaultBacklogDegradedThreshold = 100_000; + + private readonly ISyncStatus? _syncStatus; + private readonly ReplicationOptions _options; + private readonly int _syncListenPort; + private readonly long _degradedThreshold; + + /// Creates the health check. + /// + /// The replication status singleton, or when the replication engine is + /// not registered (admin-only nodes) — in which case the check is a Healthy no-op. + /// + /// The bound replication options (peer address, etc.). + /// + /// The configured sync listener port; a value greater than zero counts as "replication + /// configured" even when this node is the passive side with no peer address. + /// + /// Backlog degraded threshold; defaults to . + public LocalDbReplicationHealthCheck( + ISyncStatus? syncStatus, + IOptions options, + int syncListenPort, + long degradedThreshold = DefaultBacklogDegradedThreshold) + { + ArgumentNullException.ThrowIfNull(options); + + _syncStatus = syncStatus; + _options = options.Value; + _syncListenPort = syncListenPort; + _degradedThreshold = degradedThreshold; + } + + /// + public Task CheckHealthAsync( + HealthCheckContext context, CancellationToken cancellationToken = default) + { + // No replication engine registered at all → not applicable → Healthy. Admin-only nodes and + // any graph that did not call AddOtOpcUaLocalDb land here. + if (_syncStatus is null) + return Task.FromResult(HealthCheckResult.Healthy( + "LocalDb replication is not present on this node.")); + + var peerConfigured = + !string.IsNullOrWhiteSpace(_options.PeerAddress) || _syncListenPort > 0; + + var (status, description) = LocalDbReplicationDecision.Evaluate( + peerConfigured, _syncStatus.Connected, _syncStatus.OplogBacklog, _degradedThreshold); + + var result = status switch + { + HealthStatus.Healthy => HealthCheckResult.Healthy(description), + _ => HealthCheckResult.Degraded(description), + }; + + return Task.FromResult(result); + } +} diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Host/Observability/ObservabilityExtensions.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Observability/ObservabilityExtensions.cs index 2c89fd2f..171d9583 100644 --- a/src/Server/ZB.MOM.WW.OtOpcUa.Host/Observability/ObservabilityExtensions.cs +++ b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Observability/ObservabilityExtensions.cs @@ -1,4 +1,5 @@ using Microsoft.Extensions.Configuration; +using ZB.MOM.WW.LocalDb.Replication; using ZB.MOM.WW.OtOpcUa.Commons.Observability; using ZB.MOM.WW.Telemetry; @@ -27,7 +28,11 @@ public static class ObservabilityExtensions return services.AddZbTelemetry(o => { o.ServiceName = "otopcua"; - o.Meters = [OtOpcUaTelemetry.MeterName]; + // o.Meters is a STRICT allowlist — ZbTelemetry adds only the named meters, with no + // wildcard. The LocalDb replication engine publishes under its own meter name, so + // without this entry its localdb.sync.* / localdb.oplog.depth series are silently absent + // from /metrics on driver nodes (the exact omission a ScadaBridge live gate caught). + o.Meters = [OtOpcUaTelemetry.MeterName, LocalDbMetrics.MeterName]; o.ActivitySources = [OtOpcUaTelemetry.ActivitySourceName]; if (Enum.TryParse(configuration["OtOpcUa:Telemetry:Exporter"], ignoreCase: true, out var exporter)) o.Exporter = exporter; diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Host/Program.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Program.cs index c332916e..b9de794b 100644 --- a/src/Server/ZB.MOM.WW.OtOpcUa.Host/Program.cs +++ b/src/Server/ZB.MOM.WW.OtOpcUa.Host/Program.cs @@ -1,7 +1,9 @@ using Akka.Actor; using Akka.Hosting; +using Microsoft.AspNetCore.Server.Kestrel.Core; using Microsoft.Extensions.DependencyInjection.Extensions; using Serilog; +using ZB.MOM.WW.LocalDb.Replication; using Serilog.Events; using ZB.MOM.WW.OtOpcUa.AdminUI; using ZB.MOM.WW.OtOpcUa.AdminUI.Clients; @@ -174,6 +176,20 @@ if (hasDriver) builder.Configuration, (_, sp) => GatewayHistorian.CreateAlarmWriter(serverHistorianOptions, sp)); + // Node-local LocalDb: caches the deployed-configuration artifact so this node can boot from + // its last-known-good config when central SQL is unreachable, and optionally replicates that + // cache to its redundant pair peer. Driver-role only — admin-only nodes have nothing to cache, + // and registering here would impose LocalDb:Path-required-or-no-boot on them for nothing. + // Storage ships unconditionally; replication stays inert until LocalDb:Replication:PeerAddress + // (initiator) / LocalDb:SyncListenPort (listener) are set. See LocalDbRegistration. + builder.Services.AddOtOpcUaLocalDb(builder.Configuration); + + // gRPC server plumbing for the passive sync endpoint mapped below. Registered only under + // hasDriver so admin-only nodes expose no sync surface at all. The interceptor is the ONLY + // inbound auth on that endpoint — the replication library's LocalDbSyncService verifies + // nothing — and it fail-closes when no ApiKey is configured. + builder.Services.AddGrpc(o => o.Interceptors.Add()); + // Config-gated server-side HistoryRead backend. When the ServerHistorian section is enabled this // overrides the NullHistorianDataSource default from AddOtOpcUaRuntime (last registration wins) with // a read-only HistorianGateway-backed data source the node manager's HistoryRead overrides @@ -365,6 +381,92 @@ builder.Services.AddOtOpcUaSecrets(builder.Configuration); builder.Services.AddOtOpcUaHealth(); builder.Services.AddOtOpcUaObservability(builder.Configuration); +// --------------------------------------------------------------------------------------------- +// LocalDb sync listener (default-OFF). +// +// DANGER: any explicit Kestrel Listen* call makes Kestrel IGNORE ASPNETCORE_URLS/urls ENTIRELY +// (it logs "Overriding address(es)"). The host has no ConfigureKestrel today and binds solely via +// that configuration, so adding a listener naively would silently unbind the AdminUI + deploy API +// behind Traefik. Everything the host was already asked to serve is therefore re-bound explicitly +// in the same block. +// +// The listener is HTTP/2-ONLY on purpose: the sync client speaks prior-knowledge h2c, which a +// cleartext Http1AndHttp2 endpoint cannot negotiate (there is no ALPN without TLS). Hence a +// dedicated port rather than multiplexing onto the main one. +// +// When LocalDb:SyncListenPort is 0 (the default) none of this runs and URL binding is untouched. +// --------------------------------------------------------------------------------------------- +var syncListenPort = LocalDbRegistration.SyncListenPort(builder.Configuration); +if (hasDriver && syncListenPort > 0) +{ + // Mirror Kestrel's own source precedence: URLS wins; else the HTTP_PORTS/HTTPS_PORTS bare-port + // vars; else Kestrel's localhost:5000 default. Missing the HTTP_PORTS leg is not academic — the + // aspnet:8.0+ base images set ASPNETCORE_HTTP_PORTS=8080 as the CONTAINER default in place of + // ASPNETCORE_URLS, so a driver node that never sets URLS is really serving :8080 on all + // interfaces. Falling straight through to localhost:5000 would silently move its health/metrics + // surface to loopback (found on the docker-dev live gate). + var configuredUrls = builder.Configuration["urls"] + ?? builder.Configuration["ASPNETCORE_URLS"]; + var existingBindings = KestrelHttpBinding.Parse(configuredUrls); + + if (existingBindings.Count == 0) + { + var httpPorts = builder.Configuration["ASPNETCORE_HTTP_PORTS"] ?? builder.Configuration["HTTP_PORTS"]; + var httpsPorts = builder.Configuration["ASPNETCORE_HTTPS_PORTS"] ?? builder.Configuration["HTTPS_PORTS"]; + existingBindings = + [ + .. KestrelHttpBinding.ParsePorts(httpPorts, isSecure: false), + .. KestrelHttpBinding.ParsePorts(httpsPorts, isSecure: true), + ]; + if (existingBindings.Count > 0) + Log.Information( + "LocalDb sync listener enabled; re-binding the HTTP_PORTS surface ({HttpPorts}/{HttpsPorts}) " + + "alongside the sync port.", httpPorts, httpsPorts); + } + + // A driver-only Windows-service node gets NO URLS and NO *_PORTS (Install-Services.ps1 sets URLS + // only under $hasAdmin), so "nothing configured" is a real, supported state — not an error. + // Kestrel's own default is localhost:5000; re-state it explicitly, because taking over + // configuration means taking over the default too. Health probes hit this port. + if (existingBindings.Count == 0) + { + existingBindings = [new KestrelHttpBinding("localhost", 5000, IsSecure: false)]; + Log.Information( + "LocalDb sync listener enabled with no URLs configured; re-binding Kestrel's default " + + "http://localhost:5000 alongside the sync port."); + } + + // Re-binding an https endpoint would need its certificate configuration replayed too, which + // this host does not model (TLS is terminated at Traefik). Rather than silently serve it + // without TLS — or drop it — refuse to take over Kestrel at all and leave the existing + // surface exactly as it was. The operator gets a loud, actionable error instead of an + // AdminUI that stopped answering. + if (existingBindings.Any(b => b.IsSecure)) + { + Log.Error( + "LocalDb:SyncListenPort is set to {Port} but this host serves HTTPS endpoint(s) ({Urls}). " + + "Binding the sync listener requires re-binding every existing endpoint explicitly, and the " + + "HTTPS certificate configuration cannot be replayed safely. The sync listener is DISABLED; " + + "terminate TLS upstream (as the docker-dev rig does) or leave replication off on this node.", + syncListenPort, configuredUrls); + syncListenPort = 0; + } + else + { + builder.WebHost.ConfigureKestrel(kestrel => + { + foreach (var binding in existingBindings) + binding.Apply(kestrel); + + kestrel.ListenAnyIP(syncListenPort, o => o.Protocols = HttpProtocols.Http2); + }); + + Log.Information( + "LocalDb sync listener bound on :{SyncPort} (h2c); re-bound existing endpoint(s) {Urls}.", + syncListenPort, configuredUrls ?? "http://localhost:5000 (Kestrel default)"); + } +} + var app = builder.Build(); // AddZbSerilog registers Serilog as the MEL logging provider but does NOT assign the static @@ -395,6 +497,15 @@ if (hasAdmin) app.MapOtOpcUaDeployApi(app.Configuration); } +// Passive LocalDb sync endpoint. Gated on the same port check that bound the listener, so a +// default-OFF or admin-only graph carries no sync surface whatsoever. Mapping it would be safe +// even unauthenticated — LocalDbSyncAuthInterceptor fail-closes with no ApiKey configured — but +// not mapping it at all is the stronger default. +if (hasDriver && syncListenPort > 0) +{ + app.MapZbLocalDbSync(); +} + app.MapOtOpcUaHealth(); app.MapOtOpcUaMetrics(); diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Host/ZB.MOM.WW.OtOpcUa.Host.csproj b/src/Server/ZB.MOM.WW.OtOpcUa.Host/ZB.MOM.WW.OtOpcUa.Host.csproj index 636c7953..dd7fbd9a 100644 --- a/src/Server/ZB.MOM.WW.OtOpcUa.Host/ZB.MOM.WW.OtOpcUa.Host.csproj +++ b/src/Server/ZB.MOM.WW.OtOpcUa.Host/ZB.MOM.WW.OtOpcUa.Host.csproj @@ -33,6 +33,13 @@ + + + + diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Host/appsettings.json b/src/Server/ZB.MOM.WW.OtOpcUa.Host/appsettings.json index a955665b..e7ca5cd6 100644 --- a/src/Server/ZB.MOM.WW.OtOpcUa.Host/appsettings.json +++ b/src/Server/ZB.MOM.WW.OtOpcUa.Host/appsettings.json @@ -59,6 +59,16 @@ "Capacity": 1000000, "DeadLetterRetentionDays": 30 }, + "LocalDb": { + "_comment": "Node-local SQLite store caching the deployed-configuration artifact, so a driver node can boot from its last-known-good config when central SQL is unreachable. Registered on driver-role nodes only (see LocalDbRegistration). Storage is unconditional; replication to the redundant pair peer is default-OFF.", + "Path": "./data/otopcua-localdb.db", + "_SyncListenPortComment": "Port for the dedicated cleartext-HTTP/2 (h2c) listener serving the passive sync endpoint. 0 = no listener bound and the endpoint is not mapped. Must be a port distinct from the main HTTP listener: prior-knowledge h2c cannot share a cleartext Http1AndHttp2 port.", + "SyncListenPort": 0, + "Replication": { + "_comment": "Empty = passive/off. Set PeerAddress on exactly ONE node of the pair (the stream is bidirectional, so the other side still pushes its own deltas back). ApiKey must be BYTE-IDENTICAL on both nodes — the host interceptor fail-closes, so a typo silently stops convergence rather than degrading to unauthenticated. Supply it via ${secret:...} in production, never committed.", + "_MaxBatchSizeComment": "Row count, NOT bytes, against gRPC's 4 MB default message cap. Artifact chunk rows are ~171 KB base64, so keep this at 16 on the pair (16 x 171 KB ~= 2.7 MB worst case)." + } + }, "Deployment": { "_comment": "R2-11 (05/CONV-2): deploy-gate TagConfig strictness. Warn (default) = non-blocking warnings logged + appended to the deployment result message; Error = a config with a typo'd enum or unparseable TagConfig is rejected at the draft gate. Running servers are untouched; the gate only sees re-deploys.", "TagConfigValidationMode": "Warn" diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/DeploymentCacheSchema.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/DeploymentCacheSchema.cs new file mode 100644 index 00000000..14d29b9a --- /dev/null +++ b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/DeploymentCacheSchema.cs @@ -0,0 +1,70 @@ +using Microsoft.Data.Sqlite; + +namespace ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache; + +/// +/// DDL for the node-local deployment-artifact cache: the chunked artifact table plus the +/// per-cluster current-deployment pointer. +/// +/// +/// +/// Deliberately depends on nothing but so it can be applied +/// to any connection — the host's LocalDbSetup.OnReady in production, and a bare +/// connection in a test — without dragging the DI graph along. +/// +/// +/// Why the artifact is chunked base64 TEXT and not one BLOB row. Two independent +/// constraints force it. RegisterReplicated rejects BLOB columns outright; and the +/// replication engine batches by row count against gRPC's 4 MB default message cap, +/// so a single multi-megabyte artifact row would simply never be deliverable — at any +/// batch size. A 128 KiB raw chunk is ≈ 171 KB once base64-encoded, which keeps a +/// worst-case batch comfortably inside the cap. +/// +/// +/// Why no autoincrement PKs. Convergence is last-writer-wins keyed on the primary +/// key. Two nodes independently allocating rowid 7 for different rows would silently +/// overwrite one another. Every key here is TEXT or a composite of caller-supplied values. +/// +/// +public static class DeploymentCacheSchema +{ + /// Table holding the artifact bytes, split into base64 chunks. + public const string ArtifactsTable = "deployment_artifacts"; + + /// Table holding one current-deployment pointer per cluster. + public const string PointerTable = "deployment_pointer"; + + /// + /// Creates both cache tables if they do not already exist. Idempotent. + /// + /// + /// An already-open connection. ILocalDb.CreateConnection() hands out open, + /// pragma-configured connections — do not call Open() on one. + /// + public static void Apply(SqliteConnection connection) + { + ArgumentNullException.ThrowIfNull(connection); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = """ + CREATE TABLE IF NOT EXISTS deployment_artifacts ( + deployment_id TEXT NOT NULL, + chunk_index INTEGER NOT NULL, + cluster_id TEXT NOT NULL, + revision_hash TEXT NOT NULL, + chunk_count INTEGER NOT NULL, + chunk_base64 TEXT NOT NULL, + cached_at_utc TEXT NOT NULL, + PRIMARY KEY (deployment_id, chunk_index) + ); + CREATE TABLE IF NOT EXISTS deployment_pointer ( + cluster_id TEXT NOT NULL PRIMARY KEY, + deployment_id TEXT NOT NULL, + revision_hash TEXT NOT NULL, + artifact_sha256 TEXT NOT NULL, + applied_at_utc TEXT NOT NULL + ); + """; + cmd.ExecuteNonQuery(); + } +} diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/IDeploymentArtifactCache.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/IDeploymentArtifactCache.cs new file mode 100644 index 00000000..7d4a1c74 --- /dev/null +++ b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/IDeploymentArtifactCache.cs @@ -0,0 +1,68 @@ +namespace ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache; + +/// +/// Node-local durable cache of the deployment artifact this node last applied. +/// +/// +/// +/// Exists so a driver node can boot into its last-known-good address space when the central +/// configuration database is unreachable. Without it, a control-plane outage that outlives a +/// node restart leaves that node with no address space at all — the plant loses visibility +/// for a reason that has nothing to do with the plant. +/// +/// +public interface IDeploymentArtifactCache +{ + /// + /// Persists as the cluster's current deployment. + /// + /// + /// + /// Idempotent per : re-storing the same deployment + /// replaces its chunks rather than accumulating orphans. Retention is bounded to the two + /// newest deployments per cluster, so a long-lived node cannot grow its local database + /// without limit. + /// + /// + Task StoreAsync(string clusterId, string deploymentId, string revisionHash, + byte[] artifact, CancellationToken ct = default); + + /// + /// Reads the cached artifact for , or + /// when nothing is cached or the cached bytes fail their integrity check. + /// + /// + /// + /// A failed integrity check is deliberately indistinguishable from a miss. Booting a + /// plant from a truncated address space is strictly worse than booting from none: the + /// missing half looks like a deliberate configuration rather than an error. + /// + /// + Task GetCurrentAsync(string clusterId, CancellationToken ct = default); + + /// + /// Reads the single cached pointer without knowing the cluster id. + /// + /// + /// + /// The boot seam needs this. A node's cluster id is only derivable from an artifact it + /// has already loaded, or from the central database — which is precisely what is + /// unreachable in the scenario this cache exists to survive. So the cold path reads the + /// newest pointer unkeyed. + /// + /// + /// More than one pointer row means the node was re-homed between clusters. The newest + /// wins, but the choice is logged as a warning naming both clusters — a node silently + /// booting a neighbouring cluster's configuration is exactly the failure that must never + /// be quiet. + /// + /// + Task GetCurrentUnkeyedAsync(CancellationToken ct = default); +} + +/// +/// A deployment artifact recovered from the node-local cache, with the metadata needed to decide +/// whether it is still current once the control plane is reachable again. +/// +public sealed record CachedDeploymentArtifact( + string DeploymentId, string RevisionHash, byte[] Artifact, DateTimeOffset AppliedAtUtc); diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/LocalDbDeploymentArtifactCache.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/LocalDbDeploymentArtifactCache.cs new file mode 100644 index 00000000..a29c08e3 --- /dev/null +++ b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Deployment/LocalDbDeploymentArtifactCache.cs @@ -0,0 +1,324 @@ +using System.Globalization; +using System.Security.Cryptography; +using Microsoft.Extensions.Logging; +using ZB.MOM.WW.LocalDb; + +namespace ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache; + +/// +/// backed by the replicated LocalDb deployment-cache +/// tables. +/// +/// +/// +/// Why the artifact is stored as base64 chunks. See +/// — replication rejects BLOB columns and batches by row +/// count against gRPC's message cap, so a whole-artifact row could never be delivered. +/// Everything in this class that looks like ceremony (chunk indices, a chunk count, a +/// separate SHA over the raw bytes) exists to make that split safely reversible. +/// +/// +public sealed class LocalDbDeploymentArtifactCache : IDeploymentArtifactCache +{ + /// + /// Raw bytes per chunk, before base64 expansion. + /// + /// + /// 128 KiB raw is ≈ 171 KB encoded, which keeps a worst-case replication batch inside gRPC's + /// 4 MB default cap with room to spare. + /// + private const int ChunkSize = 128 * 1024; + + /// + /// How many deployments per cluster survive a store. + /// + /// + /// Two, not one: the previous artifact is what a rollback would need, and keeping it costs + /// one artifact's worth of disk. Three would only add a generation nobody rolls back to. + /// + private const int RetainedDeployments = 2; + + private readonly ILocalDb _db; + private readonly ILogger _logger; + + public LocalDbDeploymentArtifactCache(ILocalDb db, ILogger logger) + { + ArgumentNullException.ThrowIfNull(db); + ArgumentNullException.ThrowIfNull(logger); + + _db = db; + _logger = logger; + } + + /// + public async Task StoreAsync(string clusterId, string deploymentId, string revisionHash, + byte[] artifact, CancellationToken ct = default) + { + ArgumentException.ThrowIfNullOrWhiteSpace(clusterId); + ArgumentException.ThrowIfNullOrWhiteSpace(deploymentId); + ArgumentException.ThrowIfNullOrWhiteSpace(revisionHash); + ArgumentNullException.ThrowIfNull(artifact); + + var sha = Convert.ToHexString(SHA256.HashData(artifact)); + var cachedAtUtc = FormatTimestamp(DateTimeOffset.UtcNow); + var chunkCount = (artifact.Length + ChunkSize - 1) / ChunkSize; + + // One transaction for the whole store. A pointer that commits without its chunks — or + // chunks that commit without the pointer — is a cache that reads back as a corrupt hit + // rather than a clean miss, which is the one outcome this type exists to prevent. + await using var tx = await _db.BeginTransactionAsync(ct); + + // Delete-then-insert rather than upsert-per-chunk: a re-store with FEWER chunks than last + // time would otherwise leave the old tail behind, and those orphans read back as a + // chunk-count mismatch forever. + await tx.ExecuteAsync( + "DELETE FROM deployment_artifacts WHERE deployment_id = @DeploymentId", + new { DeploymentId = deploymentId }, + ct); + + for (var index = 0; index < chunkCount; index++) + { + var offset = index * ChunkSize; + var length = Math.Min(ChunkSize, artifact.Length - offset); + + await tx.ExecuteAsync( + """ + INSERT INTO deployment_artifacts + (deployment_id, chunk_index, cluster_id, revision_hash, chunk_count, + chunk_base64, cached_at_utc) + VALUES + (@DeploymentId, @ChunkIndex, @ClusterId, @RevisionHash, @ChunkCount, + @ChunkBase64, @CachedAtUtc) + """, + new + { + DeploymentId = deploymentId, + ChunkIndex = index, + ClusterId = clusterId, + RevisionHash = revisionHash, + ChunkCount = chunkCount, + ChunkBase64 = Convert.ToBase64String(artifact, offset, length), + CachedAtUtc = cachedAtUtc, + }, + ct); + } + + await tx.ExecuteAsync( + """ + INSERT INTO deployment_pointer + (cluster_id, deployment_id, revision_hash, artifact_sha256, applied_at_utc) + VALUES + (@ClusterId, @DeploymentId, @RevisionHash, @Sha, @AppliedAtUtc) + ON CONFLICT(cluster_id) DO UPDATE SET + deployment_id = excluded.deployment_id, + revision_hash = excluded.revision_hash, + artifact_sha256 = excluded.artifact_sha256, + applied_at_utc = excluded.applied_at_utc + """, + new + { + ClusterId = clusterId, + DeploymentId = deploymentId, + RevisionHash = revisionHash, + Sha = sha, + AppliedAtUtc = cachedAtUtc, + }, + ct); + + // Prune inside the same transaction so the cache is never briefly unbounded. + // + // The `deployment_id <> @DeploymentId` clause is load-bearing, not belt-and-braces: it makes + // "the deployment the pointer names is always present" a structural invariant instead of a + // consequence of clock resolution. Without it, three stores landing inside one timestamp + // tick fall through to the `deployment_id DESC` tiebreak — and since real deployment ids are + // GUIDs, that ordering is effectively random, so the row just written can lose. The pointer + // would then name chunks that no longer exist, which reads back as a cache miss on every + // subsequent boot until the next deploy overwrites it. Silent and permanent: exactly the + // failure this cache exists to prevent. + await tx.ExecuteAsync( + $""" + DELETE FROM deployment_artifacts + WHERE cluster_id = @ClusterId + AND deployment_id <> @DeploymentId + AND deployment_id NOT IN ( + SELECT deployment_id + FROM deployment_artifacts + WHERE cluster_id = @ClusterId + GROUP BY deployment_id + ORDER BY MAX(cached_at_utc) DESC, deployment_id DESC + LIMIT {RetainedDeployments} + ) + """, + new { ClusterId = clusterId, DeploymentId = deploymentId }, + ct); + + await tx.CommitAsync(ct); + } + + /// + public async Task GetCurrentAsync( + string clusterId, CancellationToken ct = default) + { + ArgumentException.ThrowIfNullOrWhiteSpace(clusterId); + + var pointers = await _db.QueryAsync( + """ + SELECT deployment_id, revision_hash, artifact_sha256, applied_at_utc + FROM deployment_pointer + WHERE cluster_id = @ClusterId + """, + ReadPointer, + new { ClusterId = clusterId }, + ct); + + if (pointers.Count == 0) + { + _logger.LogDebug("No cached deployment pointer for cluster {ClusterId}.", clusterId); + return null; + } + + return await ReassembleAsync(clusterId, pointers[0], ct); + } + + /// + public async Task GetCurrentUnkeyedAsync(CancellationToken ct = default) + { + // applied_at_utc is round-trip ISO-8601 UTC, so ordering it as TEXT is chronological — + // that is the reason the format is pinned rather than left to the current culture. + var pointers = await _db.QueryAsync( + """ + SELECT cluster_id, deployment_id, revision_hash, artifact_sha256, applied_at_utc + FROM deployment_pointer + ORDER BY applied_at_utc DESC + """, + r => (ClusterId: r.GetString(0), Pointer: new PointerRow( + DeploymentId: r.GetString(1), + RevisionHash: r.GetString(2), + ArtifactSha256: r.GetString(3), + AppliedAtUtc: r.GetString(4))), + parameters: null, + ct); + + if (pointers.Count == 0) + { + _logger.LogDebug("No cached deployment pointer of any cluster."); + return null; + } + + if (pointers.Count > 1) + { + // The node was re-homed between clusters. Booting the newest is the only defensible + // choice, but it must never be a silent one — a wrong-cluster address space presents as + // a plausible configuration, so the operator needs the cluster names to spot it. + _logger.LogWarning( + "Local deployment cache holds pointers for {PointerCount} clusters ({ClusterIds}); " + + "booting the newest ({ChosenClusterId}). This node appears to have been re-homed — " + + "clear the stale cache if that is unexpected.", + pointers.Count, + string.Join(", ", pointers.Select(p => p.ClusterId)), + pointers[0].ClusterId); + } + + return await ReassembleAsync(pointers[0].ClusterId, pointers[0].Pointer, ct); + } + + /// + /// Rebuilds the artifact the pointer names, returning unless every + /// integrity check passes. + /// + private async Task ReassembleAsync( + string clusterId, PointerRow pointer, CancellationToken ct) + { + var chunks = await _db.QueryAsync( + """ + SELECT chunk_count, chunk_base64 + FROM deployment_artifacts + WHERE deployment_id = @DeploymentId + ORDER BY chunk_index + """, + r => (ChunkCount: r.GetInt32(0), Base64: r.GetString(1)), + new { pointer.DeploymentId }, + ct); + + // The chunk_count carried on every row is what makes a partial replica detectable. Without + // it a missing tail chunk is indistinguishable from a shorter artifact. + var expectedChunkCount = chunks.Count > 0 ? chunks[0].ChunkCount : 0; + + if (chunks.Count != expectedChunkCount) + { + _logger.LogWarning( + "Cached deployment {DeploymentId} for cluster {ClusterId} is incomplete: found " + + "{FoundChunks} of {ExpectedChunks} chunks. Treating as a cache miss.", + pointer.DeploymentId, clusterId, chunks.Count, expectedChunkCount); + return null; + } + + byte[] artifact; + try + { + artifact = Decode(chunks.Select(c => c.Base64)); + } + catch (FormatException ex) + { + _logger.LogWarning( + ex, + "Cached deployment {DeploymentId} for cluster {ClusterId} has a chunk that is not " + + "valid base64. Treating as a cache miss.", + pointer.DeploymentId, clusterId); + return null; + } + + var actualSha = Convert.ToHexString(SHA256.HashData(artifact)); + if (!string.Equals(actualSha, pointer.ArtifactSha256, StringComparison.OrdinalIgnoreCase)) + { + _logger.LogWarning( + "Cached deployment {DeploymentId} for cluster {ClusterId} failed its SHA-256 check " + + "(expected {ExpectedSha}, computed {ActualSha}). Treating as a cache miss.", + pointer.DeploymentId, clusterId, pointer.ArtifactSha256, actualSha); + return null; + } + + if (!DateTimeOffset.TryParse(pointer.AppliedAtUtc, CultureInfo.InvariantCulture, + DateTimeStyles.RoundtripKind, out var appliedAtUtc)) + { + _logger.LogWarning( + "Cached deployment {DeploymentId} for cluster {ClusterId} has an unparseable " + + "applied_at_utc ({AppliedAtUtc}). Treating as a cache miss.", + pointer.DeploymentId, clusterId, pointer.AppliedAtUtc); + return null; + } + + return new CachedDeploymentArtifact( + pointer.DeploymentId, pointer.RevisionHash, artifact, appliedAtUtc); + } + + private static byte[] Decode(IEnumerable chunksInOrder) + { + using var buffer = new MemoryStream(); + + foreach (var chunk in chunksInOrder) + { + var bytes = Convert.FromBase64String(chunk); + buffer.Write(bytes, 0, bytes.Length); + } + + return buffer.ToArray(); + } + + private static PointerRow ReadPointer(Microsoft.Data.Sqlite.SqliteDataReader reader) + => new( + DeploymentId: reader.GetString(0), + RevisionHash: reader.GetString(1), + ArtifactSha256: reader.GetString(2), + AppliedAtUtc: reader.GetString(3)); + + /// + /// Round-trip ("O") UTC, so lexicographic ordering of the stored TEXT is chronological + /// ordering — the retention query sorts on it directly. + /// + private static string FormatTimestamp(DateTimeOffset value) + => value.ToUniversalTime().ToString("O", CultureInfo.InvariantCulture); + + private sealed record PointerRow( + string DeploymentId, string RevisionHash, string ArtifactSha256, string AppliedAtUtc); +} diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Drivers/DriverHostActor.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Drivers/DriverHostActor.cs index 4a87e53c..e3b49eba 100644 --- a/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Drivers/DriverHostActor.cs +++ b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/Drivers/DriverHostActor.cs @@ -23,6 +23,7 @@ using ZB.MOM.WW.OtOpcUa.Core.ScriptedAlarms; using ZB.MOM.WW.OtOpcUa.Core.Scripting; using ZB.MOM.WW.OtOpcUa.Core.VirtualTags; using ZB.MOM.WW.OtOpcUa.OpcUaServer; +using ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache; using ZB.MOM.WW.OtOpcUa.Runtime.ScriptedAlarms; using ZB.MOM.WW.OtOpcUa.Runtime.VirtualTags; using CommonsNodeId = ZB.MOM.WW.OtOpcUa.Commons.Types.NodeId; @@ -57,6 +58,31 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers /// Publishing interval handed to each driver's SubscribeBulk pass after an apply. private static readonly TimeSpan SubscriptionPublishingInterval = TimeSpan.FromSeconds(1); + /// + /// Cache key used when an artifact carries no cluster scoping (single-cluster or unscoped + /// deployments). A literal rather than an empty string so the row is visibly deliberate when + /// someone reads the table during an incident. + /// + private const string SingleClusterCacheKey = "__single"; + + /// + /// Node-local cache of applied deployment artifacts, or null when this node has none + /// (admin-only graphs, and tests that do not exercise it). + /// + private readonly IDeploymentArtifactCache? _deploymentArtifactCache; + + /// + /// True once this node has booted a cached artifact because central SQL was unreachable. + /// + /// + /// Surfaced on because a node running from cache looks + /// completely healthy from the outside — it serves a full address space with live values. + /// The difference is that its configuration is frozen at whatever the cache held, and no + /// deployment can reach it. Without an explicit signal that is invisible until someone + /// wonders why a deploy "succeeded" everywhere but did not take effect here. + /// + private bool _isRunningFromCache; + private readonly IDbContextFactory _dbFactory; private readonly CommonsNodeId _localNode; private readonly IActorRef? _coordinatorOverride; @@ -339,11 +365,18 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers ScriptRootLogger? scriptRootLogger = null, IActorRef? scriptedAlarmHostOverride = null, IDriverCapabilityInvokerFactory? invokerFactory = null, - Func? driverMemberCountProvider = null) => + Func? driverMemberCountProvider = null, + IDeploymentArtifactCache? deploymentArtifactCache = null) => + // WARNING: this forwarding list is POSITIONAL, and Props.Create compiles it into an + // expression tree. Six IActorRef? parameters and several interface-typed ones mean a + // mis-ordered argument is usually type-compatible and therefore compiles clean, then binds + // the wrong dependency at runtime. New parameters go LAST in all three places — this + // signature, the constructor's, and this call — and nothing else moves. Akka.Actor.Props.Create(() => new DriverHostActor( dbFactory, localNode, coordinator, driverFactory, localRoles, dependencyMux, opcUaPublishActor, healthPublisher, virtualTagEvaluator, historyWriter, virtualTagHostOverride, - loggerFactory, scriptRootLogger, scriptedAlarmHostOverride, invokerFactory, driverMemberCountProvider)); + loggerFactory, scriptRootLogger, scriptedAlarmHostOverride, invokerFactory, driverMemberCountProvider, + deploymentArtifactCache)); /// Initializes a new DriverHostActor with the specified dependencies. /// Database context factory for configuration database access. @@ -372,6 +405,10 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers /// Test seam (archreview 03/S4): overrides the count of Up /// driver-role cluster members the Primary gate reads while the role is unknown. When null the /// default reads Cluster.Get(Context.System).State.Members (0 on a non-cluster ActorRefProvider). + /// Optional node-local cache of applied deployment artifacts. + /// When supplied, each successful apply stores its artifact so the node can boot from its + /// last-known-good configuration while central SQL is unreachable. Null on admin-only nodes and in + /// tests that do not exercise the cache — caching is then simply skipped. public DriverHostActor( IDbContextFactory dbFactory, CommonsNodeId localNode, @@ -388,8 +425,10 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers ScriptRootLogger? scriptRootLogger = null, IActorRef? scriptedAlarmHostOverride = null, IDriverCapabilityInvokerFactory? invokerFactory = null, - Func? driverMemberCountProvider = null) + Func? driverMemberCountProvider = null, + IDeploymentArtifactCache? deploymentArtifactCache = null) { + _deploymentArtifactCache = deploymentArtifactCache; _dbFactory = dbFactory; _localNode = localNode; _coordinatorOverride = coordinator; @@ -541,7 +580,7 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers // space were lost on restart. Re-spawn + re-materialise + re-subscribe from the // applied deployment so a restarted/rebuilt node restores its served state instead // of waiting for a config change (whose identical-config revision would no-op). - RestoreApplied(new DeploymentId(latest.DeploymentId)); + RestoreApplied(new DeploymentId(latest.DeploymentId), revision); break; case NodeDeploymentStatus.Applying: _log.Warning("DriverHost {Node}: found orphan Applying row for deployment {Id}; replaying", @@ -560,10 +599,122 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers catch (Exception ex) { _log.Warning(ex, "DriverHost {Node}: ConfigDb unreachable on bootstrap; entering Stale", _localNode); - Become(Stale); + + // Central is unreachable, so try the node-local cache before giving up. On a hit this + // node serves its last-known-good configuration through the outage instead of coming up + // with an empty address space. On a miss, behaviour is exactly what it always was. + if (!TryBootFromCache()) + Become(Stale); } } + /// + /// Last-resort boot path: apply the artifact cached by a previous successful deploy (or + /// replicated from this node's pair peer) when central SQL cannot be reached. + /// + /// + /// when a cached artifact was applied and the actor is now Steady; + /// when the caller should fall through to Stale. + /// + /// + /// + /// The read is unkeyed. The cache is keyed by ClusterId so a pair shares one + /// entry, but ClusterId is only derivable from an artifact you already hold or from the + /// central DB — and at this seam we have neither. So the newest pointer row wins, which + /// is correct in every real topology because a node belongs to one cluster and its peer + /// replicates that same cluster's row. A node re-homed between clusters is the only + /// ambiguous case, and the cache logs a warning naming both. + /// + /// + /// This does not make the node current. _currentRevision is set from the + /// cached artifact, so a subsequent dispatch of that same revision correctly no-ops, + /// while any NEW revision still requires central — the cache holds the past, not the + /// future. Dispatch handling is unchanged. + /// + /// + /// Never throws: a fault here must degrade to Stale, which is exactly where the node + /// would have been without a cache at all. + /// + /// + private bool TryBootFromCache() + { + if (_deploymentArtifactCache is null) + return false; + + try + { + var cached = _deploymentArtifactCache.GetCurrentUnkeyedAsync().GetAwaiter().GetResult(); + if (cached is null) + { + _log.Info( + "DriverHost {Node}: no cached deployment available; entering Stale with no configuration.", + _localNode); + return false; + } + + var deploymentId = DeploymentId.Parse(cached.DeploymentId); + var revision = RevisionHash.Parse(cached.RevisionHash); + + // Steady, not Stale: this node is serving a real configuration. The retry-db timer that + // Stale would have started does not run here, so recovery rides on the next dispatch — + // matching how a normally-booted node behaves. + _currentRevision = revision; + Become(Steady); + + ApplyCachedArtifact(deploymentId, cached.Artifact); + + _isRunningFromCache = true; + _log.Warning( + "DriverHost {Node}: RUNNING FROM CACHE — central ConfigDb is unreachable, so this node " + + "booted deployment {Id} (rev {Rev}) cached at {CachedAtUtc:o} from its local database. " + + "Configuration changes cannot be applied until the ConfigDb is reachable again.", + _localNode, deploymentId, revision, cached.AppliedAtUtc); + + return true; + } + catch (Exception ex) + { + _log.Error(ex, + "DriverHost {Node}: failed to boot from the local deployment cache; falling back to Stale.", + _localNode); + return false; + } + } + + /// + /// Applies an artifact already in hand — no ConfigDb read, no ACK. + /// + /// + /// Mirrors , but sources the artifact from the cache rather than + /// re-reading it from a database that is by definition unreachable here. It also skips + /// UpsertNodeDeploymentState and SendAck for the same reason: both write to or + /// depend on central. + /// + private void ApplyCachedArtifact(DeploymentId deploymentId, byte[] blob) + { + var correlation = CorrelationId.NewId(); + + var specs = DeploymentArtifact.ParseDriverInstances(blob, _localNode.Value); + var snapshots = _children.ToDictionary( + kv => kv.Key, + kv => new DriverChildSnapshot(kv.Value.DriverType, kv.Value.LastConfigJson, kv.Value.ResilienceConfig), + StringComparer.Ordinal); + var plan = DriverSpawnPlanner.Compute(snapshots, specs); + + foreach (var id in plan.ToStop) StopChild(id); + foreach (var spec in plan.ToApplyDelta) ApplyChildDelta(spec); + foreach (var spec in plan.ToSpawn) SpawnChild(spec); + + // Hand the cached blob to the rebuild so it materialises the client-facing address space from + // it directly. Passing only the deploymentId would drive a ConfigDb read — unreachable here by + // definition — and the rebuild would no-op, leaving OPC UA clients browsing an empty server + // while the drivers below poll happily. (Found on the docker-dev live gate.) + _opcUaPublishActor?.Tell( + new ZB.MOM.WW.OtOpcUa.Runtime.OpcUa.OpcUaPublishActor.RebuildAddressSpace(correlation, deploymentId, blob)); + + PushDesiredSubscriptionsFromArtifact(deploymentId, blob); + } + private void Steady() { Receive(HandleDispatchFromSteady); @@ -1393,7 +1544,8 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers NodeId: _localNode, CurrentRevision: _currentRevision, Drivers: drivers, - AsOfUtc: DateTime.UtcNow); + AsOfUtc: DateTime.UtcNow, + RunningFromCache: _isRunningFromCache); Sender.Tell(snapshot); } @@ -1426,7 +1578,7 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers try { - ReconcileDrivers(deploymentId); + var appliedBlob = ReconcileDrivers(deploymentId); _currentRevision = revision; UpsertNodeDeploymentState(deploymentId, NodeDeploymentStatus.Applied, failureReason: null); SendAck(deploymentId, ApplyAckOutcome.Applied, failureReason: null, correlation); @@ -1437,6 +1589,12 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers // SubscribeBulk pass: hand each driver its desired tag references so live values flow into // the just-rebuilt address space instead of staying BadWaitingForInitialData. PushDesiredSubscriptions(deploymentId); + CacheAppliedArtifact(deploymentId, revision, appliedBlob); + // Reaching here means the artifact came from central, so the node is no longer serving a + // cache-sourced configuration. Note this only clears on a real apply: a dispatch of the + // revision already booted from cache short-circuits in HandleDispatchFromSteady without + // touching the ConfigDb, and the flag correctly stays set. + _isRunningFromCache = false; OtOpcUaTelemetry.DeploymentApplied.Add(1, new KeyValuePair("outcome", "ack")); _log.Info("DriverHost {Node}: applied deployment {Id} (rev {Rev}, children={Count})", _localNode, deploymentId, revision, _children.Count); @@ -1464,7 +1622,20 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers /// configured can't materialise any of the requested /// types, this is effectively a no-op. /// - private void ReconcileDrivers(DeploymentId deploymentId) + /// + /// The artifact blob that was reconciled, or when it could not be + /// loaded at all. + /// + /// + /// Returning the blob rather than swallowing it is load-bearing for the artifact cache. This + /// method catches its own DB failures, logs a warning and returns WITHOUT rethrowing — so + /// proceeds to its success path, ACKs Applied and logs success + /// having spawned zero drivers. A cache write that trusted "the apply succeeded" would then + /// persist an empty artifact as this node's last-known-good configuration, and the node would + /// later boot from it into an empty address space. The caller distinguishes the two cases by + /// this return value. + /// + private byte[]? ReconcileDrivers(DeploymentId deploymentId) { byte[] blob; try @@ -1479,7 +1650,7 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers { _log.Warning(ex, "DriverHost {Node}: failed to load artifact for {Id}; skipping reconcile", _localNode, deploymentId); - return; + return null; } var specs = DeploymentArtifact.ParseDriverInstances(blob, _localNode.Value); @@ -1500,6 +1671,74 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers // sequenced AFTER ReinitializeAsync rebuilds DependencyRefs — a synchronous re-register HERE would // re-read the STALE ref set (ApplyDelta runs asynchronously on the child), so it is deliberately NOT // done inline. + + return blob; + } + + /// + /// Store a successfully applied artifact in the node-local cache, so this node can boot from + /// it while central SQL is unreachable. + /// + /// + /// + /// Never throws. By the time this runs the deployment is already recorded Applied + /// in central SQL and an Applied ACK has been sent to the coordinator. An exception + /// escaping here would unwind into 's catch and send a second, + /// contradictory Failed ACK for a deployment the fleet already believes is live. + /// + /// + /// An empty or unloadable blob is skipped, not cached. See + /// : a DB failure there degrades silently, so "the apply + /// succeeded" does not imply "a real configuration was applied". Caching an empty + /// artifact would make it this node's last-known-good and boot it into an empty address + /// space during the next outage — strictly worse than having no cache at all. + /// + /// + /// Runs synchronously on the actor thread. That matches every other DB call on this path + /// (all of which already block), and the alternative — piping the result back as a + /// self-message — would need a handler registered in Steady, Applying AND Stale or it + /// dead-letters after the Become(Steady) in the enclosing finally. + /// + /// + private void CacheAppliedArtifact(DeploymentId deploymentId, RevisionHash revision, byte[]? blob) + { + if (_deploymentArtifactCache is null) + return; + + if (blob is null || blob.Length == 0) + { + _log.Warning( + "DriverHost {Node}: not caching deployment {Id} — its artifact was empty or could not " + + "be loaded, so it is not a usable last-known-good configuration.", + _localNode, deploymentId); + return; + } + + try + { + // The node's ClusterId is carried inside the artifact (ClusterNode rows), not in config + // or on IClusterRoleInfo. A single-cluster or unscoped artifact has none, so it shares + // one sentinel key — the pair still converges on a single pointer row either way. + var scope = DeploymentArtifact.ResolveClusterScope(blob, _localNode.Value); + var clusterId = scope.Mode == ClusterFilterMode.ScopeTo && !string.IsNullOrWhiteSpace(scope.ClusterId) + ? scope.ClusterId + : SingleClusterCacheKey; + + _deploymentArtifactCache + .StoreAsync(clusterId, deploymentId.ToString(), revision.Value, blob) + .GetAwaiter() + .GetResult(); + + _log.Debug("DriverHost {Node}: cached deployment {Id} (rev {Rev}, cluster {ClusterId}, {Bytes} bytes)", + _localNode, deploymentId, revision, clusterId, blob.Length); + } + catch (Exception ex) + { + _log.Error(ex, + "DriverHost {Node}: failed to cache applied deployment {Id}. The apply itself succeeded " + + "and is unaffected; this node simply has no local fallback for that revision.", + _localNode, deploymentId); + } } /// @@ -1511,14 +1750,24 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers /// drivers, rebuilds the address space from the applied artifact, and re-pushes SubscribeBulk. /// No re-ack: the deployment is already Applied. /// - private void RestoreApplied(DeploymentId deploymentId) + private void RestoreApplied(DeploymentId deploymentId, RevisionHash? revision) { var correlation = CorrelationId.NewId(); try { - ReconcileDrivers(deploymentId); - _opcUaPublishActor?.Tell(new ZB.MOM.WW.OtOpcUa.Runtime.OpcUa.OpcUaPublishActor.RebuildAddressSpace(correlation, deploymentId)); + var appliedBlob = ReconcileDrivers(deploymentId); + _opcUaPublishActor?.Tell(new ZB.MOM.WW.OtOpcUa.Runtime.OpcUa.OpcUaPublishActor.RebuildAddressSpace(correlation, deploymentId, appliedBlob)); PushDesiredSubscriptions(deploymentId); + // Populate the node-local cache from the artifact this restored node is now serving. The + // cache invariant is "holds what the node currently serves"; without this only a FRESH + // apply (ApplyAndAck) ever wrote it, so a node whose cache was lost — a wiped/fresh volume, + // a disk failure — would recover its served state here yet stay cache-less, unable to + // boot-from-cache on the NEXT central outage until some future new deploy happened to land. + // Replication does not heal that gap either: a peer's already-acked rows are pruned from its + // oplog, so a fully-wiped node is never back-filled. Re-caching on restore is what closes + // it. (Found on the docker-dev live gate.) + if (revision is { } rev) + CacheAppliedArtifact(deploymentId, rev, appliedBlob); _log.Info("DriverHost {Node}: restored served state for applied deployment {Id} on bootstrap", _localNode, deploymentId); } catch (Exception ex) @@ -1552,6 +1801,19 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers return; } + PushDesiredSubscriptionsFromArtifact(deploymentId, blob); + } + + /// + /// The artifact-processing half of , split out so the + /// boot-from-cache path can drive it with bytes already in hand. + /// + /// + /// Separated because the cache path runs precisely when the ConfigDb read above cannot + /// succeed — re-reading the artifact there would fail by definition. + /// + private void PushDesiredSubscriptionsFromArtifact(DeploymentId deploymentId, byte[] blob) + { AddressSpaceComposition composition; try { diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/OpcUa/OpcUaPublishActor.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/OpcUa/OpcUaPublishActor.cs index c8d9d7d3..f19127ab 100644 --- a/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/OpcUa/OpcUaPublishActor.cs +++ b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/OpcUa/OpcUaPublishActor.cs @@ -73,7 +73,17 @@ public sealed class OpcUaPublishActor : ReceiveActor, IWithTimers /// applied config + the SubscribeBulk pass. It is null only for legacy/dev callers, which /// fall back to the latest sealed deployment (lags a not-yet-sealed apply by one revision). /// - public sealed record RebuildAddressSpace(CorrelationId Correlation, DeploymentId? DeploymentId = null); + /// Correlation id for tracing this rebuild. + /// The applied deployment whose artifact to materialise, or null for the latest-sealed fallback. + /// + /// The artifact bytes already in hand, used INSTEAD of loading them from the ConfigDb. This is + /// what makes boot-from-cache actually serve its address space: on a central-SQL outage the + /// host has the cached blob but alone would drive a ConfigDb read + /// that cannot succeed, leaving the rebuild a no-op and clients browsing an empty server. Null + /// on the normal path, where loading from the ConfigDb by id is correct. + /// + public sealed record RebuildAddressSpace( + CorrelationId Correlation, DeploymentId? DeploymentId = null, byte[]? Artifact = null); /// Inject driver-discovered nodes (FixedTree) under an equipment at runtime (post-connect). /// The OPC UA NodeId of the equipment root folder to inject the @@ -345,13 +355,15 @@ public sealed class OpcUaPublishActor : ReceiveActor, IWithTimers try { - // Prefer the artifact of the deployment the host just applied — at apply time it is not - // yet Sealed, so LoadLatestArtifact would return the PREVIOUS revision and materialise a - // stale composition (variables that don't match the SubscribeBulk refs). Fall back to - // latest-sealed only for legacy callers that don't carry a DeploymentId. - var artifact = msg.DeploymentId is { } depId - ? LoadArtifact(depId) - : LoadLatestArtifact(); + // An in-hand artifact (boot-from-cache) wins: the host already has the cached bytes, and + // the ConfigDb read the DeploymentId path would do is exactly what is unreachable during + // the outage this exists to survive. Otherwise prefer the artifact of the deployment the + // host just applied — at apply time it is not yet Sealed, so LoadLatestArtifact would + // return the PREVIOUS revision and materialise a stale composition (variables that don't + // match the SubscribeBulk refs). Fall back to latest-sealed only for legacy callers that + // carry neither. + var artifact = msg.Artifact + ?? (msg.DeploymentId is { } depId ? LoadArtifact(depId) : LoadLatestArtifact()); var composition = _localNode is { } ln ? DeploymentArtifact.ParseComposition(artifact, ln.Value, inconsistency => _log.Warning("OpcUaPublish {Node}: cross-cluster binding — {Message}", ln, inconsistency)) diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/ServiceCollectionExtensions.cs b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/ServiceCollectionExtensions.cs index d3d7a4ec..0cddd870 100644 --- a/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/ServiceCollectionExtensions.cs +++ b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/ServiceCollectionExtensions.cs @@ -15,6 +15,7 @@ using ZB.MOM.WW.OtOpcUa.Core.AlarmHistorian; using ZB.MOM.WW.OtOpcUa.Core.Scripting; using ZB.MOM.WW.OtOpcUa.Core.VirtualTags; using ZB.MOM.WW.OtOpcUa.OpcUaServer; +using ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache; using ZB.MOM.WW.OtOpcUa.Runtime.Drivers; using ZB.MOM.WW.OtOpcUa.Runtime.Health; using ZB.MOM.WW.OtOpcUa.Runtime.Historian; @@ -221,6 +222,11 @@ public static class ServiceCollectionExtensions var serviceLevel = resolver.GetService() ?? NullServiceLevelPublisher.Instance; var loggerFactory = resolver.GetService() ?? NullLoggerFactory.Instance; var healthPublisher = resolver.GetService() ?? NullDriverHealthPublisher.Instance; + // Node-local deployment-artifact cache. Registered by the Host's AddOtOpcUaLocalDb on + // driver-role nodes only; deliberately left null elsewhere (admin-only graphs, test + // harnesses) rather than given a null-object, so DriverHostActor skips caching outright + // instead of pretending to cache into a sink that drops everything. + var deploymentArtifactCache = resolver.GetService(); // Root script logger backs the ScriptedAlarm host's engine + script logging. Registered in // Host DI inside the hasDriver block; may be absent in some role configs / test harnesses, // in which case the DriverHostActor gracefully skips spawning the ScriptedAlarm host. @@ -338,7 +344,8 @@ public static class ServiceCollectionExtensions historyWriter: historyWriter, loggerFactory: loggerFactory, scriptRootLogger: scriptRootLogger, - invokerFactory: invokerFactory), + invokerFactory: invokerFactory, + deploymentArtifactCache: deploymentArtifactCache), DriverHostActorName); registry.Register(driverHost); diff --git a/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/ZB.MOM.WW.OtOpcUa.Runtime.csproj b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/ZB.MOM.WW.OtOpcUa.Runtime.csproj index 98eb9cce..dc524c84 100644 --- a/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/ZB.MOM.WW.OtOpcUa.Runtime.csproj +++ b/src/Server/ZB.MOM.WW.OtOpcUa.Runtime/ZB.MOM.WW.OtOpcUa.Runtime.csproj @@ -8,6 +8,9 @@ + + diff --git a/tests/Core/ZB.MOM.WW.OtOpcUa.Configuration.Tests/GenerationSealedCacheTests.cs b/tests/Core/ZB.MOM.WW.OtOpcUa.Configuration.Tests/GenerationSealedCacheTests.cs deleted file mode 100644 index d7ed3121..00000000 --- a/tests/Core/ZB.MOM.WW.OtOpcUa.Configuration.Tests/GenerationSealedCacheTests.cs +++ /dev/null @@ -1,168 +0,0 @@ -using Shouldly; -using Xunit; -using ZB.MOM.WW.OtOpcUa.Configuration.LocalCache; - -namespace ZB.MOM.WW.OtOpcUa.Configuration.Tests; - -[Trait("Category", "Unit")] -public sealed class GenerationSealedCacheTests : IDisposable -{ - private readonly string _root = Path.Combine(Path.GetTempPath(), $"otopcua-sealed-{Guid.NewGuid():N}"); - - /// Cleans up temporary directory after test execution. - public void Dispose() - { - try - { - if (!Directory.Exists(_root)) return; - // Remove ReadOnly attribute first so Directory.Delete can clean sealed files. - foreach (var f in Directory.EnumerateFiles(_root, "*", SearchOption.AllDirectories)) - File.SetAttributes(f, FileAttributes.Normal); - Directory.Delete(_root, recursive: true); - } - catch { /* best-effort cleanup */ } - } - - private GenerationSnapshot MakeSnapshot(string clusterId, long generationId, string payload = "{\"sample\":true}") => - new() - { - ClusterId = clusterId, - GenerationId = generationId, - CachedAt = DateTime.UtcNow, - PayloadJson = payload, - }; - - /// Verifies that reading a snapshot on first boot with no existing snapshot throws. - [Fact] - public async Task FirstBoot_NoSnapshot_ReadThrows() - { - var cache = new GenerationSealedCache(_root); - - await Should.ThrowAsync( - () => cache.ReadCurrentAsync("cluster-a")); - } - - /// Verifies that sealed snapshots can be read back correctly. - [Fact] - public async Task SealThenRead_RoundTrips() - { - var cache = new GenerationSealedCache(_root); - var snapshot = MakeSnapshot("cluster-a", 42, "{\"hello\":\"world\"}"); - - await cache.SealAsync(snapshot); - - var read = await cache.ReadCurrentAsync("cluster-a"); - read.GenerationId.ShouldBe(42); - read.ClusterId.ShouldBe("cluster-a"); - read.PayloadJson.ShouldBe("{\"hello\":\"world\"}"); - } - - /// Verifies that sealed files are marked read-only on disk. - [Fact] - public async Task SealedFile_IsReadOnly_OnDisk() - { - var cache = new GenerationSealedCache(_root); - await cache.SealAsync(MakeSnapshot("cluster-a", 5)); - - var sealedPath = Path.Combine(_root, "cluster-a", "5.db"); - File.Exists(sealedPath).ShouldBeTrue(); - var attrs = File.GetAttributes(sealedPath); - attrs.HasFlag(FileAttributes.ReadOnly).ShouldBeTrue("sealed file must be read-only"); - } - - /// Verifies that the current generation pointer advances when a new generation is sealed. - [Fact] - public async Task SealingTwoGenerations_PointerAdvances_ToLatest() - { - var cache = new GenerationSealedCache(_root); - await cache.SealAsync(MakeSnapshot("cluster-a", 1)); - await cache.SealAsync(MakeSnapshot("cluster-a", 2)); - - cache.TryGetCurrentGenerationId("cluster-a").ShouldBe(2); - var read = await cache.ReadCurrentAsync("cluster-a"); - read.GenerationId.ShouldBe(2); - } - - /// Verifies that prior generation files are preserved after a new seal. - [Fact] - public async Task PriorGenerationFile_Survives_AfterNewSeal() - { - var cache = new GenerationSealedCache(_root); - await cache.SealAsync(MakeSnapshot("cluster-a", 1)); - await cache.SealAsync(MakeSnapshot("cluster-a", 2)); - - File.Exists(Path.Combine(_root, "cluster-a", "1.db")).ShouldBeTrue( - "prior generations preserved for audit; pruning is separate"); - File.Exists(Path.Combine(_root, "cluster-a", "2.db")).ShouldBeTrue(); - } - - /// Verifies that reading a corrupt sealed file fails safely. - [Fact] - public async Task CorruptSealedFile_ReadFailsClosed() - { - var cache = new GenerationSealedCache(_root); - await cache.SealAsync(MakeSnapshot("cluster-a", 7)); - - // Corrupt the sealed file: clear read-only, truncate, leave pointer intact. - var sealedPath = Path.Combine(_root, "cluster-a", "7.db"); - File.SetAttributes(sealedPath, FileAttributes.Normal); - File.WriteAllBytes(sealedPath, [0x00, 0x01, 0x02]); - - await Should.ThrowAsync( - () => cache.ReadCurrentAsync("cluster-a")); - } - - /// Verifies that reading with a missing sealed file fails safely. - [Fact] - public async Task MissingSealedFile_ReadFailsClosed() - { - var cache = new GenerationSealedCache(_root); - await cache.SealAsync(MakeSnapshot("cluster-a", 3)); - - // Delete the sealed file but leave the pointer — corruption scenario. - var sealedPath = Path.Combine(_root, "cluster-a", "3.db"); - File.SetAttributes(sealedPath, FileAttributes.Normal); - File.Delete(sealedPath); - - await Should.ThrowAsync( - () => cache.ReadCurrentAsync("cluster-a")); - } - - /// Verifies that reading with a corrupt pointer file fails safely. - [Fact] - public async Task CorruptPointerFile_ReadFailsClosed() - { - var cache = new GenerationSealedCache(_root); - await cache.SealAsync(MakeSnapshot("cluster-a", 9)); - - var pointerPath = Path.Combine(_root, "cluster-a", "CURRENT"); - File.WriteAllText(pointerPath, "not-a-number"); - - await Should.ThrowAsync( - () => cache.ReadCurrentAsync("cluster-a")); - } - - /// Verifies that sealing the same generation twice is idempotent. - [Fact] - public async Task SealSameGenerationTwice_IsIdempotent() - { - var cache = new GenerationSealedCache(_root); - await cache.SealAsync(MakeSnapshot("cluster-a", 11)); - await cache.SealAsync(MakeSnapshot("cluster-a", 11, "{\"v\":2}")); - - var read = await cache.ReadCurrentAsync("cluster-a"); - read.PayloadJson.ShouldBe("{\"sample\":true}", "sealed file is immutable; second seal no-ops"); - } - - /// Verifies that independent clusters do not interfere with each other. - [Fact] - public async Task IndependentClusters_DoNotInterfere() - { - var cache = new GenerationSealedCache(_root); - await cache.SealAsync(MakeSnapshot("cluster-a", 1)); - await cache.SealAsync(MakeSnapshot("cluster-b", 10)); - - (await cache.ReadCurrentAsync("cluster-a")).GenerationId.ShouldBe(1); - (await cache.ReadCurrentAsync("cluster-b")).GenerationId.ShouldBe(10); - } -} diff --git a/tests/Core/ZB.MOM.WW.OtOpcUa.Configuration.Tests/LiteDbConfigCacheTests.cs b/tests/Core/ZB.MOM.WW.OtOpcUa.Configuration.Tests/LiteDbConfigCacheTests.cs deleted file mode 100644 index 0da9d589..00000000 --- a/tests/Core/ZB.MOM.WW.OtOpcUa.Configuration.Tests/LiteDbConfigCacheTests.cs +++ /dev/null @@ -1,192 +0,0 @@ -using Shouldly; -using Xunit; -using ZB.MOM.WW.OtOpcUa.Configuration.LocalCache; - -namespace ZB.MOM.WW.OtOpcUa.Configuration.Tests; - -[Trait("Category", "Unit")] -public sealed class LiteDbConfigCacheTests : IDisposable -{ - private readonly string _dbPath = Path.Combine(Path.GetTempPath(), $"otopcua-cache-test-{Guid.NewGuid():N}.db"); - - /// Cleans up the temporary database file. - public void Dispose() - { - if (File.Exists(_dbPath)) File.Delete(_dbPath); - } - - private GenerationSnapshot Snapshot(string cluster, long gen) => new() - { - ClusterId = cluster, - GenerationId = gen, - CachedAt = DateTime.UtcNow, - PayloadJson = $"{{\"g\":{gen}}}", - }; - - /// Verifies that payload is preserved through a write-then-read cycle. - [Fact] - public async Task Roundtrip_preserves_payload() - { - using var cache = new LiteDbConfigCache(_dbPath); - var put = Snapshot("c-1", 42); - await cache.PutAsync(put); - - var got = await cache.GetMostRecentAsync("c-1"); - got.ShouldNotBeNull(); - got!.GenerationId.ShouldBe(42); - got.PayloadJson.ShouldBe(put.PayloadJson); - } - - /// Verifies that GetMostRecentAsync returns the latest generation when multiple exist. - [Fact] - public async Task GetMostRecent_returns_latest_when_multiple_generations_present() - { - using var cache = new LiteDbConfigCache(_dbPath); - foreach (var g in new long[] { 10, 20, 15 }) - await cache.PutAsync(Snapshot("c-1", g)); - - var got = await cache.GetMostRecentAsync("c-1"); - got!.GenerationId.ShouldBe(20); - } - - /// Verifies that GetMostRecentAsync returns null for an unknown cluster. - [Fact] - public async Task GetMostRecent_returns_null_for_unknown_cluster() - { - using var cache = new LiteDbConfigCache(_dbPath); - (await cache.GetMostRecentAsync("ghost")).ShouldBeNull(); - } - - /// Verifies that Prune keeps the latest N generations and drops older ones. - [Fact] - public async Task Prune_keeps_latest_N_and_drops_older() - { - using var cache = new LiteDbConfigCache(_dbPath); - for (long g = 1; g <= 15; g++) - await cache.PutAsync(Snapshot("c-1", g)); - - await cache.PruneOldGenerationsAsync("c-1", keepLatest: 10); - - (await cache.GetMostRecentAsync("c-1"))!.GenerationId.ShouldBe(15); - - // Drop them one by one and count — should be exactly 10 remaining - var count = 0; - while (await cache.GetMostRecentAsync("c-1") is not null) - { - count++; - await cache.PruneOldGenerationsAsync("c-1", keepLatest: Math.Max(0, 10 - count)); - if (count > 20) break; // safety - } - count.ShouldBe(10); - } - - /// Verifies that writing the same cluster/generation twice replaces rather than duplicates. - [Fact] - public async Task Put_same_cluster_generation_twice_replaces_not_duplicates() - { - using var cache = new LiteDbConfigCache(_dbPath); - var first = Snapshot("c-1", 1); - first.PayloadJson = "{\"v\":1}"; - await cache.PutAsync(first); - - var second = Snapshot("c-1", 1); - second.PayloadJson = "{\"v\":2}"; - await cache.PutAsync(second); - - (await cache.GetMostRecentAsync("c-1"))!.PayloadJson.ShouldBe("{\"v\":2}"); - } - - // ------------------------------------------------------------------------------------ - // Configuration-005 — concurrent PutAsync for the same (ClusterId, GenerationId) must - // not produce duplicate rows. The original find-then-insert was non-atomic so two racing - // callers could both observe `existing is null` and both Insert. - // ------------------------------------------------------------------------------------ - /// Verifies that concurrent PutAsync calls for the same cluster and generation do not create duplicates. - [Fact] - public async Task PutAsync_concurrent_for_same_cluster_and_generation_does_not_duplicate() - { - using var cache = new LiteDbConfigCache(_dbPath); - // Pre-seed gen=99 so prune keepLatest:1 has a sentinel that survives independent of - // any potential duplicate (gen=42) row count. - await cache.PutAsync(Snapshot("c-1", 99)); - - // Many parallel writes for the same key. Without serialization, racing find-then-insert - // would Insert multiple rows for the same (ClusterId, GenerationId=42). - var tasks = Enumerable.Range(0, 64).Select(_ => Task.Run(async () => - { - var s = Snapshot("c-1", 42); - await cache.PutAsync(s); - })).ToArray(); - - await Task.WhenAll(tasks); - - // Count rows for gen=42 directly by inspecting the LiteDB file via a fresh handle. - cache.Dispose(); - using var verify = new LiteDB.LiteDatabase(_dbPath); - var col = verify.GetCollection("generations"); - var gen42Count = col.Find(s => s.ClusterId == "c-1" && s.GenerationId == 42).Count(); - gen42Count.ShouldBe(1, - $"PutAsync must upsert atomically — found {gen42Count} rows for (c-1, gen=42) after 64 concurrent puts"); - } - - // ------------------------------------------------------------------------------------ - // Configuration-012 — the per-instance _writeGate (Configuration-005) does not protect - // against LiteDB's process-wide BsonMapper.Global lazy-init race. Many cache INSTANCES - // constructed + driven concurrently corrupt the shared global mapper, surfacing as - // "Member ClusterId not found on BsonMapper" or a bogus "duplicate key _id = 0". A private - // per-database mapper with the entity pre-registered fixes it. - // ------------------------------------------------------------------------------------ - /// Verifies that many cache instances constructed and driven concurrently do not - /// corrupt LiteDB's shared global BsonMapper — each Put/Get round-trips its own payload and - /// no insert throws a member-not-found or duplicate-_id exception. - [Fact] - public async Task Concurrent_cache_instances_do_not_race_the_shared_bson_mapper() - { - var paths = new List(); - try - { - var outer = Enumerable.Range(0, 24).Select(i => Task.Run(async () => - { - var path = Path.Combine(Path.GetTempPath(), $"otopcua-cache-mapperrace-{Guid.NewGuid():N}.db"); - lock (paths) paths.Add(path); - - using var cache = new LiteDbConfigCache(path); - // Pre-seed a sentinel, then hammer one (cluster, gen) from many threads. - await cache.PutAsync(Snapshot($"c-{i}", 99)); - var inner = Enumerable.Range(0, 16) - .Select(_ => Task.Run(() => cache.PutAsync(Snapshot($"c-{i}", 42)))) - .ToArray(); - await Task.WhenAll(inner); - - var got = await cache.GetMostRecentAsync($"c-{i}"); - got.ShouldNotBeNull(); - got!.GenerationId.ShouldBe(99); // 99 > 42, latest by GenerationId - })).ToArray(); - - // The unfixed code throws LiteException / NotSupportedException out of these tasks under - // the global-mapper race; the fixed code completes cleanly. - await Task.WhenAll(outer); - } - finally - { - foreach (var p in paths) - if (File.Exists(p)) File.Delete(p); - } - } - - /// Verifies that a corrupted cache file surfaces as LocalConfigCacheCorruptException. - [Fact] - public void Corrupt_file_surfaces_as_LocalConfigCacheCorruptException() - { - // Write a file large enough to look like a LiteDB page but with garbage contents so page - // deserialization fails on the first read probe. - File.WriteAllBytes(_dbPath, new byte[8192]); - Array.Fill(File.ReadAllBytes(_dbPath), 0xAB); - using (var fs = File.OpenWrite(_dbPath)) - { - fs.Write(new byte[8192].Select(_ => (byte)0xAB).ToArray()); - } - - Should.Throw(() => new LiteDbConfigCache(_dbPath)); - } -} diff --git a/tests/Core/ZB.MOM.WW.OtOpcUa.Configuration.Tests/ResilientConfigReaderTests.cs b/tests/Core/ZB.MOM.WW.OtOpcUa.Configuration.Tests/ResilientConfigReaderTests.cs deleted file mode 100644 index fd24bdbd..00000000 --- a/tests/Core/ZB.MOM.WW.OtOpcUa.Configuration.Tests/ResilientConfigReaderTests.cs +++ /dev/null @@ -1,355 +0,0 @@ -using Microsoft.Extensions.Logging; -using Microsoft.Extensions.Logging.Abstractions; -using Polly.Timeout; -using Shouldly; -using Xunit; -using ZB.MOM.WW.OtOpcUa.Configuration.LocalCache; - -namespace ZB.MOM.WW.OtOpcUa.Configuration.Tests; - -[Trait("Category", "Unit")] -public sealed class ResilientConfigReaderTests : IDisposable -{ - private readonly string _root = Path.Combine(Path.GetTempPath(), $"otopcua-reader-{Guid.NewGuid():N}"); - - /// Disposes temporary test files. - public void Dispose() - { - try - { - if (!Directory.Exists(_root)) return; - foreach (var f in Directory.EnumerateFiles(_root, "*", SearchOption.AllDirectories)) - File.SetAttributes(f, FileAttributes.Normal); - Directory.Delete(_root, recursive: true); - } - catch { /* best-effort */ } - } - - /// Verifies that successful central DB reads return value and mark fresh. - [Fact] - public async Task CentralDbSucceeds_ReturnsValue_MarksFresh() - { - var cache = new GenerationSealedCache(_root); - var flag = new StaleConfigFlag { }; - flag.MarkStale(); // pre-existing stale state - var reader = new ResilientConfigReader(cache, flag, NullLogger.Instance); - - var result = await reader.ReadAsync( - "cluster-a", - _ => ValueTask.FromResult("fresh-from-db"), - _ => "from-cache", - CancellationToken.None); - - result.ShouldBe("fresh-from-db"); - flag.IsStale.ShouldBeFalse("successful central-DB read clears stale flag"); - } - - /// Verifies that exhausted retries fall back to cache and mark stale. - [Fact] - public async Task CentralDbFails_ExhaustsRetries_FallsBackToCache_MarksStale() - { - var cache = new GenerationSealedCache(_root); - await cache.SealAsync(new GenerationSnapshot - { - ClusterId = "cluster-a", GenerationId = 99, CachedAt = DateTime.UtcNow, - PayloadJson = "{\"cached\":true}", - }); - var flag = new StaleConfigFlag(); - var reader = new ResilientConfigReader(cache, flag, NullLogger.Instance, - timeout: TimeSpan.FromSeconds(10), retryCount: 2); - var attempts = 0; - - var result = await reader.ReadAsync( - "cluster-a", - _ => - { - attempts++; - throw new InvalidOperationException("SQL dead"); -#pragma warning disable CS0162 - return ValueTask.FromResult("never"); -#pragma warning restore CS0162 - }, - snap => snap.PayloadJson, - CancellationToken.None); - - attempts.ShouldBe(3, "1 initial + 2 retries = 3 attempts"); - result.ShouldBe("{\"cached\":true}"); - flag.IsStale.ShouldBeTrue("cache fallback flips stale flag true"); - } - - /// Verifies that DB failure with unavailable cache throws. - [Fact] - public async Task CentralDbFails_AndCacheAlsoUnavailable_Throws() - { - var cache = new GenerationSealedCache(_root); - var flag = new StaleConfigFlag(); - var reader = new ResilientConfigReader(cache, flag, NullLogger.Instance, - timeout: TimeSpan.FromSeconds(10), retryCount: 0); - - await Should.ThrowAsync(async () => - { - await reader.ReadAsync( - "cluster-a", - _ => throw new InvalidOperationException("SQL dead"), - _ => "never", - CancellationToken.None); - }); - - flag.IsStale.ShouldBeFalse("no snapshot ever served, so flag stays whatever it was"); - } - - /// Verifies that cancellation is not retried. - [Fact] - public async Task Cancellation_NotRetried() - { - var cache = new GenerationSealedCache(_root); - var flag = new StaleConfigFlag(); - var reader = new ResilientConfigReader(cache, flag, NullLogger.Instance, - timeout: TimeSpan.FromSeconds(10), retryCount: 5); - using var cts = new CancellationTokenSource(); - cts.Cancel(); - var attempts = 0; - - await Should.ThrowAsync(async () => - { - await reader.ReadAsync( - "cluster-a", - ct => - { - attempts++; - ct.ThrowIfCancellationRequested(); - return ValueTask.FromResult("ok"); - }, - _ => "cache", - cts.Token); - }); - - attempts.ShouldBeLessThanOrEqualTo(1); - } - - // ------------------------------------------------------------------------------------ - // Configuration-006 — command-timeout TaskCanceledException and TimeoutRejectedException - // must fall back to the sealed cache, not propagate as caller cancellation. - // ------------------------------------------------------------------------------------ - - /// Verifies that command timeout TaskCanceledException falls back to cache. - [Fact] - public async Task CommandTimeout_TaskCanceledException_FallsBackToCache() - { - // A SQL command-level timeout surfaces as a TaskCanceledException thrown by the - // delegate itself (not triggered by the caller's CancellationToken). It must be - // treated as a transient failure and trigger the cache fallback, not be mistaken - // for genuine caller cancellation and propagated. - var cache = new GenerationSealedCache(_root); - await cache.SealAsync(new GenerationSnapshot - { - ClusterId = "cluster-b", GenerationId = 7, CachedAt = DateTime.UtcNow, - PayloadJson = "{\"from\":\"cache\"}", - }); - var flag = new StaleConfigFlag(); - var reader = new ResilientConfigReader(cache, flag, NullLogger.Instance, - timeout: TimeSpan.FromSeconds(10), retryCount: 0); - - // Simulate a command-level timeout: TaskCanceledException with no linked token. - var result = await reader.ReadAsync( - "cluster-b", - _ => throw new TaskCanceledException("SQL command timeout (no caller token)"), - snap => snap.PayloadJson, - CancellationToken.None); // caller token is NOT cancelled - - result.ShouldBe("{\"from\":\"cache\"}", - "command-timeout TaskCanceledException must fall back to sealed cache"); - flag.IsStale.ShouldBeTrue("cache fallback marks the stale flag"); - } - - /// Verifies that Polly timeout rejection falls back to cache. - [Fact] - public async Task PollyTimeout_TimeoutRejectedException_FallsBackToCache() - { - // When Polly's own timeout strategy fires it throws TimeoutRejectedException. - // That should trigger the cache fallback just like any other transient error. - var cache = new GenerationSealedCache(_root); - await cache.SealAsync(new GenerationSnapshot - { - ClusterId = "cluster-c", GenerationId = 8, CachedAt = DateTime.UtcNow, - PayloadJson = "{\"from\":\"polly-timeout-cache\"}", - }); - var flag = new StaleConfigFlag(); - // Set an extremely short Polly timeout so the async delay triggers it. - var reader = new ResilientConfigReader(cache, flag, NullLogger.Instance, - timeout: TimeSpan.FromMilliseconds(10), retryCount: 0); - - var result = await reader.ReadAsync( - "cluster-c", - async ct => - { - await Task.Delay(TimeSpan.FromSeconds(5), ct); // far exceeds 10 ms timeout - return "never"; - }, - snap => snap.PayloadJson, - CancellationToken.None); - - result.ShouldBe("{\"from\":\"polly-timeout-cache\"}", - "Polly TimeoutRejectedException must fall back to sealed cache"); - flag.IsStale.ShouldBeTrue("cache fallback marks the stale flag"); - } - - // ------------------------------------------------------------------------------------ - // Configuration-010 — fallback warning log must scrub connection-string fragments and - // must not include the full exception object (which carries the stack and any inner- - // exception chain). Project rule: no credential or connection-string fragment in logs. - // ------------------------------------------------------------------------------------ - - /// Verifies that fallback warnings do not log exceptions or password fragments. - [Fact] - public async Task FallbackWarning_does_not_log_full_exception_object_or_password_fragment() - { - var cache = new GenerationSealedCache(_root); - await cache.SealAsync(new GenerationSnapshot - { - ClusterId = "cluster-e", GenerationId = 1, CachedAt = DateTime.UtcNow, - PayloadJson = "{\"ok\":true}", - }); - var flag = new StaleConfigFlag(); - var capturing = new CapturingLogger(); - var reader = new ResilientConfigReader(cache, flag, capturing, - timeout: TimeSpan.FromSeconds(10), retryCount: 0); - - // Simulated SqlException-style message carrying a connection-string fragment, the - // kind of thing a poorly-wrapped delegate could surface. - const string secretBearingMessage = - "Login failed for user 'sa'. (Server=sql.example.com,1433;User Id=sa;Password=SuperSecret123!)"; - - await reader.ReadAsync( - "cluster-e", - _ => throw new InvalidOperationException(secretBearingMessage), - snap => snap.PayloadJson, - CancellationToken.None); - - var warning = capturing.Records.ShouldHaveSingleItem(); - warning.LogLevel.ShouldBe(LogLevel.Warning); - - // The exception object passed as the first arg to LogWarning(ex, ...) drives the - // formatter's stack-trace dump; capturing it lets us assert the scrubbing surface. - warning.Exception.ShouldBeNull( - "the warning must not attach the raw exception — it can carry connection-string fragments"); - - // The rendered message must not echo password / user-id strings even if the caller - // embedded them in the exception message. - warning.RenderedMessage.ShouldNotContain("Password=", Case.Insensitive); - warning.RenderedMessage.ShouldNotContain("SuperSecret123!"); - warning.RenderedMessage.ShouldNotContain("User Id=", Case.Insensitive); - } - - /// Verifies that caller cancellation propagates rather than falling back. - [Fact] - public async Task CallerCancellation_Propagates_NotFallback() - { - // Explicit caller cancellation must NOT fall back to the sealed cache — the - // caller said stop, so we must stop. - var cache = new GenerationSealedCache(_root); - await cache.SealAsync(new GenerationSnapshot - { - ClusterId = "cluster-d", GenerationId = 9, CachedAt = DateTime.UtcNow, - PayloadJson = "{\"should\":\"not be returned\"}", - }); - var flag = new StaleConfigFlag(); - var reader = new ResilientConfigReader(cache, flag, NullLogger.Instance, - timeout: TimeSpan.FromSeconds(10), retryCount: 0); - using var cts = new CancellationTokenSource(); - cts.Cancel(); - - await Should.ThrowAsync(async () => - { - await reader.ReadAsync( - "cluster-d", - ct => - { - ct.ThrowIfCancellationRequested(); - return ValueTask.FromResult("ok"); - }, - _ => "cache-should-not-be-used", - cts.Token); - }); - - flag.IsStale.ShouldBeFalse("no cache snapshot served on genuine cancellation"); - } -} - -/// Represents a captured log record for testing. -internal sealed record LogRecord(LogLevel LogLevel, string RenderedMessage, Exception? Exception); - -/// Captures log records for assertion in tests. -internal sealed class CapturingLogger : ILogger -{ - /// Gets the list of captured log records. - public List Records { get; } = new(); - - /// Begins a scope (no-op for testing). - /// The type of the scope state. - /// The scope state. - /// A disposable scope handle. - public IDisposable BeginScope(TState state) where TState : notnull => NullScope.Instance; - - /// Returns true to enable all log levels. - /// The log level to check. - /// True to indicate the log level is enabled. - public bool IsEnabled(LogLevel logLevel) => true; - - /// Logs a message by capturing it. - /// The type of the log state. - /// The log level. - /// The event identifier. - /// The log state. - /// The exception, if any. - /// Function to format the log message. - public void Log(LogLevel logLevel, EventId eventId, TState state, Exception? exception, Func formatter) - { - Records.Add(new LogRecord(logLevel, formatter(state, exception), exception)); - } - - /// No-op scope for testing. - private sealed class NullScope : IDisposable - { - /// Gets the singleton instance. - public static readonly NullScope Instance = new(); - - /// Disposes the scope (no-op). - public void Dispose() { } - } -} - -[Trait("Category", "Unit")] -public sealed class StaleConfigFlagTests -{ - /// Verifies that default state is fresh. - [Fact] - public void Default_IsFresh() - { - new StaleConfigFlag().IsStale.ShouldBeFalse(); - } - - /// Verifies that stale and fresh states toggle correctly. - [Fact] - public void MarkStale_ThenFresh_Toggles() - { - var flag = new StaleConfigFlag(); - flag.MarkStale(); - flag.IsStale.ShouldBeTrue(); - flag.MarkFresh(); - flag.IsStale.ShouldBeFalse(); - } - - /// Verifies that concurrent writes converge to the final state. - [Fact] - public void ConcurrentWrites_Converge() - { - var flag = new StaleConfigFlag(); - Parallel.For(0, 1000, i => - { - if (i % 2 == 0) flag.MarkStale(); else flag.MarkFresh(); - }); - flag.MarkFresh(); - flag.IsStale.ShouldBeFalse(); - } -} diff --git a/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDb/LocalDbPairConvergenceTests.cs b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDb/LocalDbPairConvergenceTests.cs new file mode 100644 index 00000000..bfa317d3 --- /dev/null +++ b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDb/LocalDbPairConvergenceTests.cs @@ -0,0 +1,178 @@ +using Xunit; +using ZB.MOM.WW.LocalDb; +using ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache; + +namespace ZB.MOM.WW.OtOpcUa.Host.IntegrationTests.LocalDb; + +/// +/// LocalDb Phase 1 (Task 12) — two driver nodes replicating the deployment-artifact cache over a +/// real loopback h2c transport, through the real fail-closed interceptor. +/// +/// +/// This is the test the whole phase exists to pass: does a redundant driver pair actually share +/// the cached configuration a node boots from when central SQL is down? Every upstream piece — +/// the schema, the DI wiring, the interceptor, the chunked cache — can be individually green +/// while the pair still fails to converge. +/// +[Collection("LocalDbPairConvergence")] +public sealed class LocalDbPairConvergenceTests +{ + private const string Cluster = "cluster-1"; + + /// + /// A deterministic artifact large enough to force multiple 128 KiB chunks (≈ 3 here), so the + /// tests exercise the chunk-split/reassemble path rather than a single-row shortcut. + /// + private static byte[] MultiChunkArtifact(int seed, int size = 300 * 1024) + { + var bytes = new byte[size]; + new Random(seed).NextBytes(bytes); + return bytes; + } + + private static async Task ReadArtifactAsync(IDeploymentArtifactCache cache, string cluster) + { + var cached = await cache.GetCurrentAsync(cluster, TestContext.Current.CancellationToken); + return cached?.Artifact; + } + + [Fact] + public async Task ArtifactStoredOnA_ConvergesToB_ByteIdentical() + { + await using var harness = new LocalDbPairHarness(); + await harness.StartAsync(); + + var deploymentId = Guid.NewGuid().ToString("N"); + var artifact = MultiChunkArtifact(seed: 1); + await harness.CacheA.StoreAsync(Cluster, deploymentId, "rev-1", artifact, + TestContext.Current.CancellationToken); + + // B's cache reassembles byte-for-byte what A stored — proving the chunk rows, chunk_count, + // and pointer all replicated intact through the real transport + interceptor. + await LocalDbPairHarness.WaitUntilAsync( + async () => + { + var onB = await ReadArtifactAsync(harness.CacheB, Cluster); + return onB is not null && onB.AsSpan().SequenceEqual(artifact); + }, + "the artifact stored on node A to be byte-identical on node B"); + + // The pointer row on B carries A's origin HLC + node id, not a locally re-derived stamp: the + // row_version dumps for the pointer table are identical on both nodes. This is what + // distinguishes "B replicated A's row" from "both nodes happened to write equal content". + await LocalDbPairHarness.WaitUntilAsync( + async () => + await LocalDbPairHarness.DumpRowVersionAsync(harness.A, DeploymentCacheSchema.PointerTable) + == await LocalDbPairHarness.DumpRowVersionAsync(harness.B, DeploymentCacheSchema.PointerTable), + "both nodes' pointer row_version (hlc + origin node id) to match"); + } + + [Fact] + public async Task RetentionPruneOnA_TombstonesReachB() + { + await using var harness = new LocalDbPairHarness(); + await harness.StartAsync(); + + // Three deployments for one cluster. The cache retains the newest two, so the first is pruned + // on A — and that prune must replicate to B as tombstones, not linger as live chunks. + var dep1 = Guid.NewGuid().ToString("N"); + var dep2 = Guid.NewGuid().ToString("N"); + var dep3 = Guid.NewGuid().ToString("N"); + + await harness.CacheA.StoreAsync(Cluster, dep1, "rev-1", MultiChunkArtifact(1), + TestContext.Current.CancellationToken); + await harness.CacheA.StoreAsync(Cluster, dep2, "rev-2", MultiChunkArtifact(2), + TestContext.Current.CancellationToken); + + // dep1's chunks are still present until the third store prunes them — confirm they reached B + // first, so the later disappearance is a real replicated prune rather than never-arrived. + await LocalDbPairHarness.WaitUntilAsync( + async () => await LocalDbPairHarness.ChunkCountAsync(harness.B, dep1) > 0, + "dep1's chunks to reach node B before the prune"); + + await harness.CacheA.StoreAsync(Cluster, dep3, "rev-3", MultiChunkArtifact(3), + TestContext.Current.CancellationToken); + + // A pruned dep1 locally; B must follow via replicated deletes: dep1 gone, tombstones present, + // and the surviving newest (dep3) intact. + await LocalDbPairHarness.WaitUntilAsync( + async () => + await LocalDbPairHarness.ChunkCountAsync(harness.B, dep1) == 0 + && await LocalDbPairHarness.TombstoneCountAsync(harness.B, DeploymentCacheSchema.ArtifactsTable) > 0 + && await LocalDbPairHarness.ChunkCountAsync(harness.B, dep3) > 0, + "node B to prune dep1 (with tombstones) while keeping dep3"); + } + + [Fact] + public async Task WritesWhileTransportDown_SurviveRejoin() + { + await using var harness = new LocalDbPairHarness(); + await harness.StartAsync(); + + var dep1 = Guid.NewGuid().ToString("N"); + var first = MultiChunkArtifact(seed: 10); + await harness.CacheA.StoreAsync(Cluster, dep1, "rev-1", first, + TestContext.Current.CancellationToken); + await LocalDbPairHarness.WaitUntilAsync( + async () => await ReadArtifactAsync(harness.CacheB, Cluster) is { } b && b.AsSpan().SequenceEqual(first), + "the first artifact to converge before the outage"); + + // B goes offline; A keeps deploying. B's database survives the host teardown (pre-constructed + // instance), so the rejoin is genuine rather than a fresh node. + await harness.StopPassiveAsync(); + + var dep2 = Guid.NewGuid().ToString("N"); + var second = MultiChunkArtifact(seed: 11); + await harness.CacheA.StoreAsync(Cluster, dep2, "rev-2", second, + TestContext.Current.CancellationToken); + + await harness.RestartPairAsync(); + + // Everything written during the outage catches up: B now serves the newer deployment. + await LocalDbPairHarness.WaitUntilAsync( + async () => + { + var cached = await harness.CacheB.GetCurrentAsync(Cluster, TestContext.Current.CancellationToken); + return cached is not null + && cached.DeploymentId == dep2 + && cached.Artifact.AsSpan().SequenceEqual(second); + }, + "node B to catch up on the deployment written while it was offline"); + } + + [Fact] + public async Task WrongApiKey_NeverConverges_WithMatchingKeyPositiveControl() + { + // Negative half: A and B hold different keys, so B's interceptor fail-closes on A's stream. + await using (var mismatched = new LocalDbPairHarness( + apiKeyA: LocalDbPairHarness.DefaultApiKey, apiKeyB: "a-different-key-entirely")) + { + await mismatched.StartAsync(); + + var deploymentId = Guid.NewGuid().ToString("N"); + await mismatched.CacheA.StoreAsync(Cluster, deploymentId, "rev-1", MultiChunkArtifact(20), + TestContext.Current.CancellationToken); + + var converged = await LocalDbPairHarness.HeldWithinAsync( + async () => await ReadArtifactAsync(mismatched.CacheB, Cluster) is not null, + TimeSpan.FromSeconds(5)); + + Assert.False(converged, + "a key mismatch must stop the pair converging — the interceptor is fail-closed"); + } + + // Positive control: the SAME scenario with matching keys DOES converge. Without this, the + // negative assertion above would pass even if convergence were broken for an unrelated reason. + await using var matched = new LocalDbPairHarness(); + await matched.StartAsync(); + + var controlId = Guid.NewGuid().ToString("N"); + var controlArtifact = MultiChunkArtifact(seed: 21); + await matched.CacheA.StoreAsync(Cluster, controlId, "rev-1", controlArtifact, + TestContext.Current.CancellationToken); + + await LocalDbPairHarness.WaitUntilAsync( + async () => await ReadArtifactAsync(matched.CacheB, Cluster) is { } b && b.AsSpan().SequenceEqual(controlArtifact), + "the matching-key control pair to converge"); + } +} diff --git a/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDb/LocalDbPairHarness.cs b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDb/LocalDbPairHarness.cs new file mode 100644 index 00000000..d2815590 --- /dev/null +++ b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDb/LocalDbPairHarness.cs @@ -0,0 +1,324 @@ +using System.Net; +using Microsoft.AspNetCore.Builder; +using Microsoft.AspNetCore.Hosting; +using Microsoft.AspNetCore.Hosting.Server; +using Microsoft.AspNetCore.Hosting.Server.Features; +using Microsoft.AspNetCore.Server.Kestrel.Core; +using Microsoft.Extensions.Configuration; +using Microsoft.Extensions.DependencyInjection; +using Microsoft.Extensions.Hosting; +using Microsoft.Extensions.Logging.Abstractions; +using Xunit; +using ZB.MOM.WW.LocalDb; +using ZB.MOM.WW.LocalDb.Replication; +using ZB.MOM.WW.OtOpcUa.Host.Configuration; +using ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache; + +namespace ZB.MOM.WW.OtOpcUa.Host.IntegrationTests.LocalDb; + +/// +/// Serializes the two-node convergence tests against one another: each stands up a real Kestrel +/// h2c listener plus two SQLite files, and running them concurrently under CI contention is a +/// flakiness risk. The rest of the assembly parallelizes normally. +/// +[CollectionDefinition("LocalDbPairConvergence")] +public sealed class LocalDbPairConvergenceCollection; + +/// +/// Two driver-role OtOpcUa nodes replicating the deployment-artifact cache over a REAL loopback +/// gRPC transport (Kestrel h2c on 127.0.0.1), through the REAL fail-closed +/// . Node A is the initiator (it dials the peer); node B +/// is passive (it hosts MapZbLocalDbSync). +/// +/// +/// +/// Both databases are initialised through the production — +/// a hand-written schema here would prove only that the test agrees with itself. The tables, +/// their primary keys, and the DDL→RegisterReplicated ordering under test are the ones +/// the host actually runs. +/// +/// +/// The two instances are owned by the harness and registered into the +/// hosts as pre-constructed singletons. MS.DI does not dispose instances it did not create, +/// so tearing a host down (the transport-loss scenario) leaves the databases intact and +/// writable — which is exactly what lets node A accumulate writes while B is down. +/// +/// +/// A real loopback socket (not an in-memory TestServer) is deliberate: killing the passive +/// host disposes the server and closes the socket, faulting the initiator's active stream +/// promptly. An in-memory TestServer does not model a connection drop and leaves the client +/// hanging. +/// +/// +/// The API keys are per-node so the wrong-key scenario can hand A and B different keys and +/// assert non-convergence — with a matching-key positive control proving the same harness +/// does converge when the keys agree (an absence assertion without a positive control passed +/// vacuously in ScadaBridge). +/// +/// Offline: no docker, no external services. +/// +public sealed class LocalDbPairHarness : IAsyncDisposable +{ + /// The key both nodes share unless a test overrides one of them. + public const string DefaultApiKey = "otopcua-localdb-pair-convergence-key"; + + /// How long a scenario waits for the pair to agree (or to stay diverged) before deciding. + public static readonly TimeSpan ConvergeTimeout = TimeSpan.FromSeconds(30); + + private readonly string _apiKeyA; + private readonly string _apiKeyB; + + private readonly string _pathA = Path.Combine(Path.GetTempPath(), $"otopcua-pairA-{Guid.NewGuid():N}.db"); + private readonly string _pathB = Path.Combine(Path.GetTempPath(), $"otopcua-pairB-{Guid.NewGuid():N}.db"); + + private readonly ServiceProvider _dbProviderA; + private readonly ServiceProvider _dbProviderB; + + private IHost? _serverHost; // node B — passive + private IHost? _initiatorHost; // node A — dials the peer + + static LocalDbPairHarness() => + // Grpc.Net.Client dials the loopback server over HTTP/2 cleartext (h2c). + AppContext.SetSwitch("System.Net.Http.SocketsHttpHandler.Http2UnencryptedSupport", true); + + /// Node A's replication key. Defaults to . + /// Node B's replication key. Defaults to . + public LocalDbPairHarness(string? apiKeyA = null, string? apiKeyB = null) + { + _apiKeyA = apiKeyA ?? DefaultApiKey; + _apiKeyB = apiKeyB ?? DefaultApiKey; + + _dbProviderA = BuildDatabaseProvider(_pathA); + _dbProviderB = BuildDatabaseProvider(_pathB); + } + + /// Node A — the initiator, which dials the peer. + public ILocalDb A => _dbProviderA.GetRequiredService(); + + /// Node B — the passive node, which listens. + public ILocalDb B => _dbProviderB.GetRequiredService(); + + /// The production artifact cache over node A's database. + public IDeploymentArtifactCache CacheA => + new LocalDbDeploymentArtifactCache(A, NullLogger.Instance); + + /// The production artifact cache over node B's database. + public IDeploymentArtifactCache CacheB => + new LocalDbDeploymentArtifactCache(B, NullLogger.Instance); + + // ---- lifecycle -------------------------------------------------------------------------- + + /// Forces both databases to construct (running OnReady) then brings the pair online. + public async Task StartAsync() + { + _ = A; + _ = B; + + await StartPassiveAsync(); + await StartInitiatorAsync(); + } + + /// Takes node B's listener down, leaving its database intact and writable. + public async Task StopPassiveAsync() + { + await StopHostAsync(_serverHost); + _serverHost = null; + } + + /// + /// Brings node B back on a NEW loopback port and re-dials from A. The initiator re-reads the + /// peer address on each reconnect, so this is a genuine rejoin over the same databases. + /// + public async Task RestartPairAsync() + { + await StartPassiveAsync(); + await StopHostAsync(_initiatorHost); + _initiatorHost = null; + await StartInitiatorAsync(); + } + + // ---- convergence helpers ---------------------------------------------------------------- + + /// Polls until true or the deadline passes; fails otherwise. + public static async Task WaitUntilAsync(Func> condition, string because) + { + var deadline = DateTime.UtcNow + ConvergeTimeout; + while (DateTime.UtcNow < deadline) + { + if (await condition()) + return; + await Task.Delay(50); + } + + Assert.Fail($"Timed out after {ConvergeTimeout.TotalSeconds:0}s waiting for: {because}"); + } + + /// Waits out a bounded window and returns whether ever held. + /// + /// For the negative half of the wrong-key scenario: it must NOT converge. A short, fixed + /// window keeps the test quick while still giving a matching-key control ample time to + /// converge (the control uses with the full timeout). + /// + public static async Task HeldWithinAsync(Func> condition, TimeSpan window) + { + var deadline = DateTime.UtcNow + window; + while (DateTime.UtcNow < deadline) + { + if (await condition()) + return true; + await Task.Delay(50); + } + + return false; + } + + /// + /// Dumps the __localdb_row_version rows for one table, ordered by pk, as + /// pk_json|hlc|node_id|is_tombstone. Equal dumps on both nodes prove B holds A's + /// origin-stamped row (same HLC + node id), not a locally re-derived one. + /// + public static async Task DumpRowVersionAsync(ILocalDb db, string table) + { + var rows = await db.QueryAsync( + """ + SELECT pk_json, hlc, node_id, is_tombstone + FROM __localdb_row_version + WHERE table_name = @Table + ORDER BY pk_json + """, + static r => $"{r.GetString(0)}|{r.GetInt64(1)}|{r.GetString(2)}|{r.GetInt64(3)}", + new { Table = table }); + return string.Join("\n", rows); + } + + /// Counts deployment_artifacts chunk rows for one deployment on a node. + public static async Task ChunkCountAsync(ILocalDb db, string deploymentId) + { + var rows = await db.QueryAsync( + "SELECT COUNT(*) FROM deployment_artifacts WHERE deployment_id = @DeploymentId", + static r => r.GetInt64(0), + new { DeploymentId = deploymentId }); + return rows[0]; + } + + /// Counts tombstone rows in __localdb_row_version for one table on a node. + public static async Task TombstoneCountAsync(ILocalDb db, string table) + { + var rows = await db.QueryAsync( + """ + SELECT COUNT(*) FROM __localdb_row_version + WHERE table_name = @Table AND is_tombstone = 1 + """, + static r => r.GetInt64(0), + new { Table = table }); + return rows[0]; + } + + // ---- fixture internals ------------------------------------------------------------------ + + /// + /// A provider owning one deployment-cache database, initialised through the host's own + /// — same schema, same registration order. + /// + private static ServiceProvider BuildDatabaseProvider(string path) + { + var config = new ConfigurationBuilder() + .AddInMemoryCollection(new Dictionary { ["LocalDb:Path"] = path }) + .Build(); + + return new ServiceCollection() + .AddZbLocalDb(config, LocalDbSetup.OnReady) + .BuildServiceProvider(); + } + + private IConfiguration ReplicationConfig(string apiKey, string? peerAddress) + { + var values = new Dictionary + { + // Tight flush + bounded reconnect backoff so convergence is observable well inside the + // poll deadline. The 60 s production default would let the doubling backoff overrun it + // after a peer outage. + ["LocalDb:Replication:FlushInterval"] = "00:00:00.050", + ["LocalDb:Replication:ReconnectBackoffMax"] = "00:00:02", + ["LocalDb:Replication:ApiKey"] = apiKey, + }; + + if (peerAddress is not null) + values["LocalDb:Replication:PeerAddress"] = peerAddress; + + return new ConfigurationBuilder().AddInMemoryCollection(values).Build(); + } + + /// Starts node B, the passive listener, behind the real auth interceptor. + private async Task StartPassiveAsync() + { + var config = ReplicationConfig(_apiKeyB, peerAddress: null); + + _serverHost = await new HostBuilder() + .ConfigureWebHost(web => + { + web.UseKestrel(o => + o.Listen(IPAddress.Loopback, 0, listen => listen.Protocols = HttpProtocols.Http2)); + web.ConfigureServices(services => + { + services.AddLogging(); + services.AddRouting(); + // The REAL interceptor, not a stand-in. If it rejected legitimate peer traffic, + // every matching-key scenario would fail — which is the point. + services.AddGrpc(o => o.Interceptors.Add()); + services.AddSingleton(B); + services.AddZbLocalDbReplication(config); + }) + .Configure(app => + { + app.UseRouting(); + app.UseEndpoints(e => e.MapZbLocalDbSync()); + }); + }) + .StartAsync(); + } + + /// Starts node A, which dials the passive node. + private async Task StartInitiatorAsync() + { + var config = ReplicationConfig(_apiKeyA, PassiveAddress()); + + _initiatorHost = await new HostBuilder() + .ConfigureServices(services => + { + services.AddLogging(); + services.AddSingleton(A); + services.AddZbLocalDbReplication(config); + }) + .StartAsync(); + } + + private string PassiveAddress() + => _serverHost!.Services.GetRequiredService() + .Features.Get()!.Addresses.Single(); + + private static async Task StopHostAsync(IHost? host) + { + if (host is null) + return; + try { await host.StopAsync(TimeSpan.FromSeconds(5)); } catch { /* teardown */ } + host.Dispose(); + } + + public async ValueTask DisposeAsync() + { + await StopHostAsync(_initiatorHost); + await StopHostAsync(_serverHost); + await _dbProviderA.DisposeAsync(); + await _dbProviderB.DisposeAsync(); + + Microsoft.Data.Sqlite.SqliteConnection.ClearAllPools(); + foreach (var path in new[] { _pathA, _pathB }) + { + foreach (var suffix in new[] { "", "-wal", "-shm" }) + { + try { File.Delete(path + suffix); } catch { /* best effort */ } + } + } + } +} diff --git a/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbReplicationHealthCheckTests.cs b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbReplicationHealthCheckTests.cs new file mode 100644 index 00000000..6674bf3e --- /dev/null +++ b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbReplicationHealthCheckTests.cs @@ -0,0 +1,122 @@ +using Microsoft.Extensions.Diagnostics.HealthChecks; +using Microsoft.Extensions.Options; +using Shouldly; +using Xunit; +using ZB.MOM.WW.LocalDb.Replication; +using ZB.MOM.WW.OtOpcUa.Host.Health; + +namespace ZB.MOM.WW.OtOpcUa.Host.IntegrationTests; + +/// +/// LocalDb Phase 1 (Task 10) — the replication health check. +/// +/// +/// The load-bearing case is the default-OFF one: a plain driver node with no peer and no +/// listener must report Healthy, or every unreplicated node in the fleet degrades the +/// moment this check ships. Beyond that, "connected" is not sufficient for healthy — an unknown +/// backlog (a failed oplog poll) must not read as zero. +/// +public sealed class LocalDbReplicationHealthCheckTests +{ + // ---- Pure decision core (table-tested) -------------------------------------------------- + + [Fact] + public void Unconfigured_IsHealthy_SoDefaultOffNodesDoNotDegrade() + { + LocalDbReplicationDecision.Evaluate( + peerConfigured: false, connected: false, oplogBacklog: null, degradedThreshold: 100_000) + .Status.ShouldBe(HealthStatus.Healthy); + } + + [Fact] + public void ConfiguredButNotConnected_IsDegraded() + { + LocalDbReplicationDecision.Evaluate( + peerConfigured: true, connected: false, oplogBacklog: null, degradedThreshold: 100_000) + .Status.ShouldBe(HealthStatus.Degraded); + } + + [Fact] + public void ConnectedButBacklogUnknown_IsDegraded() + { + // null backlog = the oplog poll failed. Reporting Healthy here would call a pair that cannot + // read its own oplog perfectly fine. + LocalDbReplicationDecision.Evaluate( + peerConfigured: true, connected: true, oplogBacklog: null, degradedThreshold: 100_000) + .Status.ShouldBe(HealthStatus.Degraded); + } + + [Fact] + public void ConnectedWithSmallBacklog_IsHealthy() + { + LocalDbReplicationDecision.Evaluate( + peerConfigured: true, connected: true, oplogBacklog: 12, degradedThreshold: 100_000) + .Status.ShouldBe(HealthStatus.Healthy); + } + + [Fact] + public void ConnectedWithBacklogOverThreshold_IsDegraded() + { + // A backlog past the threshold means the peer is falling behind faster than it drains — + // connected, but not keeping up. + LocalDbReplicationDecision.Evaluate( + peerConfigured: true, connected: true, oplogBacklog: 100_001, degradedThreshold: 100_000) + .Status.ShouldBe(HealthStatus.Degraded); + } + + // ---- IHealthCheck wrapper --------------------------------------------------------------- + + [Fact] + public async Task Wrapper_WithNoSyncStatusRegistered_IsHealthy() + { + // Admin-only graphs never register the replication engine. The check is registered + // unconditionally (AddOtOpcUaHealth takes no role argument), so it must no-op to Healthy + // rather than throw or degrade when LocalDb is simply absent. + var check = new LocalDbReplicationHealthCheck( + syncStatus: null, + options: Options.Create(new ReplicationOptions()), + syncListenPort: 0); + + var result = await check.CheckHealthAsync(new HealthCheckContext(), TestContext.Current.CancellationToken); + + result.Status.ShouldBe(HealthStatus.Healthy); + } + + [Fact] + public async Task Wrapper_ListenerConfiguredButNotConnected_IsDegraded() + { + // A passive node (no PeerAddress) still counts as "replication configured" when it is + // listening — otherwise a passive node that never gets dialled would look healthy while + // silently accepting nothing. + var check = new LocalDbReplicationHealthCheck( + syncStatus: new StubSyncStatus { Connected = false }, + options: Options.Create(new ReplicationOptions { PeerAddress = "" }), + syncListenPort: 9001); + + var result = await check.CheckHealthAsync(new HealthCheckContext(), TestContext.Current.CancellationToken); + + result.Status.ShouldBe(HealthStatus.Degraded); + } + + [Fact] + public async Task Wrapper_PeerConfiguredAndConnectedAndDraining_IsHealthy() + { + var check = new LocalDbReplicationHealthCheck( + syncStatus: new StubSyncStatus { Connected = true, OplogBacklog = 3 }, + options: Options.Create(new ReplicationOptions { PeerAddress = "https://peer:9001" }), + syncListenPort: 9001); + + var result = await check.CheckHealthAsync(new HealthCheckContext(), TestContext.Current.CancellationToken); + + result.Status.ShouldBe(HealthStatus.Healthy); + } + + private sealed class StubSyncStatus : ISyncStatus + { + public bool Connected { get; init; } + public string? PeerNodeId { get; init; } + public DateTimeOffset? LastSyncUtc { get; init; } + public long? OplogBacklog { get; init; } + public long ConnectionAttempts { get; init; } + } +} diff --git a/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbSetupTests.cs b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbSetupTests.cs new file mode 100644 index 00000000..2db014c2 --- /dev/null +++ b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbSetupTests.cs @@ -0,0 +1,128 @@ +using Microsoft.Data.Sqlite; +using Microsoft.Extensions.Configuration; +using Microsoft.Extensions.DependencyInjection; +using Shouldly; +using Xunit; +using ZB.MOM.WW.LocalDb; +using ZB.MOM.WW.OtOpcUa.Host.Configuration; + +namespace ZB.MOM.WW.OtOpcUa.Host.IntegrationTests; + +/// +/// LocalDb Phase 1 (Task 2) — pins the deployment-cache schema and the load-bearing +/// DDL → RegisterReplicated → writes ordering inside . +/// +/// +/// +/// The ordering assertion is the point of this file. RegisterReplicated installs the +/// capture triggers; rows written before it are never captured into the oplog and +/// therefore never reach the peer — silently, and forever. A reordered OnReady passes +/// every schema-shape test while shipping a replication path that quietly moves nothing. +/// +/// +/// Built against a real temp-file rather than a hand-written schema: +/// the library has no in-memory mode, and asserting against a schema the test itself wrote +/// would only prove the test agrees with itself. +/// +/// +public sealed class LocalDbSetupTests : IDisposable +{ + private readonly string _dbPath = + Path.Combine(Path.GetTempPath(), $"otopcua-localdb-setup-{Guid.NewGuid():N}.db"); + + private ServiceProvider? _provider; + + private ILocalDb BuildDb() + { + var configuration = new ConfigurationBuilder() + .AddInMemoryCollection(new Dictionary { ["LocalDb:Path"] = _dbPath }) + .Build(); + + _provider = new ServiceCollection() + .AddZbLocalDb(configuration, LocalDbSetup.OnReady) + .BuildServiceProvider(); + + return _provider.GetRequiredService(); + } + + [Fact] + public void OnReady_RegistersExactlyTheTwoDeploymentTables() + { + // Both directions are load-bearing. "No fewer" catches a dropped RegisterReplicated call + // (that table then never replicates). "No more" catches an accidental registration — + // every replicated table costs three triggers and oplog volume on every write. + var db = BuildDb(); + + db.ReplicatedTables.Keys.OrderBy(k => k, StringComparer.Ordinal) + .ShouldBe(["deployment_artifacts", "deployment_pointer"]); + } + + [Fact] + public void DeploymentArtifacts_PkIsDeploymentIdPlusChunkIndex() + { + // The composite PK is what makes an artifact chunkable. A single-column PK here would + // make every chunk of a deployment collide onto one row under LWW. + var db = BuildDb(); + + db.ReplicatedTables["deployment_artifacts"].PkColumns + .ShouldBe(["deployment_id", "chunk_index"]); + } + + [Fact] + public void DeploymentPointer_PkIsClusterId() + { + // One current-deployment pointer per cluster: the pair's two nodes converge on the same + // row rather than each keeping its own. + var db = BuildDb(); + + db.ReplicatedTables["deployment_pointer"].PkColumns.ShouldBe(["cluster_id"]); + } + + [Fact] + public async Task RowsWrittenAfterOnReady_EnterTheOplog() + { + // THE ordering assertion. If DDL and RegisterReplicated were swapped — or a write were + // added to OnReady ahead of registration — the triggers would not exist at write time and + // this count would stay 0 while every other test in this file still passed. + var db = BuildDb(); + + await db.ExecuteAsync( + """ + INSERT INTO deployment_pointer + (cluster_id, deployment_id, revision_hash, artifact_sha256, applied_at_utc) + VALUES (@ClusterId, @DeploymentId, @RevisionHash, @Sha, @AppliedAtUtc) + """, + new + { + ClusterId = "SITE-A", + DeploymentId = "0123456789abcdef0123456789abcdef", + RevisionHash = new string('a', 64), + Sha = new string('b', 64), + AppliedAtUtc = "2026-07-20T00:00:00.0000000Z", + }, + TestContext.Current.CancellationToken); + + var oplogRows = await db.QueryAsync( + "SELECT COUNT(*) FROM __localdb_oplog", + r => r.GetInt32(0), + parameters: null, + TestContext.Current.CancellationToken); + + oplogRows[0].ShouldBeGreaterThanOrEqualTo(1); + } + + public void Dispose() + { + _provider?.Dispose(); + + // No in-memory mode, so these are real files. Pooled connections keep a handle open past + // dispose; clearing the pools first is what makes the delete actually succeed. + SqliteConnection.ClearAllPools(); + + foreach (var path in new[] { _dbPath, $"{_dbPath}-wal", $"{_dbPath}-shm" }) + { + if (File.Exists(path)) + File.Delete(path); + } + } +} diff --git a/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbSyncAuthInterceptorTests.cs b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbSyncAuthInterceptorTests.cs new file mode 100644 index 00000000..9a506376 --- /dev/null +++ b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbSyncAuthInterceptorTests.cs @@ -0,0 +1,172 @@ +using Grpc.Core; +using Microsoft.Extensions.Logging.Abstractions; +using Microsoft.Extensions.Options; +using Shouldly; +using Xunit; +using ZB.MOM.WW.LocalDb.Replication; +using ZB.MOM.WW.OtOpcUa.Host.Configuration; + +namespace ZB.MOM.WW.OtOpcUa.Host.IntegrationTests; + +/// +/// LocalDb Phase 1 (Task 3) — the passive sync endpoint's inbound gate. +/// +/// +/// The replication library's LocalDbSyncService verifies nothing; inbound auth is +/// explicitly the host's job. Without this interceptor, anything able to reach a driver node's +/// sync port could stream arbitrary rows straight into that node's deployment-artifact cache — +/// which is exactly what the node boots from when central SQL is unreachable. +/// +public sealed class LocalDbSyncAuthInterceptorTests +{ + private const string SyncMethod = "/localdb_sync.v1.LocalDbSync/Sync"; + private const string OtherMethod = "/otopcua.SomethingElse/Call"; + + private static LocalDbSyncAuthInterceptor CreateInterceptor(string? apiKey) + => new( + Options.Create(new ReplicationOptions { ApiKey = apiKey }), + NullLogger.Instance); + + private static ServerCallContext CreateContext(string method, string? authorizationHeader) + { + var headers = new Metadata(); + if (authorizationHeader is not null) + headers.Add("authorization", authorizationHeader); + + return new FakeServerCallContext(method, headers); + } + + /// + /// Minimal carrying just a method name and request headers — + /// the only two things the interceptor reads. + /// + /// + /// Hand-rolled rather than using Grpc.Core.Testing.TestServerCallContext: that type + /// ships in the retired native Grpc.Core package and does not exist on the grpc-dotnet + /// stack this solution runs on. + /// + private sealed class FakeServerCallContext(string method, Metadata requestHeaders) + : ServerCallContext + { + protected override string MethodCore => method; + protected override string HostCore => "localhost"; + protected override string PeerCore => "ipv4:127.0.0.1:12345"; + protected override DateTime DeadlineCore => DateTime.UtcNow.AddMinutes(1); + protected override Metadata RequestHeadersCore => requestHeaders; + protected override CancellationToken CancellationTokenCore => CancellationToken.None; + protected override Metadata ResponseTrailersCore { get; } = []; + protected override Status StatusCore { get; set; } + protected override WriteOptions? WriteOptionsCore { get; set; } + + protected override AuthContext AuthContextCore { get; } = + new(null, new Dictionary>()); + + protected override ContextPropagationToken CreatePropagationTokenCore( + ContextPropagationOptions? options) + => throw new NotSupportedException(); + + protected override Task WriteResponseHeadersAsyncCore(Metadata responseHeaders) + => Task.CompletedTask; + } + + /// Invokes the interceptor's unary path with a trivial continuation. + private static Task Invoke( + LocalDbSyncAuthInterceptor interceptor, ServerCallContext context) + => interceptor.UnaryServerHandler( + "request", context, (_, _) => Task.FromResult("ok")); + + [Fact] + public async Task NonSyncMethod_PassesThrough_EvenWithNoKeyConfigured() + { + // The interceptor is registered on the whole AddGrpc pipeline, so it sees every call. + // It must be scoped strictly to the sync service. + var interceptor = CreateInterceptor(apiKey: null); + var context = CreateContext(OtherMethod, authorizationHeader: null); + + (await Invoke(interceptor, context)).ShouldBe("ok"); + } + + [Fact] + public async Task SyncMethod_WithNoKeyConfigured_IsDenied_EvenWithABearerToken() + { + // Fail-closed. "No key configured" is the DEFAULT every node ships with, so treating it + // as "no auth required" would silently expose the endpoint on exactly the configuration + // that is most common. Presenting a token must not help. + var interceptor = CreateInterceptor(apiKey: null); + var context = CreateContext(SyncMethod, "Bearer anything-at-all"); + + var ex = await Should.ThrowAsync(() => Invoke(interceptor, context)); + ex.StatusCode.ShouldBe(StatusCode.PermissionDenied); + } + + [Fact] + public async Task SyncMethod_WithNoBearerToken_IsDenied() + { + var interceptor = CreateInterceptor("the-shared-key"); + var context = CreateContext(SyncMethod, authorizationHeader: null); + + var ex = await Should.ThrowAsync(() => Invoke(interceptor, context)); + ex.StatusCode.ShouldBe(StatusCode.PermissionDenied); + } + + [Fact] + public async Task SyncMethod_WithWrongBearerToken_IsDenied() + { + var interceptor = CreateInterceptor("the-shared-key"); + var context = CreateContext(SyncMethod, "Bearer the-wrong-key"); + + var ex = await Should.ThrowAsync(() => Invoke(interceptor, context)); + ex.StatusCode.ShouldBe(StatusCode.PermissionDenied); + } + + [Fact] + public async Task SyncMethod_WithCorrectBearerToken_PassesThrough() + { + var interceptor = CreateInterceptor("the-shared-key"); + var context = CreateContext(SyncMethod, "Bearer the-shared-key"); + + (await Invoke(interceptor, context)).ShouldBe("ok"); + } + + [Fact] + public async Task SyncMethod_WithCorrectKey_ButNoBearerScheme_IsDenied() + { + // A raw key with no "Bearer " prefix is not what SyncBackgroundService sends, and + // accepting it would widen the accepted credential shape for no reason. + var interceptor = CreateInterceptor("the-shared-key"); + var context = CreateContext(SyncMethod, "the-shared-key"); + + var ex = await Should.ThrowAsync(() => Invoke(interceptor, context)); + ex.StatusCode.ShouldBe(StatusCode.PermissionDenied); + } + + [Fact] + public async Task SyncMethod_TokenComparison_IsNotAPrefixMatch() + { + // A prefix/StartsWith comparison would accept a truncated key and make the secret + // recoverable one character at a time. + var interceptor = CreateInterceptor("the-shared-key"); + var context = CreateContext(SyncMethod, "Bearer the-shared-ke"); + + var ex = await Should.ThrowAsync(() => Invoke(interceptor, context)); + ex.StatusCode.ShouldBe(StatusCode.PermissionDenied); + } + + [Fact] + public async Task DuplexStreaming_IsGated_BecauseThatIsHowSyncActuallyRuns() + { + // The sync RPC is a bidirectional stream. Gating only the unary path would leave the real + // endpoint wide open while every unary test still passed. + var interceptor = CreateInterceptor("the-shared-key"); + var context = CreateContext(SyncMethod, "Bearer the-wrong-key"); + + var ex = await Should.ThrowAsync(() => + interceptor.DuplexStreamingServerHandler( + requestStream: null!, + responseStream: null!, + context, + (_, _, _) => Task.CompletedTask)); + + ex.StatusCode.ShouldBe(StatusCode.PermissionDenied); + } +} diff --git a/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbSyncListenerTests.cs b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbSyncListenerTests.cs new file mode 100644 index 00000000..58e25082 --- /dev/null +++ b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbSyncListenerTests.cs @@ -0,0 +1,234 @@ +using System.Net; +using Microsoft.AspNetCore.Builder; +using Microsoft.AspNetCore.Server.Kestrel.Core; +using Microsoft.Extensions.DependencyInjection; +using Microsoft.Extensions.Hosting; +using Shouldly; +using Xunit; +using ZB.MOM.WW.OtOpcUa.Host.Configuration; + +namespace ZB.MOM.WW.OtOpcUa.Host.IntegrationTests; + +/// +/// LocalDb Phase 1 (Task 5) — the dedicated h2c sync listener and, more importantly, the +/// re-binding of the endpoints the host was already serving. +/// +/// +/// +/// Why this file exists. The host binds exclusively via ASPNETCORE_URLS and has +/// no ConfigureKestrel call. Any explicit Listen* makes Kestrel discard that +/// configuration wholesale — it logs "Overriding address(es)" and serves only what was +/// listed explicitly. Getting this wrong unbinds the AdminUI and the deploy API behind +/// Traefik, and it fails silently: the process starts, logs look normal, and nothing answers. +/// +/// +/// These tests drive a minimal WebApplication shaped exactly like Program.cs's +/// block rather than booting the real host, which needs SQL Server, an Akka mesh and LDAP. +/// What is under test is the Kestrel binding contract, and that is fully reproduced here. +/// +/// +public sealed class LocalDbSyncListenerTests +{ + // --------------------------------------------------------------------------------------- + // KestrelHttpBinding.Parse + // --------------------------------------------------------------------------------------- + + [Fact] + public void Parse_NullOrEmpty_YieldsNoBindings() + { + // "Nothing configured" is a real supported state: Install-Services.ps1 sets ASPNETCORE_URLS + // only for admin nodes, so a driver-only service genuinely has none. + KestrelHttpBinding.Parse(null).ShouldBeEmpty(); + KestrelHttpBinding.Parse("").ShouldBeEmpty(); + KestrelHttpBinding.Parse(" ").ShouldBeEmpty(); + } + + [Theory] + [InlineData("http://+:9000", "+", 9000)] + [InlineData("http://*:9000", "*", 9000)] + [InlineData("http://localhost:9000", "localhost", 9000)] + [InlineData("http://0.0.0.0:8080", "0.0.0.0", 8080)] + [InlineData("http://127.0.0.1:5001", "127.0.0.1", 5001)] + public void Parse_SingleUrl_ExtractsHostAndPort(string url, string expectedHost, int expectedPort) + { + // The "+" and "*" wildcards are Kestrel-isms that System.Uri cannot parse — the docker-dev + // rig uses "http://+:9000", so mishandling them would break every rig node. + var bindings = KestrelHttpBinding.Parse(url); + + bindings.Count.ShouldBe(1); + bindings[0].Host.ShouldBe(expectedHost); + bindings[0].Port.ShouldBe(expectedPort); + bindings[0].IsSecure.ShouldBeFalse(); + } + + [Fact] + public void Parse_MultipleUrls_PreservesAllOfThem() + { + // Dropping one of several configured endpoints is the exact silent-unbind failure this + // whole mechanism exists to avoid. + var bindings = KestrelHttpBinding.Parse("http://+:9000;http://localhost:9100"); + + bindings.Count.ShouldBe(2); + bindings[0].Port.ShouldBe(9000); + bindings[1].Port.ShouldBe(9100); + } + + [Fact] + public void Parse_HttpsUrl_IsFlaggedSecure() + { + // Program.cs refuses to take over Kestrel when any endpoint is HTTPS, because replaying + // certificate configuration is not modelled here. This flag is what drives that refusal. + KestrelHttpBinding.Parse("https://+:443")[0].IsSecure.ShouldBeTrue(); + } + + [Fact] + public void Parse_MalformedEntry_IsSkipped_NotThrown() + { + // A typo'd URL must not take the process down at startup. + KestrelHttpBinding.Parse("not a url;http://+:9000").ShouldHaveSingleItem().Port.ShouldBe(9000); + } + + // ---- ParsePorts (ASPNETCORE_HTTP_PORTS / HTTPS_PORTS) ---------------------------------- + + [Fact] + public void ParsePorts_NullOrEmpty_YieldsNoBindings() + { + KestrelHttpBinding.ParsePorts(null, isSecure: false).ShouldBeEmpty(); + KestrelHttpBinding.ParsePorts("", isSecure: false).ShouldBeEmpty(); + KestrelHttpBinding.ParsePorts(" ", isSecure: false).ShouldBeEmpty(); + } + + [Fact] + public void ParsePorts_BarePort_BindsAllInterfaces() + { + // The aspnet:8.0+ container default is ASPNETCORE_HTTP_PORTS=8080, on all interfaces — not + // loopback. Re-binding it as a wildcard is what keeps a driver node's health/metrics surface + // reachable from other containers once the sync listener takes over Kestrel. + var binding = KestrelHttpBinding.ParsePorts("8080", isSecure: false).ShouldHaveSingleItem(); + binding.Host.ShouldBe("+"); + binding.Port.ShouldBe(8080); + binding.IsSecure.ShouldBeFalse(); + } + + [Fact] + public void ParsePorts_SeparatedList_AndSecureFlag() + { + KestrelHttpBinding.ParsePorts("8080;8081", isSecure: false).Select(b => b.Port).ShouldBe([8080, 8081]); + KestrelHttpBinding.ParsePorts("8080,8081", isSecure: false).Select(b => b.Port).ShouldBe([8080, 8081]); + KestrelHttpBinding.ParsePorts("443", isSecure: true).ShouldHaveSingleItem().IsSecure.ShouldBeTrue(); + } + + [Fact] + public void ParsePorts_MalformedPort_IsSkipped_NotThrown() + { + KestrelHttpBinding.ParsePorts("notaport;8080;0;70000", isSecure: false) + .ShouldHaveSingleItem().Port.ShouldBe(8080); + } + + // --------------------------------------------------------------------------------------- + // The real Kestrel contract + // --------------------------------------------------------------------------------------- + + [Fact] + public async Task ExplicitListen_ReBindsTheExistingSurface_AndAddsAnH2cListener() + { + // THE test. Both ports must answer simultaneously: the re-bound HTTP/1.1 surface (proving + // the ASPNETCORE_URLS override was compensated for) and the HTTP/2-only sync port (proving + // prior-knowledge h2c works, which a cleartext Http1AndHttp2 endpoint cannot do). + var httpPort = GetFreePort(); + var syncPort = GetFreePort(); + + var builder = WebApplication.CreateBuilder(); + builder.Environment.ApplicationName = typeof(LocalDbSyncListenerTests).Assembly.GetName().Name!; + builder.Services.AddGrpc(); + + // Exactly the shape Program.cs applies. + foreach (var binding in KestrelHttpBinding.Parse($"http://localhost:{httpPort}")) + { + var captured = binding; + builder.WebHost.ConfigureKestrel(k => captured.Apply(k)); + } + + builder.WebHost.ConfigureKestrel(k => + k.ListenAnyIP(syncPort, o => o.Protocols = HttpProtocols.Http2)); + + await using var app = builder.Build(); + app.MapGet("/healthz", () => "ok"); + await app.StartAsync(TestContext.Current.CancellationToken); + + try + { + using var http1 = new HttpClient(); + var body = await http1.GetStringAsync( + $"http://localhost:{httpPort}/healthz", TestContext.Current.CancellationToken); + body.ShouldBe("ok"); + + // Prior-knowledge h2c against the sync port. A 404 is a perfectly good result — it + // proves the HTTP/2 connection was established and the request was routed, which is + // the only thing in question here. + using var http2 = new HttpClient + { + DefaultRequestVersion = HttpVersion.Version20, + DefaultVersionPolicy = HttpVersionPolicy.RequestVersionExact, + }; + using var response = await http2.GetAsync( + $"http://localhost:{syncPort}/", TestContext.Current.CancellationToken); + + response.Version.ShouldBe(HttpVersion.Version20); + } + finally + { + await app.StopAsync(TestContext.Current.CancellationToken); + } + } + + [Fact] + public async Task ExplicitListen_WithoutReBinding_SilentlyDiscardsTheConfiguredUrls() + { + // POSITIVE CONTROL for the whole task. If Kestrel did NOT override configured URLs, the + // re-binding above would be pointless ceremony. This pins the actual behaviour: a single + // explicit Listen* call makes the ASPNETCORE_URLS port stop answering entirely — no + // exception, no failed startup, just silence. That is the regression being guarded against. + var configuredPort = GetFreePort(); + var explicitPort = GetFreePort(); + + var builder = WebApplication.CreateBuilder(); + builder.Environment.ApplicationName = typeof(LocalDbSyncListenerTests).Assembly.GetName().Name!; + builder.WebHost.UseUrls($"http://localhost:{configuredPort}"); + + // Deliberately NOT re-binding configuredPort — this is the mistake being demonstrated. + builder.WebHost.ConfigureKestrel(k => k.ListenAnyIP(explicitPort)); + + await using var app = builder.Build(); + app.MapGet("/healthz", () => "ok"); + await app.StartAsync(TestContext.Current.CancellationToken); + + try + { + using var client = new HttpClient { Timeout = TimeSpan.FromSeconds(5) }; + + // The explicitly listed port serves. + (await client.GetStringAsync( + $"http://localhost:{explicitPort}/healthz", + TestContext.Current.CancellationToken)).ShouldBe("ok"); + + // The configured one does not — nothing is listening there at all. + await Should.ThrowAsync(() => client.GetStringAsync( + $"http://localhost:{configuredPort}/healthz", + TestContext.Current.CancellationToken)); + } + finally + { + await app.StopAsync(TestContext.Current.CancellationToken); + } + } + + private static int GetFreePort() + { + using var listener = new System.Net.Sockets.TcpListener(IPAddress.Loopback, 0); + listener.Start(); + var port = ((IPEndPoint)listener.LocalEndpoint).Port; + listener.Stop(); + return port; + } +} diff --git a/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbWiringTests.cs b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbWiringTests.cs new file mode 100644 index 00000000..77060548 --- /dev/null +++ b/tests/Server/ZB.MOM.WW.OtOpcUa.Host.IntegrationTests/LocalDbWiringTests.cs @@ -0,0 +1,172 @@ +using Microsoft.AspNetCore.Builder; +using Microsoft.Data.Sqlite; +using Microsoft.Extensions.Configuration; +using Microsoft.Extensions.DependencyInjection; +using Microsoft.Extensions.Diagnostics.HealthChecks; +using Microsoft.Extensions.Options; +using Shouldly; +using Xunit; +using ZB.MOM.WW.LocalDb; +using ZB.MOM.WW.LocalDb.Replication; +using ZB.MOM.WW.OtOpcUa.Host.Configuration; +using ZB.MOM.WW.OtOpcUa.Host.Health; +using ZB.MOM.WW.OtOpcUa.Host.Observability; +using ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache; + +namespace ZB.MOM.WW.OtOpcUa.Host.IntegrationTests; + +/// +/// LocalDb Phase 1 (Task 11) — DI pins over the real composition methods. +/// +/// +/// +/// DI extension methods have shipped inert three times in this family (Secrets 0.2.0 / 0.2.2, +/// ScadaBridge #22): a registration that compiles, deploys, and does nothing because the seam +/// was never resolved from a built container. These tests build a real container from the +/// same AddOtOpcUa* methods Program.cs calls and resolve through it. +/// +/// +/// They deliberately do NOT boot the full host. WebApplicationFactory<Program> +/// would start hosted services — Akka, the OPC UA server, ValidateOnStart — which need +/// SQL Server and an Akka mesh (see TwoNodeClusterHarness on why the real host is not +/// driven here). What is under test is DI resolution, so the container is built and resolved +/// without ever starting it. +/// +/// +public sealed class LocalDbWiringTests : IDisposable +{ + private readonly List _dbPaths = []; + private readonly List _apps = []; + + /// + /// Builds a container the way Program.cs's driver branch does for LocalDb: the real + /// , health, and observability methods, + /// over a temp-file LocalDb:Path. + /// + private IServiceProvider BuildDriverGraph() + { + var dbPath = Path.Combine(Path.GetTempPath(), $"otopcua-wiring-{Guid.NewGuid():N}.db"); + _dbPaths.Add(dbPath); + + var builder = WebApplication.CreateBuilder(new WebApplicationOptions { Args = [] }); + builder.Configuration.AddInMemoryCollection(new Dictionary + { + ["LocalDb:Path"] = dbPath, + }); + + // The exact LocalDb composition Program.cs runs under hasDriver. + builder.Services.AddOtOpcUaLocalDb(builder.Configuration); + builder.Services.AddGrpc(o => o.Interceptors.Add()); + builder.Services.AddOtOpcUaHealth(); + builder.Services.AddOtOpcUaObservability(builder.Configuration); + + var app = builder.Build(); + _apps.Add(app); + return app.Services; + } + + /// + /// Builds an admin-only container: the shared registrations that run on every node, but NOT + /// — exactly as Program.cs gates it under + /// hasDriver. + /// + private IServiceProvider BuildAdminOnlyGraph() + { + var builder = WebApplication.CreateBuilder(new WebApplicationOptions { Args = [] }); + + // No LocalDb:Path configured — an admin-only node has no reason to set it. + builder.Services.AddOtOpcUaHealth(); + builder.Services.AddOtOpcUaObservability(builder.Configuration); + + var app = builder.Build(); + _apps.Add(app); + return app.Services; + } + + [Fact] + public void DriverGraph_LocalDbResolvesAsASingletonWithBothCacheTablesRegistered() + { + var sp = BuildDriverGraph(); + + var first = sp.GetService(); + first.ShouldNotBeNull(); + sp.GetService().ShouldBeSameAs(first); // singleton + + // Exact set, both directions — a dropped registration replicates nothing; an extra one costs + // three triggers and oplog volume on every write. + first.ReplicatedTables.Keys.OrderBy(k => k, StringComparer.Ordinal) + .ShouldBe(["deployment_artifacts", "deployment_pointer"]); + } + + [Fact] + public void DriverGraph_ArtifactCacheResolvesToTheLocalDbBackedImplementation() + { + var sp = BuildDriverGraph(); + + sp.GetService() + .ShouldBeOfType(); + } + + [Fact] + public void DriverGraph_DefaultOff_SyncStatusReportsNotConnectedAndNoPeer() + { + // The default-OFF posture pin: registering the replication engine must not connect it. + var sp = BuildDriverGraph(); + + var status = sp.GetService(); + status.ShouldNotBeNull(); + status.Connected.ShouldBeFalse(); + status.PeerNodeId.ShouldBeNull(); + } + + [Fact] + public void DriverGraph_LocalDbReplicationHealthCheckIsRegistered() + { + var sp = BuildDriverGraph(); + + var registrations = sp.GetRequiredService>() + .Value.Registrations + .Select(r => r.Name) + .ToArray(); + + registrations.ShouldContain("localdb-replication"); + } + + [Fact] + public void AdminOnlyGraph_HasNoLocalDb_AndDoesNotDemandLocalDbPath() + { + // Boot must not require LocalDb:Path on a node that never registers LocalDb. Building the + // container without a configured path and resolving the shared services proves nothing in + // the always-on registrations reaches for ILocalDb or its options. + var sp = BuildAdminOnlyGraph(); + + sp.GetService().ShouldBeNull(); + sp.GetService().ShouldBeNull(); + + // The unconditionally-registered health check must still resolve and construct — it is meant + // to no-op to Healthy when LocalDb is absent, not to fail because ISyncStatus is missing. + var registrations = sp.GetRequiredService>() + .Value.Registrations + .Where(r => r.Name == "localdb-replication") + .ToArray(); + var localDbCheck = registrations.ShouldHaveSingleItem(); + Should.NotThrow(() => localDbCheck.Factory(sp)); + } + + public void Dispose() + { + foreach (var app in _apps) + ((IDisposable)app).Dispose(); + + SqliteConnection.ClearAllPools(); + + foreach (var basePath in _dbPaths) + { + foreach (var path in new[] { basePath, $"{basePath}-wal", $"{basePath}-shm" }) + { + if (File.Exists(path)) + File.Delete(path); + } + } + } +} diff --git a/tests/Server/ZB.MOM.WW.OtOpcUa.Runtime.Tests/Deployment/LocalDbDeploymentArtifactCacheTests.cs b/tests/Server/ZB.MOM.WW.OtOpcUa.Runtime.Tests/Deployment/LocalDbDeploymentArtifactCacheTests.cs new file mode 100644 index 00000000..86bb8154 --- /dev/null +++ b/tests/Server/ZB.MOM.WW.OtOpcUa.Runtime.Tests/Deployment/LocalDbDeploymentArtifactCacheTests.cs @@ -0,0 +1,291 @@ +using Microsoft.Data.Sqlite; +using Microsoft.Extensions.Configuration; +using Microsoft.Extensions.DependencyInjection; +using Microsoft.Extensions.Logging; +using Microsoft.Extensions.Logging.Abstractions; +using Shouldly; +using Xunit; +using ZB.MOM.WW.LocalDb; +using ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache; + +// NOT "...Tests.Deployment", even though the folder is Deployment/. A child namespace named +// Deployment under ...Runtime.Tests shadows the Deployment EF entity type for every other file in +// this assembly — the same CS0118 trap that forced the production namespace's rename. +namespace ZB.MOM.WW.OtOpcUa.Runtime.Tests.DeploymentCache; + +/// +/// LocalDb Phase 1 (Task 6) — the chunked deployment-artifact cache. +/// +/// +/// +/// Driven against a real temp-file rather than a fake. The behaviours +/// under test — chunk reassembly order, SHA-256 integrity, retention pruning, upsert +/// semantics — live entirely in SQL, so a mocked seam would assert nothing about them. +/// +/// +/// This project cannot reference the Host, so the schema setup here mirrors +/// LocalDbSetup.OnReady rather than calling it. The load-bearing +/// DDL → RegisterReplicated ordering is pinned by the Host's own +/// LocalDbSetupTests; what matters here is only that the tables exist. +/// +/// +public sealed class LocalDbDeploymentArtifactCacheTests : IDisposable +{ + private const string ClusterA = "SITE-A"; + private const string ClusterB = "SITE-B"; + + private readonly string _dbPath = + Path.Combine(Path.GetTempPath(), $"otopcua-artifact-cache-{Guid.NewGuid():N}.db"); + + private readonly ServiceProvider _provider; + private readonly ILocalDb _db; + + public LocalDbDeploymentArtifactCacheTests() + { + var configuration = new ConfigurationBuilder() + .AddInMemoryCollection(new Dictionary { ["LocalDb:Path"] = _dbPath }) + .Build(); + + _provider = new ServiceCollection() + .AddZbLocalDb(configuration, db => + { + using (var connection = db.CreateConnection()) + { + DeploymentCacheSchema.Apply(connection); + } + + db.RegisterReplicated(DeploymentCacheSchema.ArtifactsTable); + db.RegisterReplicated(DeploymentCacheSchema.PointerTable); + }) + .BuildServiceProvider(); + + _db = _provider.GetRequiredService(); + } + + private LocalDbDeploymentArtifactCache CreateCache( + ILogger? logger = null) + => new(_db, logger ?? NullLogger.Instance); + + private static byte[] RandomArtifact(int length) + { + var bytes = new byte[length]; + Random.Shared.NextBytes(bytes); + return bytes; + } + + private async Task CountChunksAsync(string? deploymentId = null) + { + var sql = deploymentId is null + ? "SELECT COUNT(*) FROM deployment_artifacts" + : "SELECT COUNT(*) FROM deployment_artifacts WHERE deployment_id = @DeploymentId"; + + var rows = await _db.QueryAsync(sql, r => r.GetInt32(0), + deploymentId is null ? null : new { DeploymentId = deploymentId }); + + return rows[0]; + } + + private async Task> DistinctDeploymentIdsAsync() + => await _db.QueryAsync( + "SELECT DISTINCT deployment_id FROM deployment_artifacts ORDER BY deployment_id", + r => r.GetString(0)); + + [Fact] + public async Task StoreThenGetCurrent_RoundTripsAMultiChunkArtifactByteForByte() + { + // 300 KiB against a 128 KiB chunk forces three chunks, so this covers the reassembly + // ordering that a single-chunk artifact would never exercise. + var artifact = RandomArtifact(300 * 1024); + var cache = CreateCache(); + + await cache.StoreAsync(ClusterA, "dep-1", "rev-1", artifact); + + var cached = await cache.GetCurrentAsync(ClusterA); + + cached.ShouldNotBeNull(); + cached.DeploymentId.ShouldBe("dep-1"); + cached.RevisionHash.ShouldBe("rev-1"); + cached.Artifact.ShouldBe(artifact); + (await CountChunksAsync("dep-1")).ShouldBe(3); + } + + [Fact] + public async Task Store_KeepsOnlyTheTwoNewestDeploymentsPerCluster() + { + var cache = CreateCache(); + + await cache.StoreAsync(ClusterA, "dep-1", "rev-1", RandomArtifact(64)); + await cache.StoreAsync(ClusterA, "dep-2", "rev-2", RandomArtifact(64)); + await cache.StoreAsync(ClusterA, "dep-3", "rev-3", RandomArtifact(64)); + + // Bounded retention is what stops a node that redeploys daily from filling its disk with + // address spaces nobody will ever roll back to. + (await DistinctDeploymentIdsAsync()).ShouldBe(["dep-2", "dep-3"]); + + var cached = await cache.GetCurrentAsync(ClusterA); + cached.ShouldNotBeNull(); + cached.DeploymentId.ShouldBe("dep-3"); + } + + [Fact] + public async Task Store_NeverPrunesTheDeploymentThePointerNames() + { + // Regression: the prune ranks survivors by cached_at_utc and falls through to a + // deployment_id tiebreak. Real deployment ids are GUIDs, so that tiebreak is effectively + // random — the deployment just written could lose and have its chunks deleted while the + // pointer still named it. The result is a cache that misses on every boot until the next + // deploy: silent, permanent, and precisely the outage this cache exists to survive. + // + // Rather than trying to race three stores into one clock tick, this forces the general + // case: the two older deployments are back-dated INTO THE FUTURE so the newest store loses + // the ranking outright. The invariant must hold regardless of timestamps. + var cache = CreateCache(); + var newest = RandomArtifact(1024); + + await cache.StoreAsync(ClusterA, "dep-old-1", "rev-1", RandomArtifact(1024)); + await cache.StoreAsync(ClusterA, "dep-old-2", "rev-2", RandomArtifact(1024)); + + await _db.ExecuteAsync( + "UPDATE deployment_artifacts SET cached_at_utc = @Future", + new { Future = "2099-01-01T00:00:00.0000000Z" }); + + await cache.StoreAsync(ClusterA, "dep-new", "rev-3", newest); + + // The pointer names dep-new, so dep-new's chunks must still be there... + (await CountChunksAsync("dep-new")).ShouldBeGreaterThan(0); + + // ...and it must actually read back, which is the behaviour that matters. + var cached = await cache.GetCurrentAsync(ClusterA); + cached.ShouldNotBeNull(); + cached.DeploymentId.ShouldBe("dep-new"); + cached.Artifact.ShouldBe(newest); + } + + [Fact] + public async Task GetCurrent_ReturnsNullWhenAChunkIsCorrupt() + { + var cache = CreateCache(); + await cache.StoreAsync(ClusterA, "dep-1", "rev-1", RandomArtifact(300 * 1024)); + + // Valid base64 of the wrong bytes: this passes decoding and the chunk-count check, so only + // the SHA-256 comparison can catch it. That is the check being pinned. + await _db.ExecuteAsync( + "UPDATE deployment_artifacts SET chunk_base64 = @Chunk WHERE deployment_id = @DeploymentId AND chunk_index = 1", + new { Chunk = Convert.ToBase64String(RandomArtifact(128 * 1024)), DeploymentId = "dep-1" }); + + (await cache.GetCurrentAsync(ClusterA)).ShouldBeNull(); + } + + [Fact] + public async Task GetCurrent_ReturnsNullWhenAChunkIsMissing() + { + var cache = CreateCache(); + await cache.StoreAsync(ClusterA, "dep-1", "rev-1", RandomArtifact(300 * 1024)); + + // A partially replicated artifact is the realistic version of this: the pointer row arrives + // before the last chunk does. Reassembling what is present would yield a plausible-looking + // but silently truncated address space. + await _db.ExecuteAsync( + "DELETE FROM deployment_artifacts WHERE deployment_id = @DeploymentId AND chunk_index = 2", + new { DeploymentId = "dep-1" }); + + (await cache.GetCurrentAsync(ClusterA)).ShouldBeNull(); + } + + [Fact] + public async Task GetCurrent_ReturnsNullWhenNoPointerExists() + { + var cache = CreateCache(); + + (await cache.GetCurrentAsync(ClusterA)).ShouldBeNull(); + } + + [Fact] + public async Task Store_IsIdempotentForTheSameDeploymentId() + { + var cache = CreateCache(); + + await cache.StoreAsync(ClusterA, "dep-1", "rev-1", RandomArtifact(300 * 1024)); + + // The re-store is smaller, so a delete-before-insert that did not happen would leave the + // tail chunks of the first artifact behind — orphans that also break the chunk-count check. + var second = RandomArtifact(200 * 1024); + await cache.StoreAsync(ClusterA, "dep-1", "rev-1b", second); + + (await CountChunksAsync("dep-1")).ShouldBe(2); + (await CountChunksAsync()).ShouldBe(2); + + var cached = await cache.GetCurrentAsync(ClusterA); + cached.ShouldNotBeNull(); + cached.RevisionHash.ShouldBe("rev-1b"); + cached.Artifact.ShouldBe(second); + } + + [Fact] + public async Task GetCurrentUnkeyed_ReturnsTheOnlyCachedArtifact() + { + var artifact = RandomArtifact(300 * 1024); + var cache = CreateCache(); + + await cache.StoreAsync(ClusterA, "dep-1", "rev-1", artifact); + + var cached = await cache.GetCurrentUnkeyedAsync(); + + cached.ShouldNotBeNull(); + cached.DeploymentId.ShouldBe("dep-1"); + cached.Artifact.ShouldBe(artifact); + } + + [Fact] + public async Task GetCurrentUnkeyed_TakesTheNewestPointerAndWarnsNamingBothClusters() + { + var newest = RandomArtifact(1024); + var logger = new CapturingLogger(); + var cache = CreateCache(logger); + + await cache.StoreAsync(ClusterA, "dep-a", "rev-a", RandomArtifact(1024)); + await cache.StoreAsync(ClusterB, "dep-b", "rev-b", newest); + + var cached = await cache.GetCurrentUnkeyedAsync(); + + cached.ShouldNotBeNull(); + cached.DeploymentId.ShouldBe("dep-b"); + cached.Artifact.ShouldBe(newest); + + // A re-homed node booting a neighbouring cluster's configuration must be operator-visible. + // Naming both clusters is what turns "wrong tags appeared" into a one-line diagnosis. + var warning = logger.Entries.ShouldHaveSingleItem(); + warning.Level.ShouldBe(LogLevel.Warning); + warning.Message.ShouldContain(ClusterA); + warning.Message.ShouldContain(ClusterB); + } + + public void Dispose() + { + _provider.Dispose(); + + // Real files, and pooled connections outlive the provider — clearing the pools is what + // makes the delete actually succeed. + SqliteConnection.ClearAllPools(); + + foreach (var path in new[] { _dbPath, $"{_dbPath}-wal", $"{_dbPath}-shm" }) + { + if (File.Exists(path)) + File.Delete(path); + } + } + + /// Minimal logger that records formatted messages so a test can assert on them. + private sealed class CapturingLogger : ILogger + { + public List<(LogLevel Level, string Message)> Entries { get; } = []; + + public IDisposable? BeginScope(TState state) where TState : notnull => null; + + public bool IsEnabled(LogLevel logLevel) => true; + + public void Log(LogLevel logLevel, EventId eventId, TState state, Exception? exception, + Func formatter) + => Entries.Add((logLevel, formatter(state, exception))); + } +} diff --git a/tests/Server/ZB.MOM.WW.OtOpcUa.Runtime.Tests/Drivers/DriverHostActorArtifactCacheTests.cs b/tests/Server/ZB.MOM.WW.OtOpcUa.Runtime.Tests/Drivers/DriverHostActorArtifactCacheTests.cs new file mode 100644 index 00000000..77f9972b --- /dev/null +++ b/tests/Server/ZB.MOM.WW.OtOpcUa.Runtime.Tests/Drivers/DriverHostActorArtifactCacheTests.cs @@ -0,0 +1,273 @@ +using System.Text.Json; +using Akka.Actor; +using Microsoft.EntityFrameworkCore; +using Shouldly; +using Xunit; +using ZB.MOM.WW.OtOpcUa.Commons.Messages.Deploy; +using ZB.MOM.WW.OtOpcUa.Commons.Types; +using ZB.MOM.WW.OtOpcUa.Configuration; +using ZB.MOM.WW.OtOpcUa.Configuration.Entities; +using ZB.MOM.WW.OtOpcUa.Configuration.Enums; +using ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache; +using ZB.MOM.WW.OtOpcUa.Runtime.Drivers; +using ZB.MOM.WW.OtOpcUa.Runtime.Tests.Harness; + +namespace ZB.MOM.WW.OtOpcUa.Runtime.Tests.Drivers; + +/// +/// LocalDb Phase 1 (Task 7) — the write half of the deployment-artifact cache. +/// +/// +/// What matters here is not that a store happens, but the conditions under which it must NOT: +/// a failed apply, and — less obviously — a "successful" apply whose artifact never actually +/// loaded. See ReconcileDrivers: it swallows its own DB failures, so apply success does +/// not imply a real configuration was applied. +/// +public sealed class DriverHostActorArtifactCacheTests : RuntimeActorTestBase +{ + private static readonly NodeId TestNode = NodeId.Parse("driver-test"); + private static readonly RevisionHash RevA = RevisionHash.Parse(new string('a', 64)); + + [Fact] + public void SuccessfulApply_StoresTheArtifactInTheCache() + { + var db = NewInMemoryDbFactory(); + var cache = new RecordingArtifactCache(); + var (deploymentId, artifact) = SeedDeployment(db, RevA, clusterId: null); + + var coordinator = CreateTestProbe(); + var actor = Sys.ActorOf(DriverHostActor.Props( + db, TestNode, coordinator.Ref, + localRoles: new HashSet { "driver" }, + deploymentArtifactCache: cache)); + + actor.Tell(new DispatchDeployment(deploymentId, RevA, CorrelationId.NewId())); + coordinator.ExpectMsg(TimeSpan.FromSeconds(5)).Outcome.ShouldBe(ApplyAckOutcome.Applied); + + AwaitAssert(() => cache.Stores.Count.ShouldBe(1), duration: TimeSpan.FromSeconds(3)); + + var store = cache.Stores[0]; + store.DeploymentId.ShouldBe(deploymentId.ToString()); + store.RevisionHash.ShouldBe(RevA.Value); + store.Artifact.ShouldBe(artifact); + } + + [Fact] + public void ApplyOfAClusterScopedArtifact_StoresUnderThatClusterId() + { + // The ClusterId is not config and not on IClusterRoleInfo — it is carried in the artifact's + // Nodes[] rows. Getting this wrong keys the pair's two nodes to different cache rows, and + // they would silently stop sharing a cache entry. + var db = NewInMemoryDbFactory(); + var cache = new RecordingArtifactCache(); + var (deploymentId, _) = SeedDeployment(db, RevA, clusterId: "SITE-A"); + + var coordinator = CreateTestProbe(); + var actor = Sys.ActorOf(DriverHostActor.Props( + db, TestNode, coordinator.Ref, + localRoles: new HashSet { "driver" }, + deploymentArtifactCache: cache)); + + actor.Tell(new DispatchDeployment(deploymentId, RevA, CorrelationId.NewId())); + coordinator.ExpectMsg(TimeSpan.FromSeconds(5)); + + AwaitAssert(() => cache.Stores.Count.ShouldBe(1), duration: TimeSpan.FromSeconds(3)); + cache.Stores[0].ClusterId.ShouldBe("SITE-A"); + } + + [Fact] + public void UnscopedArtifact_StoresUnderTheSingleClusterSentinel() + { + var db = NewInMemoryDbFactory(); + var cache = new RecordingArtifactCache(); + var (deploymentId, _) = SeedDeployment(db, RevA, clusterId: null); + + var coordinator = CreateTestProbe(); + var actor = Sys.ActorOf(DriverHostActor.Props( + db, TestNode, coordinator.Ref, + localRoles: new HashSet { "driver" }, + deploymentArtifactCache: cache)); + + actor.Tell(new DispatchDeployment(deploymentId, RevA, CorrelationId.NewId())); + coordinator.ExpectMsg(TimeSpan.FromSeconds(5)); + + AwaitAssert(() => cache.Stores.Count.ShouldBe(1), duration: TimeSpan.FromSeconds(3)); + cache.Stores[0].ClusterId.ShouldBe("__single"); + } + + [Fact] + public void EmptyArtifact_IsNotCached() + { + // THE case that makes this cache safe to boot from. ReconcileDrivers catches its own DB + // failures, logs a warning and returns without rethrowing — so an apply can reach its + // success path, ACK Applied, and have started zero drivers. Persisting that as + // last-known-good would boot the node into an empty address space during the next outage: + // strictly worse than having no cache at all. An empty blob is the observable form of it. + var db = NewInMemoryDbFactory(); + var cache = new RecordingArtifactCache(); + + var deploymentId = DeploymentId.NewId(); + using (var ctx = db.CreateDbContext()) + { + ctx.Deployments.Add(new Deployment + { + DeploymentId = deploymentId.Value, + RevisionHash = RevA.Value, + Status = DeploymentStatus.Sealed, + CreatedBy = "test", + SealedAtUtc = DateTime.UtcNow, + ArtifactBlob = Array.Empty(), + }); + ctx.SaveChanges(); + } + + var coordinator = CreateTestProbe(); + var actor = Sys.ActorOf(DriverHostActor.Props( + db, TestNode, coordinator.Ref, + localRoles: new HashSet { "driver" }, + deploymentArtifactCache: cache)); + + actor.Tell(new DispatchDeployment(deploymentId, RevA, CorrelationId.NewId())); + + // The apply still succeeds — an empty artifact is a legitimate no-op deployment. + coordinator.ExpectMsg(TimeSpan.FromSeconds(5)).Outcome.ShouldBe(ApplyAckOutcome.Applied); + + // But nothing is cached from it. + Thread.Sleep(500); + cache.Stores.ShouldBeEmpty(); + } + + [Fact] + public void AThrowingCache_DoesNotFailTheApply() + { + // By the time the cache runs, an Applied ACK has already gone to the coordinator. If a cache + // failure escaped, ApplyAndAck's catch would send a SECOND, contradictory Failed ACK for a + // deployment the fleet already believes is live. + var db = NewInMemoryDbFactory(); + var (deploymentId, _) = SeedDeployment(db, RevA, clusterId: null); + + var coordinator = CreateTestProbe(); + var actor = Sys.ActorOf(DriverHostActor.Props( + db, TestNode, coordinator.Ref, + localRoles: new HashSet { "driver" }, + deploymentArtifactCache: new ThrowingArtifactCache())); + + actor.Tell(new DispatchDeployment(deploymentId, RevA, CorrelationId.NewId())); + + coordinator.ExpectMsg(TimeSpan.FromSeconds(5)).Outcome.ShouldBe(ApplyAckOutcome.Applied); + + // Exactly one ACK, and it says Applied. A second Failed ACK arriving here would mean the + // cache failure had corrupted the deployment protocol. + coordinator.ExpectNoMsg(TimeSpan.FromMilliseconds(500)); + } + + [Fact] + public void NoCacheConfigured_AppliesNormally() + { + // Admin-only graphs and most test harnesses pass null. Caching is skipped, nothing else changes. + var db = NewInMemoryDbFactory(); + var (deploymentId, _) = SeedDeployment(db, RevA, clusterId: null); + + var coordinator = CreateTestProbe(); + var actor = Sys.ActorOf(DriverHostActor.Props( + db, TestNode, coordinator.Ref, + localRoles: new HashSet { "driver" })); + + actor.Tell(new DispatchDeployment(deploymentId, RevA, CorrelationId.NewId())); + + coordinator.ExpectMsg(TimeSpan.FromSeconds(5)).Outcome.ShouldBe(ApplyAckOutcome.Applied); + } + + /// + /// Seeds a sealed deployment whose artifact optionally scopes to a + /// cluster, and returns the exact bytes stored so a round-trip can be asserted. + /// + private static (DeploymentId Id, byte[] Artifact) SeedDeployment( + IDbContextFactory db, RevisionHash rev, string? clusterId) + { + object payload = clusterId is null + ? new + { + DriverInstances = Array.Empty(), + } + : new + { + DriverInstances = Array.Empty(), + // ResolveClusterScope only filters when the artifact declares MORE than one cluster; + // a single-cluster artifact is treated as unscoped. + Clusters = new object[] + { + new { ClusterId = clusterId }, + new { ClusterId = "OTHER" }, + }, + Nodes = new object[] + { + new { NodeId = TestNode.Value, ClusterId = clusterId }, + }, + }; + + var artifact = JsonSerializer.SerializeToUtf8Bytes(payload); + + var id = DeploymentId.NewId(); + using var ctx = db.CreateDbContext(); + ctx.Deployments.Add(new Deployment + { + DeploymentId = id.Value, + RevisionHash = rev.Value, + Status = DeploymentStatus.Sealed, + CreatedBy = "test", + SealedAtUtc = DateTime.UtcNow, + ArtifactBlob = artifact, + }); + ctx.SaveChanges(); + return (id, artifact); + } + + /// Records every store so the test can assert on what was cached. + private sealed class RecordingArtifactCache : IDeploymentArtifactCache + { + private readonly Lock _gate = new(); + private readonly List _stores = []; + + public IReadOnlyList Stores + { + get { lock (_gate) { return _stores.ToArray(); } } + } + + public Task StoreAsync(string clusterId, string deploymentId, string revisionHash, + byte[] artifact, CancellationToken ct = default) + { + lock (_gate) + { + _stores.Add(new StoreCall(clusterId, deploymentId, revisionHash, artifact)); + } + + return Task.CompletedTask; + } + + public Task GetCurrentAsync( + string clusterId, CancellationToken ct = default) + => Task.FromResult(null); + + public Task GetCurrentUnkeyedAsync(CancellationToken ct = default) + => Task.FromResult(null); + + internal sealed record StoreCall( + string ClusterId, string DeploymentId, string RevisionHash, byte[] Artifact); + } + + /// Fails every store, to prove a cache fault cannot reach the deployment protocol. + private sealed class ThrowingArtifactCache : IDeploymentArtifactCache + { + public Task StoreAsync(string clusterId, string deploymentId, string revisionHash, + byte[] artifact, CancellationToken ct = default) + => throw new InvalidOperationException("disk full"); + + public Task GetCurrentAsync( + string clusterId, CancellationToken ct = default) + => Task.FromResult(null); + + public Task GetCurrentUnkeyedAsync(CancellationToken ct = default) + => Task.FromResult(null); + } +} diff --git a/tests/Server/ZB.MOM.WW.OtOpcUa.Runtime.Tests/Drivers/DriverHostActorBootFromCacheTests.cs b/tests/Server/ZB.MOM.WW.OtOpcUa.Runtime.Tests/Drivers/DriverHostActorBootFromCacheTests.cs new file mode 100644 index 00000000..dc03c043 --- /dev/null +++ b/tests/Server/ZB.MOM.WW.OtOpcUa.Runtime.Tests/Drivers/DriverHostActorBootFromCacheTests.cs @@ -0,0 +1,300 @@ +using System.Text.Json; +using Akka.Actor; +using Microsoft.EntityFrameworkCore; +using Shouldly; +using Xunit; +using ZB.MOM.WW.OtOpcUa.Commons.Interfaces; +using ZB.MOM.WW.OtOpcUa.Commons.Messages.Fleet; +using ZB.MOM.WW.OtOpcUa.Commons.Types; +using ZB.MOM.WW.OtOpcUa.Configuration; +using ZB.MOM.WW.OtOpcUa.Configuration.Entities; +using ZB.MOM.WW.OtOpcUa.Configuration.Enums; +using ZB.MOM.WW.OtOpcUa.Runtime.DeploymentCache; +using ZB.MOM.WW.OtOpcUa.Runtime.Drivers; +using ZB.MOM.WW.OtOpcUa.Runtime.OpcUa; +using ZB.MOM.WW.OtOpcUa.Runtime.Tests.Harness; + +namespace ZB.MOM.WW.OtOpcUa.Runtime.Tests.Drivers; + +/// +/// LocalDb Phase 1 (Task 8) — booting from the node-local artifact cache when central SQL is +/// unreachable. +/// +/// +/// The behaviour that matters is asymmetric: the cache must rescue a node whose ConfigDb read +/// failed, and must be completely invisible otherwise. A cache consulted on the happy path would +/// be a route for stale configuration to override fresh configuration. +/// +public sealed class DriverHostActorBootFromCacheTests : RuntimeActorTestBase +{ + private static readonly NodeId TestNode = NodeId.Parse("driver-test"); + private static readonly RevisionHash CachedRev = RevisionHash.Parse(new string('c', 64)); + + [Fact] + public void CentralUnreachableAndCacheHasArtifact_BootsFromCache() + { + var cached = new CachedDeploymentArtifact( + DeploymentId.NewId().ToString(), + CachedRev.Value, + EmptyArtifact(), + DateTimeOffset.UtcNow.AddHours(-1)); + var cache = new StubArtifactCache(cached); + + var actor = Sys.ActorOf(DriverHostActor.Props( + new ThrowingDbFactory(), TestNode, coordinator: null, + localRoles: new HashSet { "driver" }, + deploymentArtifactCache: cache)); + + var snapshot = AskDiagnostics(actor); + + snapshot.RunningFromCache.ShouldBeTrue(); + snapshot.CurrentRevision.ShouldBe(CachedRev); + + // Positive control for the UnkeyedReads counter the "never consults the cache" tests rely + // on. Without this, those ShouldBe(0) assertions would pass just as happily if the counter + // were never incremented at all. + cache.UnkeyedReads.ShouldBe(1); + } + + [Fact] + public void BootFromCache_HandsTheCachedArtifactToTheAddressSpaceRebuild() + { + // The client-facing address space rebuild must materialise from the CACHED bytes, not from a + // ConfigDb read keyed by deployment id — that read is exactly what is unreachable during the + // outage boot-from-cache exists to survive. Without the blob, the node logs "running from + // cache", spins up its drivers, and still serves OPC UA clients an empty server. (Regression + // for the defect the docker-dev live gate caught: 16 nodes on the healthy peer, 1 here.) + var blob = EmptyArtifact(); + var cached = new CachedDeploymentArtifact( + DeploymentId.NewId().ToString(), CachedRev.Value, blob, DateTimeOffset.UtcNow.AddHours(-1)); + var publishProbe = CreateTestProbe(); + + Sys.ActorOf(DriverHostActor.Props( + new ThrowingDbFactory(), TestNode, coordinator: null, + localRoles: new HashSet { "driver" }, + opcUaPublishActor: publishProbe.Ref, + deploymentArtifactCache: new StubArtifactCache(cached))); + + var rebuild = publishProbe.ExpectMsg(TimeSpan.FromSeconds(5)); + rebuild.Artifact.ShouldNotBeNull(); + rebuild.Artifact.ShouldBe(blob); + } + + [Fact] + public void CentralUnreachableAndCacheEmpty_StaysStaleWithNoConfiguration() + { + // The pre-cache behaviour, unchanged. A cache miss must not invent a configuration. + var actor = Sys.ActorOf(DriverHostActor.Props( + new ThrowingDbFactory(), TestNode, coordinator: null, + localRoles: new HashSet { "driver" }, + deploymentArtifactCache: new StubArtifactCache(current: null))); + + var snapshot = AskDiagnostics(actor); + + snapshot.RunningFromCache.ShouldBeFalse(); + snapshot.CurrentRevision.ShouldBeNull(); + } + + [Fact] + public void CentralUnreachableAndNoCacheConfigured_StaysStale() + { + // Admin-only graphs and older harnesses pass null; behaviour is identical to a cache miss. + var actor = Sys.ActorOf(DriverHostActor.Props( + new ThrowingDbFactory(), TestNode, coordinator: null, + localRoles: new HashSet { "driver" })); + + var snapshot = AskDiagnostics(actor); + + snapshot.RunningFromCache.ShouldBeFalse(); + snapshot.CurrentRevision.ShouldBeNull(); + } + + [Fact] + public void CentralReachable_NeverConsultsTheCache() + { + // THE guard against stale config winning. If the cache were read on the happy path, a node + // whose cache held an older revision could quietly serve it over the deployed one. + var cache = new StubArtifactCache(new CachedDeploymentArtifact( + DeploymentId.NewId().ToString(), CachedRev.Value, EmptyArtifact(), DateTimeOffset.UtcNow)); + + var actor = Sys.ActorOf(DriverHostActor.Props( + NewInMemoryDbFactory(), TestNode, coordinator: null, + localRoles: new HashSet { "driver" }, + deploymentArtifactCache: cache)); + + var snapshot = AskDiagnostics(actor); + + snapshot.RunningFromCache.ShouldBeFalse(); + cache.UnkeyedReads.ShouldBe(0); + } + + [Fact] + public void CentralReachableWithAPriorAppliedDeployment_NeverConsultsTheCache() + { + // The RestoreApplied path — the other way a boot can succeed against central. + var db = NewInMemoryDbFactory(); + var rev = RevisionHash.Parse(new string('d', 64)); + var deploymentId = SeedAppliedDeployment(db, rev); + + var cache = new StubArtifactCache(new CachedDeploymentArtifact( + deploymentId.ToString(), CachedRev.Value, EmptyArtifact(), DateTimeOffset.UtcNow)); + + var actor = Sys.ActorOf(DriverHostActor.Props( + db, TestNode, coordinator: null, + localRoles: new HashSet { "driver" }, + deploymentArtifactCache: cache)); + + var snapshot = AskDiagnostics(actor); + + snapshot.RunningFromCache.ShouldBeFalse(); + snapshot.CurrentRevision.ShouldBe(rev); + cache.UnkeyedReads.ShouldBe(0); + } + + [Fact] + public void AThrowingCache_DegradesToStale_RatherThanFailingTheActor() + { + // A cache fault must leave the node exactly where it would have been without a cache. + var actor = Sys.ActorOf(DriverHostActor.Props( + new ThrowingDbFactory(), TestNode, coordinator: null, + localRoles: new HashSet { "driver" }, + deploymentArtifactCache: new ThrowingReadCache())); + + var snapshot = AskDiagnostics(actor); + + snapshot.RunningFromCache.ShouldBeFalse(); + snapshot.CurrentRevision.ShouldBeNull(); + } + + [Fact] + public void RestoringAnAppliedDeploymentOnBootstrap_RepopulatesTheCache() + { + // The cache invariant is "holds what the node serves". A node recovering an already-applied + // deployment (the RestoreApplied path) must (re)write its cache — otherwise a node whose cache + // was lost (fresh volume, disk failure) recovers its served state from central yet stays + // cache-less, unable to boot-from-cache on the NEXT outage. Replication does not heal it: a + // peer's already-acked rows are pruned from its oplog, so a wiped node is never back-filled. + // (Regression for the docker-dev live-gate finding.) + var db = NewInMemoryDbFactory(); + var rev = RevisionHash.Parse(new string('e', 64)); + var deploymentId = SeedAppliedDeployment(db, rev); + + // A cache that starts empty — exactly the wiped-node state. + var cache = new StubArtifactCache(current: null); + + var actor = Sys.ActorOf(DriverHostActor.Props( + db, TestNode, coordinator: null, + localRoles: new HashSet { "driver" }, + deploymentArtifactCache: cache)); + + var snapshot = AskDiagnostics(actor); + snapshot.CurrentRevision.ShouldBe(rev); // recovered the applied deployment + snapshot.RunningFromCache.ShouldBeFalse(); // via central, not the cache + + // The point: the restore path wrote the served artifact back into the cache. + AwaitAssert( + () => cache.Stores.ShouldBe(1), + duration: TimeSpan.FromSeconds(5)); + cache.LastStoredDeploymentId.ShouldBe(deploymentId.ToString()); + } + + private NodeDiagnosticsSnapshot AskDiagnostics(IActorRef actor) + { + var probe = CreateTestProbe(); + // GetDiagnostics is serviced in Steady, Applying AND Stale, so this observes the node + // whichever way the boot resolved. + AwaitAssert( + () => + { + actor.Tell(new GetDiagnostics(CorrelationId.NewId()), probe.Ref); + probe.ExpectMsg(TimeSpan.FromSeconds(2)); + }, + duration: TimeSpan.FromSeconds(5)); + + actor.Tell(new GetDiagnostics(CorrelationId.NewId()), probe.Ref); + return probe.ExpectMsg(TimeSpan.FromSeconds(5)); + } + + /// A syntactically valid artifact with no driver instances. + private static byte[] EmptyArtifact() + => JsonSerializer.SerializeToUtf8Bytes(new { DriverInstances = Array.Empty() }); + + private static DeploymentId SeedAppliedDeployment( + IDbContextFactory db, RevisionHash rev) + { + var id = DeploymentId.NewId(); + using var ctx = db.CreateDbContext(); + ctx.Deployments.Add(new Deployment + { + DeploymentId = id.Value, + RevisionHash = rev.Value, + Status = DeploymentStatus.Sealed, + CreatedBy = "test", + SealedAtUtc = DateTime.UtcNow, + ArtifactBlob = EmptyArtifact(), + }); + ctx.NodeDeploymentStates.Add(new NodeDeploymentState + { + NodeId = TestNode.Value, + DeploymentId = id.Value, + Status = NodeDeploymentStatus.Applied, + StartedAtUtc = DateTime.UtcNow, + }); + ctx.SaveChanges(); + return id; + } + + /// Stands in for an unreachable central SQL Server. + private sealed class ThrowingDbFactory : IDbContextFactory + { + public OtOpcUaConfigDbContext CreateDbContext() + => throw new InvalidOperationException("ConfigDb unreachable"); + } + + private sealed class StubArtifactCache(CachedDeploymentArtifact? current) : IDeploymentArtifactCache + { + private int _unkeyedReads; + private int _stores; + + /// How many times the boot path consulted this cache. + public int UnkeyedReads => Volatile.Read(ref _unkeyedReads); + + /// How many times the node wrote an applied artifact to this cache. + public int Stores => Volatile.Read(ref _stores); + + /// The deployment id of the most recent store. + public string? LastStoredDeploymentId { get; private set; } + + public Task StoreAsync(string clusterId, string deploymentId, string revisionHash, + byte[] artifact, CancellationToken ct = default) + { + Interlocked.Increment(ref _stores); + LastStoredDeploymentId = deploymentId; + return Task.CompletedTask; + } + + public Task GetCurrentAsync( + string clusterId, CancellationToken ct = default) + => Task.FromResult(current); + + public Task GetCurrentUnkeyedAsync(CancellationToken ct = default) + { + Interlocked.Increment(ref _unkeyedReads); + return Task.FromResult(current); + } + } + + private sealed class ThrowingReadCache : IDeploymentArtifactCache + { + public Task StoreAsync(string clusterId, string deploymentId, string revisionHash, + byte[] artifact, CancellationToken ct = default) + => Task.CompletedTask; + + public Task GetCurrentAsync( + string clusterId, CancellationToken ct = default) + => throw new InvalidOperationException("cache unreadable"); + + public Task GetCurrentUnkeyedAsync(CancellationToken ct = default) + => throw new InvalidOperationException("cache unreadable"); + } +}