refactor(sf,site): both stores take ILocalDb instead of a connection string
Tasks 5 and 6 of the Phase 2 plan, committed together because their test
fallout is entangled — several fixtures construct both stores.
StoreAndForwardStorage and SiteStorageService now take ILocalDb. Connections
come from ILocalDb.CreateConnection(), which hands out an already-open,
pragma-configured connection carrying the zb_hlc_next() UDF the capture triggers
call; a raw connection would lack the UDF and every write to a replicated table
would fail closed. Deleted with the connection strings: S&F's
EnsureDatabaseDirectoryExists and its per-open busy_timeout pragma, and the site
service's BusyTimeoutFloorSeconds normalization — LocalDb owns all of it now.
DI: AddSiteRuntime's string overload is gone (nothing left to supply), so the
Host calls the no-arg form. ScadaBridge:Database:SiteDbPath and
StoreAndForwardOptions.SqliteDbPath survive only as the migrator's source
locations in Tasks 8/9.
Two things the plan did not anticipate, both worth reading:
1. FOUND A REAL LATENT DEFECT, from Phase 1, now fixed. The plan assumed
directory creation simply moved to LocalDb along with file ownership. It did
not: the LocalDb library never creates the parent directory, and
SqliteLocalDb opens the file eagerly in its constructor — so a missing
directory is a hard boot failure ("SQLite Error 14: unable to open database
file"), not a degraded start. The default site config points at the RELATIVE
path ./data/site-localdb.db, so any site node without a pre-existing data/
directory fails to boot. The docker rig escapes only because its volume mount
happens to create /app/data — a coincidence that would have hidden this until
a bare-metal or fresh deployment. This has been latent since Phase 1 made
LocalDb:Path required; deleting S&F's EnsureDatabaseDirectoryExists here
would have widened it. Re-established the guarantee at the layer that now
owns the path (SiteLocalDbDirectory.Ensure, called before AddZbLocalDb) and
pinned it with SiteLocalDbDirectoryTests. Non-vacuity is not assumed: two
tests written against the wrong assumption failed with exactly this
SQLite Error 14 before the fix existed.
2. Test fallout was ~7x the plan's estimate. The plan named "fixtures" in one
project; the constructor change actually reaches 40 files across 7 test
projects, and most used Mode=Memory;Cache=Shared — which LocalDb has no
equivalent for, so every one had to move to a real temp file. Rather than
copy the Phase 1 TestLocalDb fixture into 7 projects, added a shared
tests/ZB.MOM.WW.ScadaBridge.TestSupport library (not a test project) so the
WAL-sidecar cleanup and the "real, not stubbed" rationale live in one place.
Retargeted rather than deleted, in both directions: the S&F WAL test now asserts
against the LocalDb-backed store (WAL genuinely is LocalDb's job), while the
directory-creation test moved to Host.Tests (that guarantee is NOT LocalDb's).
SiteStorageServiceTests.Initialize_EnablesWalJournalMode got the same treatment.
DeploymentManagerMediumFindingsTests induced a persistence failure via an
unopenable path, which no longer reaches the assertion since the fixture now
throws first; it induces the same failure shape via an uninitialized store.
Verified: full solution build 0 warnings; SiteRuntime 532, Host 318,
AuditLog 355, ExternalSystemGateway 142, HealthMonitoring 97,
StoreAndForward 153 — 1597 passed, 0 failed.
Claude-Session: https://claude.ai/code/session_01BL2Vu1ESDQ9SCN4gVKkdts
This commit is contained in:
+53
-25
@@ -4,6 +4,7 @@ using ZB.MOM.WW.ScadaBridge.SiteRuntime.Actors;
|
||||
using ZB.MOM.WW.ScadaBridge.SiteRuntime.Persistence;
|
||||
using ZB.MOM.WW.ScadaBridge.StoreAndForward;
|
||||
using ZB.MOM.WW.ScadaBridge.Commons.Types.Enums;
|
||||
using ZB.MOM.WW.ScadaBridge.TestSupport;
|
||||
|
||||
namespace ZB.MOM.WW.ScadaBridge.IntegrationTests.Cluster;
|
||||
|
||||
@@ -25,40 +26,67 @@ public class SfBufferResyncPredicateTests
|
||||
var p2 = TwoNodeClusterFixture.GetFreeTcpPort();
|
||||
var (portHigh, portLow) = p1 > p2 ? (p1, p2) : (p2, p1);
|
||||
|
||||
await using var fixture = await TwoNodeClusterFixture.StartAsync(
|
||||
var fixture = await TwoNodeClusterFixture.StartAsync(
|
||||
role: "site-int", portA: portHigh, portB: portLow);
|
||||
|
||||
// Real S&F storage + replication actor per node, production default predicate
|
||||
// (no isActiveOverride) — the exact wiring under test.
|
||||
var (storageOldest, _) = await CreateReplicationActorAsync(fixture.NodeA, "oldest");
|
||||
var (storageJoiner, _) = await CreateReplicationActorAsync(fixture.NodeB, "joiner");
|
||||
// The S&F stores AND SiteStorageService take an ILocalDb (LocalDb has no in-memory
|
||||
// mode), so each node gets its own temp-file local databases. They are disposed
|
||||
// AFTER the cluster is shut down — the actors hold connections while the systems
|
||||
// are alive — and only then are the files (plus their WAL sidecars) deleted.
|
||||
var localDbs = new List<TestLocalDb>();
|
||||
|
||||
// The delivering (oldest) node has a live buffered row the standby never saw.
|
||||
await storageOldest.EnqueueAsync(NewMessage("live-row"));
|
||||
try
|
||||
{
|
||||
// Real S&F storage + replication actor per node, production default predicate
|
||||
// (no isActiveOverride) — the exact wiring under test.
|
||||
var (storageOldest, _, sfDbOldest, siteDbOldest) =
|
||||
await CreateReplicationActorAsync(fixture.NodeA, "oldest");
|
||||
localDbs.Add(sfDbOldest);
|
||||
localDbs.Add(siteDbOldest);
|
||||
var (storageJoiner, _, sfDbJoiner, siteDbJoiner) =
|
||||
await CreateReplicationActorAsync(fixture.NodeB, "joiner");
|
||||
localDbs.Add(sfDbJoiner);
|
||||
localDbs.Add(siteDbJoiner);
|
||||
|
||||
// Trigger peer (re)tracking on both sides: each actor got InitialStateAsSnapshot
|
||||
// in PreStart, but the enqueue raced it — re-deliver via a fresh MemberUp is not
|
||||
// needed; OnPeerTracked already fired on join. The resync exchange is async:
|
||||
// wait until the JOINER holds the row (proves the snapshot flowed oldest→joiner,
|
||||
// the correct direction). Pre-fix this times out (the joiner, as leader, never
|
||||
// requests) AND the oldest node's row is deleted by the stale wipe.
|
||||
await AwaitAsync(async () => await storageJoiner.GetMessageByIdAsync("live-row") != null,
|
||||
TimeSpan.FromSeconds(20),
|
||||
"joiner never received the resync snapshot (resync ran in the wrong direction)");
|
||||
// The delivering (oldest) node has a live buffered row the standby never saw.
|
||||
await storageOldest.EnqueueAsync(NewMessage("live-row"));
|
||||
|
||||
// And the delivering node's buffer is untouched — the N1 wipe assertion.
|
||||
Assert.NotNull(await storageOldest.GetMessageByIdAsync("live-row"));
|
||||
// Trigger peer (re)tracking on both sides: each actor got InitialStateAsSnapshot
|
||||
// in PreStart, but the enqueue raced it — re-deliver via a fresh MemberUp is not
|
||||
// needed; OnPeerTracked already fired on join. The resync exchange is async:
|
||||
// wait until the JOINER holds the row (proves the snapshot flowed oldest→joiner,
|
||||
// the correct direction). Pre-fix this times out (the joiner, as leader, never
|
||||
// requests) AND the oldest node's row is deleted by the stale wipe.
|
||||
await AwaitAsync(async () => await storageJoiner.GetMessageByIdAsync("live-row") != null,
|
||||
TimeSpan.FromSeconds(20),
|
||||
"joiner never received the resync snapshot (resync ran in the wrong direction)");
|
||||
|
||||
// And the delivering node's buffer is untouched — the N1 wipe assertion.
|
||||
Assert.NotNull(await storageOldest.GetMessageByIdAsync("live-row"));
|
||||
}
|
||||
finally
|
||||
{
|
||||
await fixture.DisposeAsync();
|
||||
|
||||
foreach (var localDb in localDbs)
|
||||
{
|
||||
var path = localDb.Path;
|
||||
localDb.Dispose();
|
||||
TestLocalDb.DeleteFiles(path);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private static async Task<(StoreAndForwardStorage Storage, IActorRef Actor)> CreateReplicationActorAsync(
|
||||
ActorSystem node, string tag)
|
||||
private static async Task<(
|
||||
StoreAndForwardStorage Storage, IActorRef Actor, TestLocalDb SfLocalDb, TestLocalDb SiteLocalDb)>
|
||||
CreateReplicationActorAsync(ActorSystem node, string tag)
|
||||
{
|
||||
var sfDb = Path.Combine(Path.GetTempPath(), $"sf-resync-{tag}-{Guid.NewGuid():N}.db");
|
||||
var siteDb = Path.Combine(Path.GetTempPath(), $"site-resync-{tag}-{Guid.NewGuid():N}.db");
|
||||
var sfStorage = new StoreAndForwardStorage($"Data Source={sfDb}",
|
||||
var sfLocalDb = TestLocalDb.CreateTemp($"sf-resync-{tag}");
|
||||
var sfStorage = new StoreAndForwardStorage(sfLocalDb.Db,
|
||||
NullLogger<StoreAndForwardStorage>.Instance);
|
||||
await sfStorage.InitializeAsync();
|
||||
var siteStorage = new SiteStorageService($"Data Source={siteDb}",
|
||||
var siteLocalDb = TestLocalDb.CreateTemp($"site-resync-{tag}");
|
||||
var siteStorage = new SiteStorageService(siteLocalDb.Db,
|
||||
NullLogger<SiteStorageService>.Instance);
|
||||
var replicationService = new ReplicationService(
|
||||
new StoreAndForwardOptions(), NullLogger<ReplicationService>.Instance);
|
||||
@@ -67,7 +95,7 @@ public class SfBufferResyncPredicateTests
|
||||
siteStorage, sfStorage, replicationService, "site-int",
|
||||
NullLogger<SiteReplicationActor>.Instance, null, null, null, null)),
|
||||
"site-replication");
|
||||
return (sfStorage, actor);
|
||||
return (sfStorage, actor, sfLocalDb, siteLocalDb);
|
||||
}
|
||||
|
||||
private static StoreAndForwardMessage NewMessage(string id) => new()
|
||||
|
||||
Reference in New Issue
Block a user