f2efeb37b7
Tasks 5 and 6 of the Phase 2 plan, committed together because their test
fallout is entangled — several fixtures construct both stores.
StoreAndForwardStorage and SiteStorageService now take ILocalDb. Connections
come from ILocalDb.CreateConnection(), which hands out an already-open,
pragma-configured connection carrying the zb_hlc_next() UDF the capture triggers
call; a raw connection would lack the UDF and every write to a replicated table
would fail closed. Deleted with the connection strings: S&F's
EnsureDatabaseDirectoryExists and its per-open busy_timeout pragma, and the site
service's BusyTimeoutFloorSeconds normalization — LocalDb owns all of it now.
DI: AddSiteRuntime's string overload is gone (nothing left to supply), so the
Host calls the no-arg form. ScadaBridge:Database:SiteDbPath and
StoreAndForwardOptions.SqliteDbPath survive only as the migrator's source
locations in Tasks 8/9.
Two things the plan did not anticipate, both worth reading:
1. FOUND A REAL LATENT DEFECT, from Phase 1, now fixed. The plan assumed
directory creation simply moved to LocalDb along with file ownership. It did
not: the LocalDb library never creates the parent directory, and
SqliteLocalDb opens the file eagerly in its constructor — so a missing
directory is a hard boot failure ("SQLite Error 14: unable to open database
file"), not a degraded start. The default site config points at the RELATIVE
path ./data/site-localdb.db, so any site node without a pre-existing data/
directory fails to boot. The docker rig escapes only because its volume mount
happens to create /app/data — a coincidence that would have hidden this until
a bare-metal or fresh deployment. This has been latent since Phase 1 made
LocalDb:Path required; deleting S&F's EnsureDatabaseDirectoryExists here
would have widened it. Re-established the guarantee at the layer that now
owns the path (SiteLocalDbDirectory.Ensure, called before AddZbLocalDb) and
pinned it with SiteLocalDbDirectoryTests. Non-vacuity is not assumed: two
tests written against the wrong assumption failed with exactly this
SQLite Error 14 before the fix existed.
2. Test fallout was ~7x the plan's estimate. The plan named "fixtures" in one
project; the constructor change actually reaches 40 files across 7 test
projects, and most used Mode=Memory;Cache=Shared — which LocalDb has no
equivalent for, so every one had to move to a real temp file. Rather than
copy the Phase 1 TestLocalDb fixture into 7 projects, added a shared
tests/ZB.MOM.WW.ScadaBridge.TestSupport library (not a test project) so the
WAL-sidecar cleanup and the "real, not stubbed" rationale live in one place.
Retargeted rather than deleted, in both directions: the S&F WAL test now asserts
against the LocalDb-backed store (WAL genuinely is LocalDb's job), while the
directory-creation test moved to Host.Tests (that guarantee is NOT LocalDb's).
SiteStorageServiceTests.Initialize_EnablesWalJournalMode got the same treatment.
DeploymentManagerMediumFindingsTests induced a persistence failure via an
unopenable path, which no longer reaches the assertion since the fixture now
throws first; it induces the same failure shape via an uninitialized store.
Verified: full solution build 0 warnings; SiteRuntime 532, Host 318,
AuditLog 355, ExternalSystemGateway 142, HealthMonitoring 97,
StoreAndForward 153 — 1597 passed, 0 failed.
Claude-Session: https://claude.ai/code/session_01BL2Vu1ESDQ9SCN4gVKkdts
123 lines
5.6 KiB
C#
123 lines
5.6 KiB
C#
using Akka.Actor;
|
|
using Microsoft.Extensions.Logging.Abstractions;
|
|
using ZB.MOM.WW.ScadaBridge.SiteRuntime.Actors;
|
|
using ZB.MOM.WW.ScadaBridge.SiteRuntime.Persistence;
|
|
using ZB.MOM.WW.ScadaBridge.StoreAndForward;
|
|
using ZB.MOM.WW.ScadaBridge.Commons.Types.Enums;
|
|
using ZB.MOM.WW.ScadaBridge.TestSupport;
|
|
|
|
namespace ZB.MOM.WW.ScadaBridge.IntegrationTests.Cluster;
|
|
|
|
/// <summary>
|
|
/// N1 regression (review 02 round 2, Critical): the resync authority must use the same
|
|
/// oldest-Up predicate as the S&F delivery gate. Divergence scenario = the delivering node
|
|
/// is OLDEST but not LEADER (leader = lowest address), the exact state a rolling restart of
|
|
/// the lower-address node produces. Pre-fix the delivering node requests a resync from the
|
|
/// stale peer and ReplaceAllAsync wipes its live buffer.
|
|
/// </summary>
|
|
public class SfBufferResyncPredicateTests
|
|
{
|
|
[Fact]
|
|
public async Task OldestButNotLeaderNode_KeepsItsBuffer_AndSeedsTheJoiner()
|
|
{
|
|
// Two explicit ports, deliberately assigned so the FIRST-started (oldest,
|
|
// delivering) node has the HIGHER address → the second node is cluster leader.
|
|
var p1 = TwoNodeClusterFixture.GetFreeTcpPort();
|
|
var p2 = TwoNodeClusterFixture.GetFreeTcpPort();
|
|
var (portHigh, portLow) = p1 > p2 ? (p1, p2) : (p2, p1);
|
|
|
|
var fixture = await TwoNodeClusterFixture.StartAsync(
|
|
role: "site-int", portA: portHigh, portB: portLow);
|
|
|
|
// The S&F stores AND SiteStorageService take an ILocalDb (LocalDb has no in-memory
|
|
// mode), so each node gets its own temp-file local databases. They are disposed
|
|
// AFTER the cluster is shut down — the actors hold connections while the systems
|
|
// are alive — and only then are the files (plus their WAL sidecars) deleted.
|
|
var localDbs = new List<TestLocalDb>();
|
|
|
|
try
|
|
{
|
|
// Real S&F storage + replication actor per node, production default predicate
|
|
// (no isActiveOverride) — the exact wiring under test.
|
|
var (storageOldest, _, sfDbOldest, siteDbOldest) =
|
|
await CreateReplicationActorAsync(fixture.NodeA, "oldest");
|
|
localDbs.Add(sfDbOldest);
|
|
localDbs.Add(siteDbOldest);
|
|
var (storageJoiner, _, sfDbJoiner, siteDbJoiner) =
|
|
await CreateReplicationActorAsync(fixture.NodeB, "joiner");
|
|
localDbs.Add(sfDbJoiner);
|
|
localDbs.Add(siteDbJoiner);
|
|
|
|
// The delivering (oldest) node has a live buffered row the standby never saw.
|
|
await storageOldest.EnqueueAsync(NewMessage("live-row"));
|
|
|
|
// Trigger peer (re)tracking on both sides: each actor got InitialStateAsSnapshot
|
|
// in PreStart, but the enqueue raced it — re-deliver via a fresh MemberUp is not
|
|
// needed; OnPeerTracked already fired on join. The resync exchange is async:
|
|
// wait until the JOINER holds the row (proves the snapshot flowed oldest→joiner,
|
|
// the correct direction). Pre-fix this times out (the joiner, as leader, never
|
|
// requests) AND the oldest node's row is deleted by the stale wipe.
|
|
await AwaitAsync(async () => await storageJoiner.GetMessageByIdAsync("live-row") != null,
|
|
TimeSpan.FromSeconds(20),
|
|
"joiner never received the resync snapshot (resync ran in the wrong direction)");
|
|
|
|
// And the delivering node's buffer is untouched — the N1 wipe assertion.
|
|
Assert.NotNull(await storageOldest.GetMessageByIdAsync("live-row"));
|
|
}
|
|
finally
|
|
{
|
|
await fixture.DisposeAsync();
|
|
|
|
foreach (var localDb in localDbs)
|
|
{
|
|
var path = localDb.Path;
|
|
localDb.Dispose();
|
|
TestLocalDb.DeleteFiles(path);
|
|
}
|
|
}
|
|
}
|
|
|
|
private static async Task<(
|
|
StoreAndForwardStorage Storage, IActorRef Actor, TestLocalDb SfLocalDb, TestLocalDb SiteLocalDb)>
|
|
CreateReplicationActorAsync(ActorSystem node, string tag)
|
|
{
|
|
var sfLocalDb = TestLocalDb.CreateTemp($"sf-resync-{tag}");
|
|
var sfStorage = new StoreAndForwardStorage(sfLocalDb.Db,
|
|
NullLogger<StoreAndForwardStorage>.Instance);
|
|
await sfStorage.InitializeAsync();
|
|
var siteLocalDb = TestLocalDb.CreateTemp($"site-resync-{tag}");
|
|
var siteStorage = new SiteStorageService(siteLocalDb.Db,
|
|
NullLogger<SiteStorageService>.Instance);
|
|
var replicationService = new ReplicationService(
|
|
new StoreAndForwardOptions(), NullLogger<ReplicationService>.Instance);
|
|
// Name MUST be "site-replication" — SendToPeer targets /user/site-replication.
|
|
var actor = node.ActorOf(Props.Create(() => new SiteReplicationActor(
|
|
siteStorage, sfStorage, replicationService, "site-int",
|
|
NullLogger<SiteReplicationActor>.Instance, null, null, null, null)),
|
|
"site-replication");
|
|
return (sfStorage, actor, sfLocalDb, siteLocalDb);
|
|
}
|
|
|
|
private static StoreAndForwardMessage NewMessage(string id) => new()
|
|
{
|
|
Id = id,
|
|
Category = StoreAndForwardCategory.Notification,
|
|
Target = "central",
|
|
PayloadJson = "{}",
|
|
CreatedAt = DateTimeOffset.UtcNow,
|
|
Status = StoreAndForwardMessageStatus.Pending,
|
|
MaxRetries = 0,
|
|
};
|
|
|
|
private static async Task AwaitAsync(Func<Task<bool>> condition, TimeSpan timeout, string why)
|
|
{
|
|
var deadline = DateTime.UtcNow + timeout;
|
|
while (DateTime.UtcNow < deadline)
|
|
{
|
|
if (await condition()) return;
|
|
await Task.Delay(250);
|
|
}
|
|
throw new TimeoutException(why);
|
|
}
|
|
}
|