Files
ScadaBridge/tests/ZB.MOM.WW.ScadaBridge.IntegrationTests/LocalDbSitePairHarness.cs
T
Joseph Doherty 2bbe66311d test(localdb): port store-and-forward replication intents as CDC convergence specs
Task 10. Specifications first, deletion second: the bespoke ReplicationService
(explicit Add/Remove/Park/Requeue over Akka) dies at Task 14, so its behaviour
is restated here as outcomes the CDC replacement must still deliver. Written
in terms of ROWS, not operations — under CDC there is no Add or Park message to
observe, only a row that must end up right on both nodes.

Not ported: ReplicationOperations_AreDispatchedInIssueOrder. It asserts the
mechanism (inline fire-and-forget dispatch), and CDC capture is asynchronous
and batched by construction. Its portable content is the ordering OUTCOME —
add-then-remove must never converge to present — which is a test here, with
that reasoning recorded in the file so it does not read as an accidental drop.

DEVIATION: extracted the Phase 1 fixture into LocalDbSitePairHarness rather
than duplicating ~150 lines. Phase 1's tests now derive from it and still pass
unchanged. The harness registers the Phase 2 tables itself, since production
OnReady does not until Task 14; that method is marked for deletion at the
cutover, and the 8-table list is written literally so a cutover registering the
wrong set fails these tests instead of agreeing with itself.

Non-vacuity verified by unregistering sf_messages: 6 of 7 failed. The 7th —
the ordering test — PASSED, because an absent row is also what a pair that
replicates nothing looks like. Fixed with a control row that must converge in
the same window, so the absence is evidence rather than silence.

Also corrected two comments from Task 9 that claimed Task 14 makes
notification_lists/smtp_configurations replicated. It explicitly does not
register them, for the same reason the migrator skips them.

Claude-Session: https://claude.ai/code/session_01BL2Vu1ESDQ9SCN4gVKkdts
2026-07-20 03:50:48 -04:00

287 lines
12 KiB
C#

using System.Net;
using Microsoft.AspNetCore.Builder;
using Microsoft.AspNetCore.Hosting;
using Microsoft.AspNetCore.Hosting.Server;
using Microsoft.AspNetCore.Hosting.Server.Features;
using Microsoft.AspNetCore.Server.Kestrel.Core;
using Microsoft.Extensions.Configuration;
using Microsoft.Extensions.DependencyInjection;
using Microsoft.Extensions.Hosting;
using ZB.MOM.WW.LocalDb;
using ZB.MOM.WW.LocalDb.Replication;
using ZB.MOM.WW.ScadaBridge.Host;
namespace ZB.MOM.WW.ScadaBridge.IntegrationTests;
/// <summary>
/// Serializes the site-pair convergence tests against each other: each one stands up a real
/// Kestrel listener plus two SQLite files, and running them concurrently under CI
/// contention is a flakiness risk.
/// </summary>
[CollectionDefinition("LocalDbSitePairConvergence")]
public sealed class LocalDbSitePairConvergenceCollection;
/// <summary>
/// Two ScadaBridge site nodes replicating the consolidated site database over a REAL
/// loopback gRPC transport, through the REAL fail-closed auth interceptor.
/// </summary>
/// <remarks>
/// <para>
/// Extracted from the Phase 1 convergence tests once Phase 2 needed the same pair for the
/// store-and-forward buffer and the configuration tables. Everything here is fixture; the
/// derived classes hold only their own data helpers and scenarios.
/// </para>
/// <para>
/// It uses <see cref="SiteLocalDbSetup.OnReady"/>, not a hand-written schema, so the tables,
/// their primary keys, and the registration ORDER under test are the ones the host actually
/// runs. A separate schema here would prove only that the test agrees with itself.
/// </para>
/// <para>
/// Offline: no docker, no external services. Loopback Kestrel with h2c.
/// </para>
/// </remarks>
public abstract class LocalDbSitePairHarness : IAsyncLifetime
{
private const string SharedApiKey = "site-pair-convergence-key";
/// <summary>How long a scenario waits for the pair to agree before failing.</summary>
protected static readonly TimeSpan ConvergeTimeout = TimeSpan.FromSeconds(30);
private readonly string _pathA = Path.Combine(Path.GetTempPath(), $"sitepairA-{Guid.NewGuid():N}.db");
private readonly string _pathB = Path.Combine(Path.GetTempPath(), $"sitepairB-{Guid.NewGuid():N}.db");
// The databases are owned by the fixture, in their own providers, and registered into the
// hosts as pre-constructed instances. MS.DI does not dispose instances it did not create,
// so tearing a host down (the offline-peer scenario) leaves the databases intact and
// writable — which is exactly what lets node A accumulate writes while B is down.
private ServiceProvider _dbProviderA = null!;
private ServiceProvider _dbProviderB = null!;
private IHost? _serverHost; // node B — passive
private IHost? _initiatorHost; // node A — dials the peer
static LocalDbSitePairHarness() =>
// Grpc.Net.Client dials the loopback server over HTTP/2 cleartext.
AppContext.SetSwitch("System.Net.Http.SocketsHttpHandler.Http2UnencryptedSupport", true);
/// <summary>Node A — the initiator, which dials the peer.</summary>
protected ILocalDb A => _dbProviderA.GetRequiredService<ILocalDb>();
/// <summary>Node B — the passive node, which listens.</summary>
protected ILocalDb B => _dbProviderB.GetRequiredService<ILocalDb>();
public async Task InitializeAsync()
{
_dbProviderA = BuildDatabaseProvider(_pathA, "node-a");
_dbProviderB = BuildDatabaseProvider(_pathB, "node-b");
// Force construction (and therefore OnReady) before anything replicates.
_ = A;
_ = B;
await StartPassiveAsync();
await StartInitiatorAsync();
}
public async Task DisposeAsync()
{
await StopHostAsync(_initiatorHost);
await StopHostAsync(_serverHost);
await _dbProviderA.DisposeAsync();
await _dbProviderB.DisposeAsync();
Microsoft.Data.Sqlite.SqliteConnection.ClearAllPools();
foreach (var path in new[] { _pathA, _pathB })
{
foreach (var suffix in new[] { "", "-wal", "-shm" })
{
try { File.Delete(path + suffix); } catch { /* best effort */ }
}
}
}
// ---- fixture internals ------------------------------------------------------------
/// <summary>
/// A provider owning one consolidated site database, initialized through the host's own
/// <see cref="SiteLocalDbSetup.OnReady"/> — same schema, same registration order.
/// </summary>
private static ServiceProvider BuildDatabaseProvider(string path, string nodeName)
{
var config = new ConfigurationBuilder()
.AddInMemoryCollection(new Dictionary<string, string?>
{
["LocalDb:Path"] = path,
["ScadaBridge:Node:NodeName"] = nodeName,
// Point the legacy migrators at paths that do not exist, so they no-op rather
// than picking up stray files from the test working directory. The two Phase 2
// defaults matter most: unlike the Phase 1 pair they resolve inside ./data/,
// and a migration would also RENAME whatever it found.
["ScadaBridge:OperationTracking:ConnectionString"] =
$"Data Source={Path.Combine(Path.GetTempPath(), $"absent-{Guid.NewGuid():N}.db")}",
["ScadaBridge:SiteEventLog:DatabasePath"] =
Path.Combine(Path.GetTempPath(), $"absent-{Guid.NewGuid():N}.db"),
["ScadaBridge:StoreAndForward:SqliteDbPath"] =
Path.Combine(Path.GetTempPath(), $"absent-{Guid.NewGuid():N}.db"),
["ScadaBridge:Database:SiteDbPath"] =
Path.Combine(Path.GetTempPath(), $"absent-{Guid.NewGuid():N}.db"),
})
.Build();
return new ServiceCollection()
.AddZbLocalDb(config, db =>
{
SiteLocalDbSetup.OnReady(db, config);
RegisterPhase2TablesUntilCutover(db);
})
.BuildServiceProvider();
}
/// <summary>
/// Registers the Phase 2 tables for capture, which production <c>OnReady</c> does not do
/// until the Task 14 cutover.
/// </summary>
/// <remarks>
/// <b>Delete this method at Task 14</b>, along with its call above. Until the cutover the
/// bespoke <c>SiteReplicationActor</c> and <c>ReplicationService</c> still own these
/// tables, so <c>OnReady</c> deliberately leaves them unregistered — but the convergence
/// specifications the cutover has to satisfy need capture triggers to mean anything. The
/// tables are empty at this point, so registering here captures nothing retroactively;
/// the ordering guarantee under test is unaffected.
/// <para>
/// After Task 14 these registrations come from <c>OnReady</c> itself, and leaving this
/// method behind would mask a cutover that forgot one — the whole point of these tests.
/// </para>
/// </remarks>
private static void RegisterPhase2TablesUntilCutover(ILocalDb db)
{
foreach (var table in Phase2ReplicatedTables)
db.RegisterReplicated(table);
}
/// <summary>
/// The eight tables Task 14 registers. Listed literally rather than derived from
/// production code, so that a cutover which registers the wrong set fails these tests
/// instead of agreeing with itself.
/// </summary>
/// <remarks>
/// <c>notification_lists</c> and <c>smtp_configurations</c> are absent by design. They
/// are permanently empty (no site writer since 2026-07-10, the migrator skips them, the
/// active-node purge keeps them empty), so registering them would open a standing
/// replication channel whose only historical payload was plaintext SMTP passwords.
/// </remarks>
protected static readonly string[] Phase2ReplicatedTables =
[
"sf_messages",
"deployed_configurations", "static_attribute_overrides", "shared_scripts",
"external_systems", "database_connections", "data_connection_definitions",
"native_alarm_state",
];
private static IConfiguration ReplicationConfig(string? peerAddress)
{
var values = new Dictionary<string, string?>
{
// Tight flush + bounded reconnect backoff so convergence is observable well
// inside the poll deadline. The 60 s production default would let the doubling
// backoff overrun it after a peer outage.
["LocalDb:Replication:FlushInterval"] = "00:00:00.050",
["LocalDb:Replication:ReconnectBackoffMax"] = "00:00:02",
// Both nodes share one key — the interceptor is fail-closed, so a mismatch here
// turns every scenario below red (verified by deliberately breaking it).
["LocalDb:Replication:ApiKey"] = SharedApiKey,
};
if (peerAddress is not null)
values["LocalDb:Replication:PeerAddress"] = peerAddress;
return new ConfigurationBuilder().AddInMemoryCollection(values).Build();
}
/// <summary>Starts node B, the passive listener.</summary>
protected async Task StartPassiveAsync()
{
var config = ReplicationConfig(peerAddress: null);
_serverHost = await new HostBuilder()
.ConfigureWebHost(web =>
{
web.UseKestrel(o =>
o.Listen(IPAddress.Loopback, 0, listen => listen.Protocols = HttpProtocols.Http2));
web.ConfigureServices(services =>
{
services.AddLogging();
services.AddRouting();
// The REAL interceptor, not a stand-in. If it rejected legitimate peer
// traffic, every scenario below would fail — which is the point.
services.AddGrpc(o => o.Interceptors.Add<LocalDbSyncAuthInterceptor>());
services.AddSingleton(B);
services.AddZbLocalDbReplication(config);
});
web.Configure(app =>
{
app.UseRouting();
app.UseEndpoints(e => e.MapZbLocalDbSync());
});
})
.StartAsync();
}
/// <summary>Starts node A, which dials the passive node.</summary>
protected async Task StartInitiatorAsync()
{
var config = ReplicationConfig(PassiveAddress());
_initiatorHost = await new HostBuilder()
.ConfigureServices(services =>
{
services.AddLogging();
services.AddSingleton(A);
services.AddZbLocalDbReplication(config);
})
.StartAsync();
}
/// <summary>Takes node B's listener down, leaving its database intact and writable.</summary>
protected async Task StopPassiveAsync()
{
await StopHostAsync(_serverHost);
_serverHost = null;
}
/// <summary>
/// Brings node B back on a NEW loopback port and re-dials from A. The initiator's channel
/// factory re-reads the peer address on each reconnect, so this is a genuine rejoin.
/// </summary>
protected async Task RestartPairAsync()
{
await StartPassiveAsync();
await StopHostAsync(_initiatorHost);
await StartInitiatorAsync();
}
private string PassiveAddress()
=> _serverHost!.Services.GetRequiredService<IServer>()
.Features.Get<IServerAddressesFeature>()!.Addresses.Single();
private static async Task StopHostAsync(IHost? host)
{
if (host is null) return;
try { await host.StopAsync(TimeSpan.FromSeconds(5)); } catch { /* teardown */ }
host.Dispose();
}
/// <summary>Polls <paramref name="condition"/> until true or the deadline passes.</summary>
protected static async Task WaitUntilAsync(Func<Task<bool>> condition, string because)
{
var deadline = DateTime.UtcNow + ConvergeTimeout;
while (DateTime.UtcNow < deadline)
{
if (await condition()) return;
await Task.Delay(50);
}
Assert.Fail($"Timed out after {ConvergeTimeout.TotalSeconds:0}s waiting for: {because}");
}
}