feat(mesh): FetchAndCache bootstrap from the LocalDb pointer; no driver-side config SQL read
Phase 3 Task 6. In FetchAndCache mode Bootstrap() branches to BootstrapFromCache(), which restores served state from the LocalDb pointer (GetCurrentUnkeyedAsync → set _currentRevision → ApplyCachedArtifact from the cached bytes) with NO central-SQL read. Distinct from the Direct-mode TryBootFromCache fallback: reading the cache is NORMAL operation here so _isRunningFromCache stays false (the node can still fetch new deploys). An empty OR unreadable cache lands Steady-with-no-revision (the first dispatch fetches), never Stale — Stale means 'central SQL down, retry it', and FetchAndCache has no config SQL read to recover. TryRecoverFromStale gains a defensive FetchAndCache guard (it is unreachable in that mode, since the retry-db timer only starts in Stale). Proven with a ThrowingDbFactory in every test: reaching Steady + applying a dispatch proves the boot never touched central SQL. Sabotaging the mode branch (fall through to the SQL read) reddens all three — test A flips RunningFromCache; B/C enter Stale, which ignores the dispatch. Claude-Session: https://claude.ai/code/session_01GASWkNEi68FSCtvr6rLoEW
This commit is contained in:
@@ -595,6 +595,14 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers
|
||||
|
||||
private void Bootstrap()
|
||||
{
|
||||
// Per-cluster mesh Phase 3: a FetchAndCache node NEVER reads central SQL for config, at boot
|
||||
// or otherwise. It restores its served state from the LocalDb pointer instead.
|
||||
if (_fetchAndCacheMode)
|
||||
{
|
||||
BootstrapFromCache();
|
||||
return;
|
||||
}
|
||||
|
||||
// Read the most-recent NodeDeploymentState for this node; if it's Applied, jump
|
||||
// to Steady with that revision. If Applying (orphan from a crash), discard and replay.
|
||||
// If the DB is unreachable, fall back to Stale and start the reconnect loop.
|
||||
@@ -659,6 +667,60 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers
|
||||
}
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Per-cluster mesh Phase 3 boot for <c>FetchAndCache</c> mode. Restores served state from the
|
||||
/// LocalDb pointer without any central-SQL read.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// <para>
|
||||
/// Distinct from <see cref="TryBootFromCache"/> (the Direct-mode SQL-unreachable fallback):
|
||||
/// reading from the cache is <b>normal</b> operation here, not a degraded state, so
|
||||
/// <see cref="_isRunningFromCache"/> stays <see langword="false"/> — the node can still
|
||||
/// fetch new deployments. An empty or unreadable cache lands Steady-with-no-revision (the
|
||||
/// first dispatch fetches), never <c>Stale</c>: Stale means "central SQL is down, retry it",
|
||||
/// and FetchAndCache has no config SQL read to recover.
|
||||
/// </para>
|
||||
/// </remarks>
|
||||
private void BootstrapFromCache()
|
||||
{
|
||||
try
|
||||
{
|
||||
var cached = _deploymentArtifactCache?.GetCurrentUnkeyedAsync().GetAwaiter().GetResult();
|
||||
if (cached is null)
|
||||
{
|
||||
_log.Info(
|
||||
"DriverHost {Node}: FetchAndCache boot — the LocalDb cache is empty; entering Steady " +
|
||||
"(no revision). The first dispatch will fetch this node's configuration from central.",
|
||||
_localNode);
|
||||
Become(Steady);
|
||||
return;
|
||||
}
|
||||
|
||||
var deploymentId = DeploymentId.Parse(cached.DeploymentId);
|
||||
var revision = RevisionHash.Parse(cached.RevisionHash);
|
||||
|
||||
_currentRevision = revision;
|
||||
Become(Steady);
|
||||
// Restore drivers + address space + subscriptions from the cached bytes — no SQL read, no
|
||||
// re-ack (the deployment is already Applied). Same in-hand-bytes apply as the Direct-mode
|
||||
// cache boot, but this is the steady-state config source, not an outage fallback.
|
||||
ApplyCachedArtifact(deploymentId, cached.Artifact);
|
||||
|
||||
_log.Info(
|
||||
"DriverHost {Node}: FetchAndCache boot — restored served state for deployment {Id} " +
|
||||
"(rev {Rev}) from the LocalDb pointer.",
|
||||
_localNode, deploymentId, revision);
|
||||
}
|
||||
catch (Exception ex)
|
||||
{
|
||||
_log.Warning(ex,
|
||||
"DriverHost {Node}: FetchAndCache boot — failed to read the LocalDb pointer; entering " +
|
||||
"Steady (no revision). The next dispatch will fetch.",
|
||||
_localNode);
|
||||
Become(Steady);
|
||||
}
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Last-resort boot path: apply the artifact cached by a previous successful deploy (or
|
||||
/// replicated from this node's pair peer) when central SQL cannot be reached.
|
||||
@@ -2511,6 +2573,16 @@ public sealed class DriverHostActor : ReceiveActor, IWithTimers
|
||||
|
||||
private void TryRecoverFromStale()
|
||||
{
|
||||
// Defensive: FetchAndCache never enters Stale (BootstrapFromCache always Becomes Steady, so
|
||||
// the retry-db timer that drives this is never started), and its recovery must NOT read
|
||||
// central SQL. If we somehow land here, leave Steady rather than re-reading Deployments.
|
||||
if (_fetchAndCacheMode)
|
||||
{
|
||||
Timers.Cancel("retry-db");
|
||||
Become(Steady);
|
||||
return;
|
||||
}
|
||||
|
||||
try
|
||||
{
|
||||
using var db = _dbFactory.CreateDbContext();
|
||||
|
||||
Reference in New Issue
Block a user