feat(adminui): manual failover control on the cluster redundancy page
Claude-Session: https://claude.ai/code/session_01GASWkNEi68FSCtvr6rLoEW
This commit is contained in:
+93
@@ -1,10 +1,16 @@
|
||||
@page "/clusters/{ClusterId}/redundancy"
|
||||
@attribute [Authorize(Policy = AdminUiPolicies.AuthenticatedRead)]
|
||||
@rendermode RenderMode.InteractiveServer
|
||||
@using Microsoft.AspNetCore.Authorization
|
||||
@using Microsoft.EntityFrameworkCore
|
||||
@using ZB.MOM.WW.OtOpcUa.AdminUI.Redundancy
|
||||
@using ZB.MOM.WW.OtOpcUa.Configuration
|
||||
@using ZB.MOM.WW.OtOpcUa.Configuration.Entities
|
||||
@using ZB.MOM.WW.OtOpcUa.ControlPlane.Redundancy
|
||||
@inject IDbContextFactory<OtOpcUaConfigDbContext> DbFactory
|
||||
@inject IManualFailoverService FailoverService
|
||||
@inject AuthenticationStateProvider AuthState
|
||||
@inject IAuthorizationService AuthorizationService
|
||||
|
||||
@if (!_loaded)
|
||||
{
|
||||
@@ -45,6 +51,67 @@ else
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<section class="panel rise mt-3" style="animation-delay:.11s">
|
||||
<div class="panel-head">Live redundancy</div>
|
||||
<div style="padding:1rem">
|
||||
<div class="kv">
|
||||
<span class="k">Driver Primary</span>
|
||||
<span class="v mono">@(_failover.Snapshot?.PrimaryAddress ?? "—")</span>
|
||||
</div>
|
||||
<div class="kv">
|
||||
<span class="k">Up driver members</span>
|
||||
<span class="v mono">
|
||||
@(_failover.Snapshot is { DriverAddresses.Count: > 0 } s
|
||||
? string.Join(", ", s.DriverAddresses)
|
||||
: "—")
|
||||
</span>
|
||||
</div>
|
||||
|
||||
<p class="text-muted small mt-2 mb-0">
|
||||
Read live from cluster state on this node. <strong>Mesh-wide scope:</strong> the Primary is
|
||||
elected once per Akka mesh, not per cluster row — in the current single-mesh topology this
|
||||
acts on the whole mesh's Primary, which may be a node of another
|
||||
<span class="mono">Cluster</span>. See <span class="mono">docs/Redundancy.md</span>.
|
||||
</p>
|
||||
|
||||
<AuthorizeView Policy="@ManualFailoverPageModel.RequiredPolicy">
|
||||
<Authorized>
|
||||
<div class="mt-3">
|
||||
<button class="btn btn-sm btn-outline-danger"
|
||||
disabled="@(!_failover.CanFailOver)"
|
||||
title="@(_failover.DisabledReason ?? "Gracefully move the driver Primary to its peer")"
|
||||
@onclick="() => _failover.RequestFailover()">
|
||||
Trigger failover
|
||||
</button>
|
||||
@if (_failover.DisabledReason is { } reason)
|
||||
{
|
||||
<span class="text-muted small ms-2">@reason</span>
|
||||
}
|
||||
</div>
|
||||
|
||||
@if (_failover.ConfirmOpen)
|
||||
{
|
||||
<div class="panel notice mt-3">
|
||||
<strong>Fail over the driver Primary?</strong>
|
||||
<ul class="mb-2 mt-2">
|
||||
<li><span class="mono">@(_failover.Snapshot?.PrimaryAddress)</span> leaves the cluster and its process restarts.</li>
|
||||
<li>Its peer becomes Primary and advertises <span class="mono">ServiceLevel</span> 250.</li>
|
||||
<li>Connected OPC UA clients re-select the new Primary.</li>
|
||||
</ul>
|
||||
<button class="btn btn-sm btn-danger" @onclick="ConfirmFailoverAsync">Confirm failover</button>
|
||||
<button class="btn btn-sm btn-outline-secondary ms-2" @onclick="() => _failover.CancelFailover()">Cancel</button>
|
||||
</div>
|
||||
}
|
||||
</Authorized>
|
||||
</AuthorizeView>
|
||||
|
||||
@if (_failover.StatusMessage is { } msg)
|
||||
{
|
||||
<div class="mt-3 @(_failover.StatusIsError ? "text-danger" : "text-success")">@msg</div>
|
||||
}
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<section class="panel rise mt-3" style="animation-delay:.14s">
|
||||
<div class="panel-head">Node service-level configuration</div>
|
||||
@if (_nodes is null || _nodes.Count == 0)
|
||||
@@ -91,8 +158,16 @@ else
|
||||
private ServerCluster? _cluster;
|
||||
private List<ClusterNode>? _nodes;
|
||||
|
||||
// Everything consequential about the failover control (peer guard, confirm flow, outcome text)
|
||||
// lives in this pure model rather than in the markup — the repo has no bUnit, so logic left in a
|
||||
// .razor is verified only by driving the page. Covered by ManualFailoverPageModelTests.
|
||||
private ManualFailoverPageModel _failover = default!;
|
||||
|
||||
protected override async Task OnInitializedAsync()
|
||||
{
|
||||
_failover = new ManualFailoverPageModel(FailoverService);
|
||||
_failover.Refresh();
|
||||
|
||||
await using var db = await DbFactory.CreateDbContextAsync();
|
||||
_cluster = await db.ServerClusters.AsNoTracking()
|
||||
.FirstOrDefaultAsync(c => c.ClusterId == ClusterId);
|
||||
@@ -105,4 +180,22 @@ else
|
||||
}
|
||||
_loaded = true;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Defense-in-depth: the button is FleetAdmin-gated in markup, but this handler runs on the
|
||||
/// server circuit — re-check the policy before bouncing a production node (the same pattern the
|
||||
/// certificate-store actions use).
|
||||
/// </summary>
|
||||
private async Task ConfirmFailoverAsync()
|
||||
{
|
||||
var authState = await AuthState.GetAuthenticationStateAsync();
|
||||
if (!(await AuthorizationService.AuthorizeAsync(
|
||||
authState.User, null, ManualFailoverPageModel.RequiredPolicy)).Succeeded)
|
||||
{
|
||||
_failover.CancelFailover();
|
||||
return;
|
||||
}
|
||||
|
||||
await _failover.ConfirmFailoverAsync(authState.User.Identity?.Name ?? "system");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,134 @@
|
||||
using ZB.MOM.WW.OtOpcUa.ControlPlane.Redundancy;
|
||||
using ZB.MOM.WW.OtOpcUa.Security.Auth;
|
||||
|
||||
namespace ZB.MOM.WW.OtOpcUa.AdminUI.Redundancy;
|
||||
|
||||
/// <summary>
|
||||
/// Presentation state for the Trigger-failover control on the cluster redundancy page.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// <para>
|
||||
/// A pure model behind a thin razor shell — the same split the driver-typed tag editors use,
|
||||
/// and for the same reason: this repo has no bUnit (see
|
||||
/// <c>PageAuthorizationGuardTests</c>), so anything left inside the <c>.razor</c> is verified
|
||||
/// only by driving the page live. The peer guard, the confirm flow, and the
|
||||
/// refused-vs-succeeded outcome are consequential enough to be unit-testable, so they live
|
||||
/// here.
|
||||
/// </para>
|
||||
/// <para>
|
||||
/// What is deliberately NOT here: the markup authorization gate. It is expressed in the razor
|
||||
/// as <c><AuthorizeView Policy="@ManualFailoverPageModel.RequiredPolicy"></c> and
|
||||
/// re-checked server-side before the call, so <see cref="RequiredPolicy"/> is the single
|
||||
/// source of truth for both.
|
||||
/// </para>
|
||||
/// </remarks>
|
||||
public sealed class ManualFailoverPageModel
|
||||
{
|
||||
/// <summary>
|
||||
/// The policy gating the control, in markup and in the server-side re-check.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// <see cref="AdminUiPolicies.FleetAdmin"/> (Administrator only) rather than the
|
||||
/// <see cref="AdminUiPolicies.ConfigEditor"/> the neighbouring cluster-authoring pages use:
|
||||
/// this does not edit configuration, it restarts a production node. ConfigEditor also admits the
|
||||
/// Designer role, which must not be able to bounce the Primary.
|
||||
/// </remarks>
|
||||
public const string RequiredPolicy = AdminUiPolicies.FleetAdmin;
|
||||
|
||||
private readonly IManualFailoverService _service;
|
||||
|
||||
/// <summary>Creates the model over the failover service.</summary>
|
||||
/// <param name="service">The manual-failover seam; faked in tests.</param>
|
||||
public ManualFailoverPageModel(IManualFailoverService service)
|
||||
=> _service = service ?? throw new ArgumentNullException(nameof(service));
|
||||
|
||||
/// <summary>The last read cluster snapshot, or <see langword="null"/> if it could not be read.</summary>
|
||||
public ManualFailoverSnapshot? Snapshot { get; private set; }
|
||||
|
||||
/// <summary>Whether the confirmation dialog is open.</summary>
|
||||
public bool ConfirmOpen { get; private set; }
|
||||
|
||||
/// <summary>Status text from the last completed action, if any.</summary>
|
||||
public string? StatusMessage { get; private set; }
|
||||
|
||||
/// <summary>Whether <see cref="StatusMessage"/> reports a failure.</summary>
|
||||
public bool StatusIsError { get; private set; }
|
||||
|
||||
/// <summary>Whether the button is enabled — a peer must exist to take over.</summary>
|
||||
public bool CanFailOver => Snapshot?.CanFailOver == true;
|
||||
|
||||
/// <summary>
|
||||
/// Why the button is disabled, or <see langword="null"/> when it is enabled. Rendered as the
|
||||
/// button's tooltip so a disabled control explains itself.
|
||||
/// </summary>
|
||||
public string? DisabledReason => Snapshot is null
|
||||
? "Cluster state is unavailable on this node."
|
||||
: Snapshot.CanFailOver
|
||||
? null
|
||||
: $"Failover needs a driver peer to take over; this mesh has {Snapshot.DriverAddresses.Count} Up driver member(s).";
|
||||
|
||||
/// <summary>Re-reads live cluster state. Never throws — an unreadable cluster disables the button.</summary>
|
||||
public void Refresh()
|
||||
{
|
||||
try
|
||||
{
|
||||
Snapshot = _service.GetSnapshot();
|
||||
}
|
||||
catch (Exception ex)
|
||||
{
|
||||
// A node whose ActorSystem is not yet up (or not a cluster provider) must render the page,
|
||||
// not 500. The button stays disabled with DisabledReason explaining why.
|
||||
Snapshot = null;
|
||||
StatusIsError = true;
|
||||
StatusMessage = $"Could not read cluster state: {ex.Message}";
|
||||
}
|
||||
}
|
||||
|
||||
/// <summary>Opens the confirmation dialog. No-op when the peer guard is closed.</summary>
|
||||
public void RequestFailover()
|
||||
{
|
||||
if (!CanFailOver) return;
|
||||
StatusMessage = null;
|
||||
StatusIsError = false;
|
||||
ConfirmOpen = true;
|
||||
}
|
||||
|
||||
/// <summary>Closes the confirmation dialog without acting.</summary>
|
||||
public void CancelFailover() => ConfirmOpen = false;
|
||||
|
||||
/// <summary>
|
||||
/// Performs the failover the dialog is confirming.
|
||||
/// </summary>
|
||||
/// <param name="actor">The authenticated user name, recorded in the audit event.</param>
|
||||
public async Task ConfirmFailoverAsync(string actor)
|
||||
{
|
||||
if (!ConfirmOpen) return;
|
||||
ConfirmOpen = false;
|
||||
|
||||
try
|
||||
{
|
||||
var target = await _service.FailOverDriverPrimaryAsync(actor).ConfigureAwait(false);
|
||||
if (target is null)
|
||||
{
|
||||
// The service re-evaluates the peer guard against live state, which may have changed
|
||||
// since Refresh(). A refusal is not an error — but it must not read as a success.
|
||||
StatusIsError = true;
|
||||
StatusMessage = "Failover refused: no driver peer is available to take over.";
|
||||
}
|
||||
else
|
||||
{
|
||||
StatusIsError = false;
|
||||
StatusMessage =
|
||||
$"Failover requested. {target} is leaving the cluster; its peer takes over as Primary "
|
||||
+ "and the node restarts as the youngest member.";
|
||||
}
|
||||
}
|
||||
catch (Exception ex)
|
||||
{
|
||||
StatusIsError = true;
|
||||
StatusMessage = $"Failover failed: {ex.Message}";
|
||||
}
|
||||
|
||||
Refresh();
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user