using Microsoft.Extensions.Hosting;
using Microsoft.Extensions.Logging;
namespace ZB.MOM.WW.ScadaBridge.HealthMonitoring;
///
/// Site Event Logging — site-side hosted service that
/// periodically reads the cumulative event-log write-failure count and pushes
/// it into so the next
/// emits a fresh
/// SiteEventLogWriteFailures field on the site health report.
///
///
///
/// Why a Func<long> and not ISiteEventLogger directly.
/// A direct HealthMonitoring → SiteEventLogging reference is avoided
/// to prevent an undesirable low-level coupling: SiteEventLogging is a
/// leaf component that should not pull in higher-level infrastructure. Note that
/// HealthMonitoring → StoreAndForward → SiteEventLogging already
/// exists as a transitive path (confirmed: StoreAndForward.csproj references
/// SiteEventLogging.csproj), so a direct reference would NOT introduce a
/// cycle — the delegate is purely a coupling-avoidance measure. The
/// seam lets the caller (Host site wiring) capture
/// ISiteEventLogger.FailedWriteCount as a lambda at registration time; this
/// service reads only the numeric result. The delegate approach is a standard
/// pattern for counter bridges and keeps the registration path self-documenting.
///
///
/// Cadence. 30 s by default — the same cadence as
/// SiteAuditBacklogReporter, which is coarse enough to stay within
/// the health-report interval budget while keeping the central dashboard
/// current.
///
///
/// Failure containment. Any unexpected exception during the probe is
/// caught and logged; the next tick retries. Mirrors
/// SiteAuditBacklogReporter's "exception logged, not propagated"
/// contract.
///
///
public sealed class SiteEventLogFailureCountReporter : IHostedService, IDisposable
{
///
/// Default poll cadence. Matches SiteAuditBacklogReporter.DefaultRefreshInterval
/// (30 s) — coarse enough to amortise the read across many reports, fine
/// enough that the central dashboard never lags by more than one
/// health-report interval.
///
internal static readonly TimeSpan DefaultRefreshInterval = TimeSpan.FromSeconds(30);
private readonly Func _failedWriteCountProvider;
private readonly ISiteHealthCollector _collector;
private readonly ILogger _logger;
private readonly TimeSpan _refreshInterval;
private CancellationTokenSource? _cts;
private Task? _loop;
/// Initializes a new instance of .
///
/// A delegate that returns the current cumulative event-log write-failure count.
/// Typically wired as () => sp.GetRequiredService<ISiteEventLogger>().FailedWriteCount
/// in the Host site composition root.
///
/// The site health collector that receives the failure-count snapshot.
/// Logger instance.
/// Poll interval override; defaults to (30 s).
public SiteEventLogFailureCountReporter(
Func failedWriteCountProvider,
ISiteHealthCollector collector,
ILogger logger,
TimeSpan? refreshInterval = null)
{
_failedWriteCountProvider = failedWriteCountProvider
?? throw new ArgumentNullException(nameof(failedWriteCountProvider));
_collector = collector ?? throw new ArgumentNullException(nameof(collector));
_logger = logger ?? throw new ArgumentNullException(nameof(logger));
_refreshInterval = refreshInterval ?? DefaultRefreshInterval;
}
/// Starts the background polling loop, running an immediate first probe before entering the timed cycle.
/// Cancellation token signalling host shutdown.
/// A task that represents the asynchronous operation.
public Task StartAsync(CancellationToken ct)
{
// Linked CTS lets StopAsync's cancellation AND the host's shutdown
// token both terminate the loop; either side firing aborts the
// pending Task.Delay.
var cts = CancellationTokenSource.CreateLinkedTokenSource(ct);
_cts = cts;
// Read Token on the caller's thread, not inside the lambda: the lambda runs
// whenever the thread pool gets to it, so a Dispose landing first would make
// the deferred _cts.Token read throw and fault the loop task the host awaits.
var token = cts.Token;
_loop = Task.Run(() => RunLoopAsync(token), CancellationToken.None);
return Task.CompletedTask;
}
private async Task RunLoopAsync(CancellationToken ct)
{
// First tick runs immediately so the very first health report after
// process start carries a real failure-count snapshot — without this
// the dashboard would show 0 for the first 30 s after a deploy even
// if failures had already accumulated.
SafeProbe();
while (!ct.IsCancellationRequested)
{
try
{
await Task.Delay(_refreshInterval, ct).ConfigureAwait(false);
}
catch (OperationCanceledException)
{
break;
}
SafeProbe();
}
}
private void SafeProbe()
{
try
{
var count = _failedWriteCountProvider();
_collector.SetSiteEventLogWriteFailures(count);
}
catch (Exception ex)
{
// Catch-all is deliberate: the hosted service must survive every
// class of probe failure so the next tick gets a chance. Mirrors
// SiteAuditBacklogReporter's "exception logged, not propagated" contract.
_logger.LogWarning(ex, "SiteEventLogFailureCountReporter probe failed; next tick will retry.");
}
}
/// Signals the polling loop to stop and waits for it to complete.
/// Cancellation token (not used; the internal CTS governs shutdown).
/// A task that represents the asynchronous operation.
public Task StopAsync(CancellationToken ct)
{
try
{
_cts?.Cancel();
}
catch (ObjectDisposedException)
{
// Stop-after-Dispose is a legal ordering; Dispose already cancelled the
// loop. Letting this escape would abort the host's shutdown sequence.
}
return _loop ?? Task.CompletedTask;
}
/// Releases the internal used to stop the polling loop.
public void Dispose()
{
// Cancel before disposing so the loop is always signalled even when the host
// disposes the container without having driven StopAsync first, and so the
// loop's pending Task.Delay(interval, token) sees an already-cancelled token
// rather than registering against a dead source.
try
{
_cts?.Cancel();
}
catch (ObjectDisposedException)
{
// Already disposed — Dispose is idempotent.
}
_cts?.Dispose();
}
}