refactor: rename ScadaLink → ZB.MOM.WW.ScadaBridge (code + projects + namespaces)
Solution + 23 src projects + 26 test projects renamed; folders, csproj, namespaces, and ScadaLinkDbContext/ScadaBridgeDbContext class updated. ActorSystem "scadalink" → "scadabridge", Akka seed-node URLs migrated. SQL roles/logins, LDAP domains, CLI command name, and CLI config dir (~/.scadalink → ~/.scadabridge) also renamed. Build green; 5 Host.Tests fail awaiting SQL login rename in next commit. Pre-existing StaleTagMonitor timing flakes unchanged. Rename script committed at tools/rename-to-scadabridge.sh.
This commit is contained in:
@@ -0,0 +1,334 @@
|
||||
using System.Threading.Channels;
|
||||
using Microsoft.Data.Sqlite;
|
||||
using Microsoft.Extensions.Logging;
|
||||
using Microsoft.Extensions.Options;
|
||||
|
||||
namespace ZB.MOM.WW.ScadaBridge.SiteEventLogging;
|
||||
|
||||
/// <summary>
|
||||
/// Records operational events to a local SQLite database.
|
||||
/// Only the active node generates events. Not replicated to standby.
|
||||
/// On failover, the new active node starts a fresh log.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// <para>
|
||||
/// A single <see cref="SqliteConnection"/> is owned here and is NOT thread-safe.
|
||||
/// All access — recording, querying, purging — must be funnelled through
|
||||
/// <see cref="WithConnection"/>, which serialises callers on a shared lock.
|
||||
/// </para>
|
||||
/// <para>
|
||||
/// Event recording is offloaded to a dedicated background writer thread (fed by a
|
||||
/// <em>bounded</em> <see cref="Channel{T}"/>; capacity <see cref="SiteEventLogOptions.WriteQueueCapacity"/>,
|
||||
/// default 10 000, overflow <see cref="BoundedChannelFullMode.DropOldest"/>).
|
||||
/// <see cref="LogEventAsync"/> only validates its arguments and enqueues, so callers —
|
||||
/// typically Akka actor threads on hot paths — never block on disk I/O or on
|
||||
/// contention for the write lock. The returned <see cref="Task"/> completes once the
|
||||
/// event is durably persisted and faults if the write fails. SiteEventLogging-015:
|
||||
/// when a queued event is evicted to make room for a newer one, that event's Task
|
||||
/// is faulted with <see cref="InvalidOperationException"/> and
|
||||
/// <see cref="FailedWriteCount"/> is incremented so the drop is observable.
|
||||
/// </para>
|
||||
/// </remarks>
|
||||
public class SiteEventLogger : ISiteEventLogger, IDisposable
|
||||
{
|
||||
private readonly SqliteConnection _connection;
|
||||
private readonly ILogger<SiteEventLogger> _logger;
|
||||
private readonly object _writeLock = new();
|
||||
private readonly Channel<PendingEvent> _writeQueue;
|
||||
private readonly Task _writerLoop;
|
||||
private long _failedWriteCount;
|
||||
private bool _disposed;
|
||||
|
||||
/// <summary>
|
||||
/// Initializes the event logger, opens the SQLite connection, and starts the background writer loop.
|
||||
/// </summary>
|
||||
/// <param name="options">Site event log configuration (database path, retention settings).</param>
|
||||
/// <param name="logger">Logger for write-failure diagnostics.</param>
|
||||
/// <param name="connectionStringOverride">Optional connection string override; uses the configured path when null.</param>
|
||||
public SiteEventLogger(
|
||||
IOptions<SiteEventLogOptions> options,
|
||||
ILogger<SiteEventLogger> logger,
|
||||
string? connectionStringOverride = null)
|
||||
{
|
||||
_logger = logger;
|
||||
|
||||
// SiteEventLogging-022: Cache=Shared is a cross-connection optimisation
|
||||
// that lets multiple SqliteConnections share an in-process page cache.
|
||||
// This logger owns exactly one SqliteConnection and serialises all
|
||||
// access through _writeLock, so the mode is dormant — at best dead
|
||||
// configuration, at worst a small future foot-gun for any second
|
||||
// connection opened to the same file. A test path that genuinely
|
||||
// needs Cache=Shared can still inject it via connectionStringOverride.
|
||||
var connectionString = connectionStringOverride
|
||||
?? $"Data Source={options.Value.DatabasePath}";
|
||||
_connection = new SqliteConnection(connectionString);
|
||||
_connection.Open();
|
||||
|
||||
InitializeSchema();
|
||||
|
||||
// SiteEventLogging-015: bounded queue with DropOldest preserves the
|
||||
// "callers never block" guarantee (SiteEventLogging-005) while putting an
|
||||
// upper bound on memory under sustained writer slowness. Drops are
|
||||
// observable — itemDropped faults the evicted Task and increments
|
||||
// FailedWriteCount.
|
||||
var capacity = Math.Max(1, options.Value.WriteQueueCapacity);
|
||||
_writeQueue = Channel.CreateBounded<PendingEvent>(
|
||||
new BoundedChannelOptions(capacity)
|
||||
{
|
||||
SingleReader = true,
|
||||
SingleWriter = false,
|
||||
FullMode = BoundedChannelFullMode.DropOldest,
|
||||
},
|
||||
itemDropped: dropped =>
|
||||
{
|
||||
Interlocked.Increment(ref _failedWriteCount);
|
||||
dropped.Completion.TrySetException(
|
||||
new InvalidOperationException(
|
||||
$"Event was dropped because the write queue exceeded its bounded capacity ({capacity})."));
|
||||
});
|
||||
_writerLoop = Task.Run(ProcessWriteQueueAsync);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// SiteEventLogging-018: number of event writes that have failed (SQLite
|
||||
/// error, disk full, bounded-queue overflow drop, etc.) since this logger
|
||||
/// was created. Available for future Health Monitoring integration — the
|
||||
/// counter is correct and observable, but the central health-metric
|
||||
/// pipeline does not yet poll it, so a sustained non-zero value currently
|
||||
/// goes unnoticed in production beyond the per-failure log line. Wiring
|
||||
/// the metric into the 30-second site-metric publish is tracked
|
||||
/// separately; promoted to <see cref="ISiteEventLogger"/> so the eventual
|
||||
/// consumer reads it without a concrete-type downcast.
|
||||
/// </summary>
|
||||
public long FailedWriteCount => Interlocked.Read(ref _failedWriteCount);
|
||||
|
||||
/// <summary>
|
||||
/// Runs <paramref name="action"/> against the shared connection while holding the
|
||||
/// write lock, so purge / query / record callers on different threads never use
|
||||
/// the non-thread-safe <see cref="SqliteConnection"/> concurrently.
|
||||
/// Returns <see langword="false"/> without invoking the action if the logger has
|
||||
/// been disposed.
|
||||
/// </summary>
|
||||
/// <param name="action">The action to run against the shared connection.</param>
|
||||
internal bool WithConnection(Action<SqliteConnection> action)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(action);
|
||||
lock (_writeLock)
|
||||
{
|
||||
if (_disposed) return false;
|
||||
action(_connection);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Runs <paramref name="func"/> against the shared connection while holding the
|
||||
/// write lock. Throws <see cref="ObjectDisposedException"/> if the logger has
|
||||
/// been disposed (callers that need a result cannot proceed without the database).
|
||||
/// </summary>
|
||||
/// <typeparam name="T">The return type of the function.</typeparam>
|
||||
/// <param name="func">The function to run against the shared connection.</param>
|
||||
internal T WithConnection<T>(Func<SqliteConnection, T> func)
|
||||
{
|
||||
ArgumentNullException.ThrowIfNull(func);
|
||||
lock (_writeLock)
|
||||
{
|
||||
ObjectDisposedException.ThrowIf(_disposed, this);
|
||||
return func(_connection);
|
||||
}
|
||||
}
|
||||
|
||||
private void InitializeSchema()
|
||||
{
|
||||
// auto_vacuum must be set before any table is created for it to take effect
|
||||
// on a fresh database. With INCREMENTAL mode, PRAGMA incremental_vacuum can
|
||||
// later reclaim free pages so the storage-cap purge can shrink the file.
|
||||
using (var pragmaCmd = _connection.CreateCommand())
|
||||
{
|
||||
pragmaCmd.CommandText = "PRAGMA auto_vacuum = INCREMENTAL";
|
||||
pragmaCmd.ExecuteNonQuery();
|
||||
}
|
||||
|
||||
using var cmd = _connection.CreateCommand();
|
||||
cmd.CommandText = """
|
||||
CREATE TABLE IF NOT EXISTS site_events (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
timestamp TEXT NOT NULL,
|
||||
event_type TEXT NOT NULL,
|
||||
severity TEXT NOT NULL,
|
||||
instance_id TEXT,
|
||||
source TEXT NOT NULL,
|
||||
message TEXT NOT NULL,
|
||||
details TEXT
|
||||
);
|
||||
CREATE INDEX IF NOT EXISTS idx_events_timestamp ON site_events(timestamp);
|
||||
CREATE INDEX IF NOT EXISTS idx_events_type ON site_events(event_type);
|
||||
CREATE INDEX IF NOT EXISTS idx_events_instance ON site_events(instance_id);
|
||||
CREATE INDEX IF NOT EXISTS idx_events_severity ON site_events(severity);
|
||||
""";
|
||||
// The query service also supports keyword search via leading-wildcard
|
||||
// LIKE on message/source. A leading-wildcard LIKE cannot use a B-tree
|
||||
// index, so that path intentionally full-scans; severity/event_type/
|
||||
// instance_id/timestamp filters above are all covered.
|
||||
cmd.ExecuteNonQuery();
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// SiteEventLogging-020: closed set of allowed severities. Case-sensitive to
|
||||
/// match the SQLite default <c>BINARY</c> collation used by the query filter —
|
||||
/// a row stored as <c>"error"</c> would be invisible to a query filtering on
|
||||
/// <c>"Error"</c>, so the contract on the way in must match the contract on
|
||||
/// the way out.
|
||||
/// </summary>
|
||||
private static readonly HashSet<string> AllowedSeverities =
|
||||
new(StringComparer.Ordinal) { "Info", "Warning", "Error" };
|
||||
|
||||
/// <inheritdoc />
|
||||
public Task LogEventAsync(
|
||||
string eventType,
|
||||
string severity,
|
||||
string? instanceId,
|
||||
string source,
|
||||
string message,
|
||||
string? details = null)
|
||||
{
|
||||
ArgumentException.ThrowIfNullOrWhiteSpace(eventType);
|
||||
ArgumentException.ThrowIfNullOrWhiteSpace(severity);
|
||||
ArgumentException.ThrowIfNullOrWhiteSpace(source);
|
||||
ArgumentException.ThrowIfNullOrWhiteSpace(message);
|
||||
|
||||
// SiteEventLogging-020: reject unknown severities so the query-time filter
|
||||
// (case-sensitive BINARY collation) and the documented enum stay in sync.
|
||||
if (!AllowedSeverities.Contains(severity))
|
||||
{
|
||||
throw new ArgumentException(
|
||||
$"Severity '{severity}' is not one of the allowed values: Info, Warning, Error.",
|
||||
nameof(severity));
|
||||
}
|
||||
|
||||
var pending = new PendingEvent(
|
||||
DateTimeOffset.UtcNow.ToString("o"),
|
||||
eventType,
|
||||
severity,
|
||||
instanceId,
|
||||
source,
|
||||
message,
|
||||
details);
|
||||
|
||||
// Enqueue only — the actual SQLite write happens on the background writer
|
||||
// thread so the caller (an Akka actor thread on a hot path) never blocks
|
||||
// on disk I/O or on contention for the write lock.
|
||||
if (!_writeQueue.Writer.TryWrite(pending))
|
||||
{
|
||||
// The channel is unbounded, so the only way TryWrite fails is that the
|
||||
// writer has been completed (logger disposed). The event cannot be
|
||||
// persisted — fault the Task (SiteEventLogging-012) rather than
|
||||
// reporting false success, so a caller that awaits a critical audit
|
||||
// event can tell it was dropped.
|
||||
pending.Completion.TrySetException(
|
||||
new ObjectDisposedException(nameof(SiteEventLogger),
|
||||
"Event could not be recorded: the event logger has been disposed."));
|
||||
}
|
||||
|
||||
return pending.Completion.Task;
|
||||
}
|
||||
|
||||
private async Task ProcessWriteQueueAsync()
|
||||
{
|
||||
await foreach (var pending in _writeQueue.Reader.ReadAllAsync().ConfigureAwait(false))
|
||||
{
|
||||
try
|
||||
{
|
||||
var written = WithConnection(connection =>
|
||||
{
|
||||
using var cmd = connection.CreateCommand();
|
||||
cmd.CommandText = """
|
||||
INSERT INTO site_events (timestamp, event_type, severity, instance_id, source, message, details)
|
||||
VALUES ($timestamp, $event_type, $severity, $instance_id, $source, $message, $details)
|
||||
""";
|
||||
cmd.Parameters.AddWithValue("$timestamp", pending.Timestamp);
|
||||
cmd.Parameters.AddWithValue("$event_type", pending.EventType);
|
||||
cmd.Parameters.AddWithValue("$severity", pending.Severity);
|
||||
cmd.Parameters.AddWithValue("$instance_id", (object?)pending.InstanceId ?? DBNull.Value);
|
||||
cmd.Parameters.AddWithValue("$source", pending.Source);
|
||||
cmd.Parameters.AddWithValue("$message", pending.Message);
|
||||
cmd.Parameters.AddWithValue("$details", (object?)pending.Details ?? DBNull.Value);
|
||||
cmd.ExecuteNonQuery();
|
||||
});
|
||||
|
||||
if (written)
|
||||
{
|
||||
pending.Completion.TrySetResult();
|
||||
}
|
||||
else
|
||||
{
|
||||
// WithConnection returns false only when the logger has been
|
||||
// disposed mid-drain; the event was not persisted. Fault the
|
||||
// Task (SiteEventLogging-012) instead of reporting false
|
||||
// success for a dropped audit event.
|
||||
pending.Completion.TrySetException(
|
||||
new ObjectDisposedException(nameof(SiteEventLogger),
|
||||
"Event could not be recorded: the event logger was disposed before the write completed."));
|
||||
}
|
||||
}
|
||||
catch (Exception ex)
|
||||
{
|
||||
// SiteEventLogging-008: a write failure must be observable. Count it
|
||||
// (Health Monitoring reads FailedWriteCount) and fault the caller's
|
||||
// Task instead of silently discarding the exception.
|
||||
Interlocked.Increment(ref _failedWriteCount);
|
||||
_logger.LogError(ex, "Failed to record event: {EventType} from {Source}",
|
||||
pending.EventType, pending.Source);
|
||||
pending.Completion.TrySetException(ex);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Stops accepting new events, drains the write queue, and disposes the SQLite connection.
|
||||
/// </summary>
|
||||
public void Dispose()
|
||||
{
|
||||
Task? writerLoop = null;
|
||||
lock (_writeLock)
|
||||
{
|
||||
if (_disposed) return;
|
||||
_disposed = true;
|
||||
// Stop accepting new events and let the writer loop drain.
|
||||
_writeQueue.Writer.TryComplete();
|
||||
writerLoop = _writerLoop;
|
||||
}
|
||||
|
||||
// Wait for the writer loop to finish outside the lock — the loop itself
|
||||
// acquires the lock for each write.
|
||||
try
|
||||
{
|
||||
writerLoop?.Wait(TimeSpan.FromSeconds(5));
|
||||
}
|
||||
catch (AggregateException)
|
||||
{
|
||||
// A faulted writer loop has already been logged per event; nothing more
|
||||
// to do during disposal.
|
||||
}
|
||||
|
||||
lock (_writeLock)
|
||||
{
|
||||
_connection.Dispose();
|
||||
}
|
||||
}
|
||||
|
||||
/// <summary>An event awaiting persistence by the background writer.</summary>
|
||||
private sealed record PendingEvent(
|
||||
string Timestamp,
|
||||
string EventType,
|
||||
string Severity,
|
||||
string? InstanceId,
|
||||
string Source,
|
||||
string Message,
|
||||
string? Details)
|
||||
{
|
||||
/// <summary>Completes when the event has been durably persisted, or faults on write failure.</summary>
|
||||
public TaskCompletionSource Completion { get; } =
|
||||
new(TaskCreationOptions.RunContinuationsAsynchronously);
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user