Files
mxaccessgw/src/ZB.MOM.WW.MxGateway.Tests/Security/Authorization/ApiKeyFailureLimiterTests.cs
T
Joseph Doherty 5b681ee59b fix(SEC-31,SEC-32): identify a probe-slot reservation by version, not by timestamp
ReleaseProbe recognised its own reservation by comparing NextProbeAtTicks to
now + _probeIntervalTicks. RecordInto's rearm-on-trip writes that identical
expression, so a concurrent RecordFailure on the same WindowState whose `now`
lands on the claimer's tick — routine at ~1 ms clock resolution under load — was
mistaken for the caller's own claim. The release then stomped the legitimate
fresh re-arm back to the stale previousProbeAtTicks, which is already due, handing
the next arrival a free probe the re-arm had just closed.

WindowState gains a monotonic ProbeVersion bumped by every writer of
NextProbeAtTicks (TryConsumeProbe's claim and RecordInto's re-arm alike).
TryConsumeProbe returns the stamp it set as part of a ProbeClaim; ReleaseProbe
restores the previous value only while the state's version still equals that
stamp, checking and restoring in one lock(state) section and bumping the version
again on restore so no other stale release can match either.

Test: ProbeSlotRestore_DoesNotStompConcurrentRearmAtSameTick, with the clock held
still so the claim and the interleaved failure necessarily share a tick. Making it
deterministic needed a seam — the claim-to-release window is a few nanoseconds and
racing threads do not hit it (an earlier thread-based attempt passed against the
defective guard three runs out of three, and its end state was ordering-dependent
rather than correctness-dependent, so it was dropped rather than shipped as
theatre). The seam is an internal ProbeReleaseInterleaveHook, null in production,
costing one null check on the already-refused path. Verified as a genuine red
against the timestamp guard: Expected ThrottledByPeer, Actual ProbeAdmitted.
2026-08-07 06:10:14 -04:00

441 lines
20 KiB
C#

using ZB.MOM.WW.MxGateway.Server.Security.Authorization;
using ZB.MOM.WW.MxGateway.Tests.TestSupport;
namespace ZB.MOM.WW.MxGateway.Tests.Security.Authorization;
/// <summary>
/// Unit tests for the two-layer API-key failure limiter (SEC-31 / SEC-32): composite
/// <c>(transport peer, key id)</c> partitions, the cross-peer per-key-id aggregate, probe
/// admission, the per-peer key-id partition cap, and the eviction preference order.
/// </summary>
public sealed class ApiKeyFailureLimiterTests
{
private static readonly TimeSpan Window = TimeSpan.FromSeconds(60);
private static readonly TimeSpan ProbeInterval = TimeSpan.FromSeconds(5);
/// <summary>A partition below the failure limit is admitted without consulting a probe slot.</summary>
[Fact]
public void Check_BelowLimit_Allows()
{
ManualTimeProvider clock = new(DateTimeOffset.UnixEpoch);
ApiKeyFailureLimiter limiter = CreateLimiter(clock, limit: 3);
ApiKeyThrottlePartition partition = new("ipv4:10.0.0.1:1", "victim");
limiter.RecordFailure(partition);
limiter.RecordFailure(partition);
Assert.Equal(ApiKeyThrottleDecision.Allowed, limiter.Check(partition));
}
/// <summary>Reaching the limit inside the window throttles the composite partition.</summary>
[Fact]
public void Check_AtLimit_ThrottlesCompositePartition()
{
ManualTimeProvider clock = new(DateTimeOffset.UnixEpoch);
ApiKeyFailureLimiter limiter = CreateLimiter(clock, limit: 3);
ApiKeyThrottlePartition partition = new("ipv4:10.0.0.1:1", "victim");
RecordFailures(limiter, partition, 3);
Assert.Equal(ApiKeyThrottleDecision.ThrottledByPeer, limiter.Check(partition));
}
/// <summary>Failures older than the sliding window are pruned, releasing the throttle.</summary>
[Fact]
public void Window_PrunesExpiredFailures_ReleasesThrottle()
{
ManualTimeProvider clock = new(DateTimeOffset.UnixEpoch);
ApiKeyFailureLimiter limiter = CreateLimiter(clock, limit: 3);
ApiKeyThrottlePartition partition = new("ipv4:10.0.0.1:1", "victim");
RecordFailures(limiter, partition, 3);
Assert.Equal(ApiKeyThrottleDecision.ThrottledByPeer, limiter.Check(partition));
clock.Advance(Window + TimeSpan.FromSeconds(1));
Assert.Equal(ApiKeyThrottleDecision.Allowed, limiter.Check(partition));
}
/// <summary>
/// The composite partition binds a throttle to the failing address: the same key id presented
/// from a different transport peer is unaffected. This is the structural half of the SEC-31 fix.
/// </summary>
[Fact]
public void CompositePartition_ThrottleDoesNotFollowKeyIdToAnotherPeer()
{
ManualTimeProvider clock = new(DateTimeOffset.UnixEpoch);
ApiKeyFailureLimiter limiter = CreateLimiter(clock, limit: 3, aggregateLimit: 1000);
RecordFailures(limiter, new ApiKeyThrottlePartition("ipv4:10.0.0.1:1", "victim"), 3);
Assert.Equal(
ApiKeyThrottleDecision.ThrottledByPeer,
limiter.Check(new ApiKeyThrottlePartition("ipv4:10.0.0.1:1", "victim")));
Assert.Equal(
ApiKeyThrottleDecision.Allowed,
limiter.Check(new ApiKeyThrottlePartition("ipv4:10.0.0.2:1", "victim")));
}
/// <summary>
/// Failures for one key id spread across many peers trip the per-key aggregate layer, so a
/// rotating-source sprayer is still bounded even though no single composite partition trips.
/// </summary>
[Fact]
public void AggregateLayer_TripsAcrossDistinctPeers()
{
ManualTimeProvider clock = new(DateTimeOffset.UnixEpoch);
ApiKeyFailureLimiter limiter = CreateLimiter(clock, limit: 100, aggregateLimit: 5);
for (int peer = 0; peer < 5; peer++)
{
limiter.RecordFailure(new ApiKeyThrottlePartition($"ipv4:10.0.0.{peer}:1", "victim"));
}
Assert.Equal(
ApiKeyThrottleDecision.ThrottledByAggregate,
limiter.Check(new ApiKeyThrottlePartition("ipv4:10.0.9.9:1", "victim")));
// The aggregate is per key id: an unrelated key from the same fresh peer is untouched.
Assert.Equal(
ApiKeyThrottleDecision.Allowed,
limiter.Check(new ApiKeyThrottlePartition("ipv4:10.0.9.9:1", "other-key")));
}
/// <summary>
/// A throttled partition is a valve, not a wall: exactly one request per probe interval is
/// admitted to the real verifier, so the holder of the correct secret can always get through.
/// </summary>
[Fact]
public void ProbeAdmission_AdmitsOneRequestPerInterval()
{
ManualTimeProvider clock = new(DateTimeOffset.UnixEpoch);
ApiKeyFailureLimiter limiter = CreateLimiter(clock, limit: 3);
ApiKeyThrottlePartition partition = new("ipv4:10.0.0.1:1", "victim");
RecordFailures(limiter, partition, 3);
Assert.Equal(ApiKeyThrottleDecision.ThrottledByPeer, limiter.Check(partition));
clock.Advance(ProbeInterval);
Assert.Equal(ApiKeyThrottleDecision.ProbeAdmitted, limiter.Check(partition));
// The granted slot is consumed: the next request inside the same interval is throttled.
Assert.Equal(ApiKeyThrottleDecision.ThrottledByPeer, limiter.Check(partition));
clock.Advance(ProbeInterval);
Assert.Equal(ApiKeyThrottleDecision.ProbeAdmitted, limiter.Check(partition));
}
/// <summary>
/// The probe slot is claimed atomically: when a crowd of requests arrives at the same interval
/// boundary exactly one is admitted and the rest are still refused. A check-then-act grant would
/// let every arrival observe "due" and hand the whole burst through to the verifier.
/// </summary>
[Fact]
public void ProbeAdmission_UnderConcurrentArrivals_GrantsExactlyOneSlot()
{
ManualTimeProvider clock = new(DateTimeOffset.UnixEpoch);
ApiKeyFailureLimiter limiter = CreateLimiter(clock, limit: 3, aggregateLimit: 5);
ApiKeyThrottlePartition partition = new("ipv4:10.0.0.1:1", "victim");
// Five failures on one partition trip both layers (limit 3, aggregate 5), so the concurrent
// arrivals contend for the composite probe slot AND the aggregate's.
RecordFailures(limiter, partition, 5);
clock.Advance(ProbeInterval);
// The unsafe window between reading "probe due" and advancing the slot is nanoseconds wide,
// so one burst can miss it by luck. Repeating the boundary makes the red reliable, while the
// atomic implementation must yield exactly one admission in every round.
const int arrivals = 8;
const int rounds = 200;
int totalAdmitted = 0;
for (int round = 0; round < rounds; round++)
{
// Re-arm: the failures keep both layers over their limits, and recording pushes the next
// probe one interval out, which the advance below then reaches.
RecordFailures(limiter, partition, 5);
clock.Advance(ProbeInterval);
ApiKeyThrottleDecision[] decisions = new ApiKeyThrottleDecision[arrivals];
using (Barrier startLine = new(arrivals))
{
Thread[] threads = new Thread[arrivals];
for (int index = 0; index < arrivals; index++)
{
int slot = index;
threads[slot] = new Thread(() =>
{
startLine.SignalAndWait();
decisions[slot] = limiter.Check(partition);
});
threads[slot].Start();
}
foreach (Thread thread in threads)
{
Assert.True(thread.Join(TimeSpan.FromSeconds(30)), "probe-contention thread did not finish");
}
}
int admitted = decisions.Count(decision => decision == ApiKeyThrottleDecision.ProbeAdmitted);
Assert.Equal(
arrivals - admitted,
decisions.Count(decision => decision is ApiKeyThrottleDecision.ThrottledByPeer
or ApiKeyThrottleDecision.ThrottledByAggregate));
totalAdmitted += admitted;
}
Assert.Equal(rounds, totalAdmitted);
}
/// <summary>
/// The layers claim their probe slots one at a time, so a slot claimed on the composite partition
/// must be returned when the aggregate then refuses. Otherwise a refused request would silently
/// spend the partition's next slot and push the legitimate holder out by a full interval.
/// </summary>
[Fact]
public void ProbeAdmission_WhenAggregateRefuses_ReturnsTheClaimedPeerSlot()
{
ManualTimeProvider clock = new(DateTimeOffset.UnixEpoch);
ApiKeyFailureLimiter limiter = CreateLimiter(clock, limit: 3, aggregateLimit: 5);
ApiKeyThrottlePartition holder = new("ipv4:10.0.0.1:1", "victim");
// Both layers trip together, arming both slots for t0 + interval.
RecordFailures(limiter, holder, 5);
// A failure from a second address re-arms only the aggregate (its own composite is far under
// the limit), so the aggregate's slot now opens one interval later than the partition's.
clock.Advance(TimeSpan.FromSeconds(3));
limiter.RecordFailure(new ApiKeyThrottlePartition("ipv4:10.0.0.2:1", "victim"));
// t0 + 5s: the partition's slot is due, the aggregate's is not — the request is refused and
// the partition's slot must be handed back rather than consumed.
clock.Advance(TimeSpan.FromSeconds(2));
Assert.Equal(ApiKeyThrottleDecision.ThrottledByAggregate, limiter.Check(holder));
// t0 + 8s: the aggregate's slot opens. The partition's slot was restored, so this passes; had
// it been consumed above it would not reopen until t0 + 10s and this would be ThrottledByPeer.
clock.Advance(TimeSpan.FromSeconds(3));
Assert.Equal(ApiKeyThrottleDecision.ProbeAdmitted, limiter.Check(holder));
}
/// <summary>
/// A compensating release must never undo a re-arm written by a concurrent failure on the same
/// partition. Both writers store the identical <c>now + interval</c> value when they share a
/// clock tick, so identifying the caller's own reservation by timestamp would let the release
/// stomp a fresh re-arm back to an already-due value and reopen the probe slot early. The clock
/// is deliberately held still here, which forces exactly that collision.
/// </summary>
[Fact]
public void ProbeSlotRestore_DoesNotStompConcurrentRearmAtSameTick()
{
ManualTimeProvider clock = new(DateTimeOffset.UnixEpoch);
ApiKeyFailureLimiter limiter = CreateLimiter(clock, limit: 3, aggregateLimit: 5);
ApiKeyThrottlePartition holder = new("ipv4:10.0.0.1:1", "victim");
ApiKeyThrottlePartition other = new("ipv4:10.0.0.2:1", "victim");
// Trip both layers, then push the aggregate's slot one interval past the partition's, so
// every Check below claims the partition's slot and is then refused by the aggregate — the
// claim-and-compensate path under test.
RecordFailures(limiter, holder, 5);
clock.Advance(TimeSpan.FromSeconds(3));
limiter.RecordFailure(other);
clock.Advance(TimeSpan.FromSeconds(2));
// Land a failure on the same partition inside the claim-to-release window — the interleaving
// a concurrent RecordFailure produces, forced here so the assertion is deterministic. It
// shares the frozen clock tick with the claim, so both write the identical slot value.
int interleaved = 0;
limiter.ProbeReleaseInterleaveHook = () =>
{
if (Interlocked.Exchange(ref interleaved, 1) == 0)
{
limiter.RecordFailure(holder);
}
};
Assert.Equal(ApiKeyThrottleDecision.ThrottledByAggregate, limiter.Check(holder));
Assert.Equal(1, interleaved);
limiter.ProbeReleaseInterleaveHook = null;
// Drop the aggregate so the next decision reflects the composite partition alone.
limiter.Reset(other);
// The interleaved failure pushed the slot one interval past the (still unadvanced) clock, so
// no probe may be due. Restoring over it would leave the already-due earlier value and hand
// the next arrival a free probe.
Assert.Equal(ApiKeyThrottleDecision.ThrottledByPeer, limiter.Check(holder));
}
/// <summary>A zero probe interval restores absolute blocking (documented as not recommended).</summary>
[Fact]
public void ProbeIntervalZero_BlocksAbsolutely()
{
ManualTimeProvider clock = new(DateTimeOffset.UnixEpoch);
ApiKeyFailureLimiter limiter = CreateLimiter(clock, limit: 3, probeInterval: TimeSpan.Zero);
ApiKeyThrottlePartition partition = new("ipv4:10.0.0.1:1", "victim");
RecordFailures(limiter, partition, 3);
// Well past several probe intervals but still inside the failure window: no slot opens.
clock.Advance(ProbeInterval + ProbeInterval + ProbeInterval);
Assert.Equal(ApiKeyThrottleDecision.ThrottledByPeer, limiter.Check(partition));
}
/// <summary>A successful verification clears both the composite partition and the key aggregate.</summary>
[Fact]
public void Reset_ClearsCompositeAndAggregateLayers()
{
ManualTimeProvider clock = new(DateTimeOffset.UnixEpoch);
ApiKeyFailureLimiter limiter = CreateLimiter(clock, limit: 3, aggregateLimit: 5);
ApiKeyThrottlePartition partition = new("ipv4:10.0.0.1:1", "victim");
RecordFailures(limiter, partition, 3);
for (int peer = 1; peer < 3; peer++)
{
limiter.RecordFailure(new ApiKeyThrottlePartition($"ipv4:10.0.1.{peer}:1", "victim"));
}
Assert.Equal(ApiKeyThrottleDecision.ThrottledByPeer, limiter.Check(partition));
limiter.Reset(partition);
Assert.Equal(ApiKeyThrottleDecision.Allowed, limiter.Check(partition));
Assert.Equal(0, limiter.TrackedAggregateCount);
Assert.Equal(
ApiKeyThrottleDecision.Allowed,
limiter.Check(new ApiKeyThrottlePartition("ipv4:10.0.9.9:1", "victim")));
}
/// <summary>
/// A success whose key id was squeezed into the address's shared fallback bucket must not clear
/// that bucket: it carries failures from other key ids at the same address, so clearing it would
/// make one successful authentication a reset button for an in-progress spray.
/// </summary>
[Fact]
public void Reset_WithOverCapKeyId_DoesNotClearSharedFallbackPartition()
{
ManualTimeProvider clock = new(DateTimeOffset.UnixEpoch);
ApiKeyFailureLimiter limiter = CreateLimiter(clock, limit: 3, aggregateLimit: 100, maxPartitions: 4096);
const string peer = "ipv4:10.0.0.1:1";
// Fill the per-peer key-id cap, then spray past it so the overflow lands on — and trips —
// the address's shared fallback partition.
for (int i = 0; i < ApiKeyFailureLimiter.MaxKeyIdPartitionsPerPeer; i++)
{
limiter.RecordFailure(new ApiKeyThrottlePartition(peer, $"key{i}"));
}
for (int i = 0; i < 5; i++)
{
limiter.RecordFailure(new ApiKeyThrottlePartition(peer, $"overflow{i}"));
}
ApiKeyThrottlePartition overCap = new(peer, "overflow0");
Assert.Equal(ApiKeyThrottleDecision.ThrottledByPeer, limiter.Check(overCap));
limiter.Reset(overCap);
Assert.True(limiter.IsTracked(overCap));
Assert.Equal(ApiKeyThrottleDecision.ThrottledByPeer, limiter.Check(overCap));
}
/// <summary>
/// SEC-32: a spray of unique junk partitions (junk tokens resolve to the sender's fallback
/// partition, one per address) must not evict a partition that is currently throttled — the
/// LRU cap bounds memory, it must not be a reset button for the block.
/// </summary>
[Fact]
public void JunkTokenSpray_DoesNotEvictBlockedEntry()
{
ManualTimeProvider clock = new(DateTimeOffset.UnixEpoch);
ApiKeyFailureLimiter limiter = CreateLimiter(clock, limit: 3, maxPartitions: 8);
ApiKeyThrottlePartition blocked = new("ipv4:10.0.0.1:1", "victim");
RecordFailures(limiter, blocked, 3);
Assert.Equal(ApiKeyThrottleDecision.ThrottledByPeer, limiter.Check(blocked));
for (int i = 0; i < 16; i++)
{
clock.Advance(TimeSpan.FromMilliseconds(1));
limiter.RecordFailure(new ApiKeyThrottlePartition($"ipv4:10.9.{i / 256}.{i % 256}:1", KeyId: null));
}
Assert.True(limiter.IsTracked(blocked));
Assert.Equal(ApiKeyThrottleDecision.ThrottledByPeer, limiter.Check(blocked));
}
/// <summary>
/// SEC-32: one address may mint at most <see cref="ApiKeyFailureLimiter.MaxKeyIdPartitionsPerPeer"/>
/// key-id partitions; the overflow collapses into that address's fallback partition, which then
/// throttles the address wholesale instead of letting the spray mint unbounded state.
/// </summary>
[Fact]
public void UniqueMxgwKeyIdSpray_FromOnePeer_CollapsesAtPerPeerCap()
{
ManualTimeProvider clock = new(DateTimeOffset.UnixEpoch);
ApiKeyFailureLimiter limiter = CreateLimiter(clock, limit: 3, maxPartitions: 4096);
const string peer = "ipv4:10.0.0.1:1";
for (int i = 0; i < 1000; i++)
{
limiter.RecordFailure(new ApiKeyThrottlePartition(peer, $"key{i}"));
}
Assert.True(
limiter.TrackedPartitionCount <= ApiKeyFailureLimiter.MaxKeyIdPartitionsPerPeer + 1,
$"expected at most {ApiKeyFailureLimiter.MaxKeyIdPartitionsPerPeer + 1} partitions, saw {limiter.TrackedPartitionCount}");
Assert.Equal(
ApiKeyThrottleDecision.ThrottledByPeer,
limiter.Check(new ApiKeyThrottlePartition(peer, "key999")));
}
/// <summary>
/// SEC-32 eviction preference: with the map at capacity a new failure evicts an entry whose
/// window has fully expired rather than an entry that is still counting.
/// </summary>
[Fact]
public void Eviction_PrefersExpiredWindows()
{
ManualTimeProvider clock = new(DateTimeOffset.UnixEpoch);
ApiKeyFailureLimiter limiter = CreateLimiter(clock, limit: 3, maxPartitions: 2);
ApiKeyThrottlePartition expired = new("ipv4:10.0.0.1:1", KeyId: null);
ApiKeyThrottlePartition active = new("ipv4:10.0.0.2:1", KeyId: null);
ApiKeyThrottlePartition arriving = new("ipv4:10.0.0.3:1", KeyId: null);
limiter.RecordFailure(expired);
clock.Advance(Window + TimeSpan.FromSeconds(1));
limiter.RecordFailure(active);
limiter.RecordFailure(arriving);
Assert.Equal(2, limiter.TrackedPartitionCount);
Assert.False(limiter.IsTracked(expired));
Assert.True(limiter.IsTracked(active));
Assert.True(limiter.IsTracked(arriving));
}
private static void RecordFailures(ApiKeyFailureLimiter limiter, ApiKeyThrottlePartition partition, int count)
{
for (int i = 0; i < count; i++)
{
limiter.RecordFailure(partition);
}
}
private static ApiKeyFailureLimiter CreateLimiter(
TimeProvider clock,
int limit,
int aggregateLimit = 0,
int maxPartitions = 1024,
TimeSpan? probeInterval = null)
{
return new ApiKeyFailureLimiter(
limit,
Window,
maxPartitions,
aggregateLimit,
probeInterval ?? ProbeInterval,
clock);
}
}