From 48c9cd5893396d8d7c466be9f31bcde652d1d867 Mon Sep 17 00:00:00 2001 From: zacala Date: Sat, 3 Oct 2026 17:57:37 +0000 Subject: [PATCH 01/29] Fix lock ownership on Linux, lock-wait robustness, and generic struct support Code review findings, fixed: * Linux: a live lock owner was reported as an orphan, so any waiter stole its write lock. Process.StartTime is derived per process from a wall-clock boot-time snapshot, so the value an owner records and the value another process computes for it differ by milliseconds and never compare equal. Record and compare /proc//stat starttime (kernel ticks) instead. Values written by 3.0.0 (negative DateTime binaries) are treated as unknown and fall back to the PID-only check. * Orphan recovery only probed the owner at the start of a wait and at 75% of the timeout, so a wait with Timeout.InfiniteTimeSpan never recovered from an owner that died later. Probe every 250 ms instead. * Dispose() unmapped the header while another thread was still spinning in TryAcquireWriteLock/TryAcquireReadLock, which is a process-fatal AccessViolationException. Lock waiters now register with Dispose(), leave with ObjectDisposedException, and Dispose() waits for them before unmapping. * ReleaseWriteLock silently ignored a release from a non-owner thread, so `await` inside a lock scope left the lock held forever with no diagnostic. It now throws SynchronizationLockException, as do the StructuredMemory WriteLock/ReadLock guards when disposed on another thread (before touching any state, so the owning thread can still release correctly). * TypeLayoutFingerprint used Marshal.SizeOf/OffsetOf, which reject generic types, so ValueTuple/KeyValuePair<,> failed in every typed container. Fall back to a managed-layout descriptor for those types. Fingerprints of types that already worked are byte-for-byte unchanged (pinned by a test), so 3.0.0 and newer processes can still share regions. Tests: cross-process regression tests (live owner, infinite and finite wait recovery) with a new hold_write_lock worker role, Dispose-with-waiter, guard-on-another-thread, generic struct and pinned fingerprint tests. Two tests that asserted the old "silently ignored" behaviour now expect the exception. --- InterprocessMemory.TestWorker/Program.cs | 16 ++ .../ChangeVerificationTests.cs | 15 +- InterprocessMemory.Tests/CrossProcessTests.cs | 119 +++++++++++ .../LibraryHardeningTests.cs | 117 ++++++++++- .../TypeLayoutFingerprintTests.cs | 92 +++++++++ InterprocessMemory/IMemoryRegion.cs | 12 +- InterprocessMemory/MemoryRegion.cs | 188 +++++++++++++++--- InterprocessMemory/StructuredMemory.cs | 67 ++++++- InterprocessMemory/TimeoutHelper.cs | 7 - InterprocessMemory/TypeLayoutFingerprint.cs | 77 ++++++- README.md | 13 +- 11 files changed, 667 insertions(+), 56 deletions(-) create mode 100644 InterprocessMemory.Tests/TypeLayoutFingerprintTests.cs diff --git a/InterprocessMemory.TestWorker/Program.cs b/InterprocessMemory.TestWorker/Program.cs index e5e48f4..07e5dcd 100644 --- a/InterprocessMemory.TestWorker/Program.cs +++ b/InterprocessMemory.TestWorker/Program.cs @@ -18,6 +18,7 @@ /// concurrent_producer <name> <producerId> — enqueue 1000 unique integers /// try_write_lock <name> — try the cross-process write lock for 250 ms /// orphan_write_lock <name> — acquire a write lock and exit without releasing it +/// hold_write_lock <name> — acquire a write lock, print "holding", and keep it until killed /// if (args.Length < 2) { @@ -43,6 +44,7 @@ args.Length >= 3 ? int.Parse(args[2]) : 0), "try_write_lock" => TryWriteLock(bufferName), "orphan_write_lock" => OrphanWriteLock(bufferName), + "hold_write_lock" => HoldWriteLock(bufferName), _ => Error($"Unknown role: {role}") }; @@ -197,6 +199,20 @@ static int OrphanWriteLock(string name) return 0; // Deliberately skip Dispose/Release; process teardown closes only the mapping handle. } +static int HoldWriteLock(string name) +{ + using var region = MemoryRegion.OpenExisting(name); + if (!region.TryAcquireWriteLock(TimeSpan.FromSeconds(5))) + return Error("failed to acquire held test lock"); + + // The parent reads this line, probes the lock while we are alive, and then kills us + // to simulate a crash while the lock is held. + Console.WriteLine("holding"); + Console.Out.Flush(); + Thread.Sleep(Timeout.Infinite); + return 0; +} + static int Error(string msg) { Console.Error.WriteLine(msg); diff --git a/InterprocessMemory.Tests/ChangeVerificationTests.cs b/InterprocessMemory.Tests/ChangeVerificationTests.cs index 803c093..222eddc 100644 --- a/InterprocessMemory.Tests/ChangeVerificationTests.cs +++ b/InterprocessMemory.Tests/ChangeVerificationTests.cs @@ -1041,18 +1041,19 @@ public void Audit4_FieldDefinition_NullOrEmptyName_AllFactoriesReject() // ── AUDIT-5: ReleaseWriteLock CAS-by-owner ─────────────────────────────── [Test] - public void Audit5_ReleaseWriteLock_WithoutAcquire_IsNoOp() + public void Audit5_ReleaseWriteLock_WithoutAcquire_Throws() { - // Caller bug: releasing a lock that wasn't acquired by this process. Without the - // owner-CAS guard, ReleaseWriteLock would zero ownership metadata and free the lock — - // dangerous when another process legitimately holds it. With the fix it's a logged no-op. + // Caller bug: releasing a lock that wasn't acquired by this thread. Without the owner + // guard, ReleaseWriteLock would zero ownership metadata and free the lock — dangerous + // when another process legitimately holds it. The guard leaves the lock state untouched + // and reports the misuse instead of hiding it. using var buf = new MemoryRegion(N("Audit5_NoAcquire"), new MemoryRegionOptions { Capacity = 4096 }); - // No acquire here. Release should be a safe no-op (logs a warning). - Assert.DoesNotThrow(() => buf.ReleaseWriteLock()); + // No acquire here. + Assert.Throws(() => buf.ReleaseWriteLock()); - // Now actually acquire — must still work normally after the no-op release. + // Now actually acquire — must still work normally after the rejected release. Assert.That(buf.TryAcquireWriteLock(TimeSpan.FromSeconds(1)), Is.True); buf.ReleaseWriteLock(); } diff --git a/InterprocessMemory.Tests/CrossProcessTests.cs b/InterprocessMemory.Tests/CrossProcessTests.cs index 60e05f5..46dd7bd 100644 --- a/InterprocessMemory.Tests/CrossProcessTests.cs +++ b/InterprocessMemory.Tests/CrossProcessTests.cs @@ -293,6 +293,125 @@ public void CrossProcess_OrphanWriteLock_IsRecovered() region.ReleaseWriteLock(); } + /// + /// Starts a child that holds the write lock until it is killed, and returns once the child + /// has reported that it holds the lock. + /// + private static Process StartLockHolder(string bufferName) + { + Process process = Process.Start(CreateHelperStartInfo("hold_write_lock", bufferName))!; + string? line = process.StandardOutput.ReadLine(); + if (line != "holding") + { + try + { process.Kill(); } + catch (InvalidOperationException) { /* already exited */ } + process.Dispose(); + Assert.Fail($"lock holder did not report 'holding' (got '{line}')"); + } + + return process; + } + + /// Kills the process (a no-op when it has already exited) and waits for it to be gone. + private static void KillAndWait(Process process) + { + try + { process.Kill(); } + catch (InvalidOperationException) { /* already exited */ } + process.WaitForExit(); + } + + [Test, Timeout(30000)] + public void CrossProcess_LiveWriteLockOwner_IsNotReportedAsOrphan() + { + // Process.StartTime is derived per observing process from a wall-clock boot-time snapshot on + // Linux, so comparing the owner's recorded value with the observer's value reported every + // live owner as an impostor and let waiters steal its lock. + string name = GetUniqueName("LiveOwner"); + using var region = MemoryRegion.CreateOrOpen(name, 256); + using Process holder = StartLockHolder(name); + try + { + for (int i = 0; i < 50; i++) + { + Assert.That(region.IsWriteLockOrphaned(), Is.False, $"probe {i}"); + Thread.Sleep(10); + } + + Assert.That(region.TryAcquireWriteLock(TimeSpan.FromMilliseconds(300)), Is.False, + "a live owner's lock must not be taken over"); + } + finally + { + KillAndWait(holder); + } + } + + [Test, Timeout(30000)] + public void CrossProcess_InfiniteWait_RecoversWhenOwnerProcessDies() + { + string name = GetUniqueName("InfiniteOrphan"); + using var region = MemoryRegion.CreateOrOpen(name, 256); + using Process holder = StartLockHolder(name); + try + { + // Release on the acquiring thread: the write lock is thread-affine. + Task waiter = Task.Run(() => + { + bool acquired = region.TryAcquireWriteLock(Timeout.InfiniteTimeSpan); + if (acquired) + region.ReleaseWriteLock(); + return acquired; + }); + + // Give the waiter time to run its first orphan probe while the owner is still alive. + Thread.Sleep(500); + Assert.That(waiter.IsCompleted, Is.False, "the owner is alive, so the waiter must still be waiting"); + + KillAndWait(holder); + + Assert.That(waiter.Wait(TimeSpan.FromSeconds(10)), Is.True, + "an infinite wait must keep probing the owner and recover once it has died"); + Assert.That(waiter.Result, Is.True); + } + finally + { + KillAndWait(holder); + } + } + + [Test, Timeout(40000)] + public void CrossProcess_FiniteWait_RecoversPromptlyWhenOwnerDies() + { + // The orphan probe used to run only at the start of the wait and at 75% of the timeout, so + // with a 30 s timeout an owner that died after one second was noticed after ~22 s. + string name = GetUniqueName("FiniteOrphan"); + using var region = MemoryRegion.CreateOrOpen(name, 256); + using Process holder = StartLockHolder(name); + try + { + Task waiter = Task.Run(() => + { + bool acquired = region.TryAcquireWriteLock(TimeSpan.FromSeconds(30)); + if (acquired) + region.ReleaseWriteLock(); + return acquired; + }); + + Thread.Sleep(500); + KillAndWait(holder); + + Assert.That(waiter.Wait(TimeSpan.FromSeconds(5)), Is.True, + "recovery must not wait for most of the lock timeout"); + Assert.That(waiter.Result, Is.True); + } + finally + { + KillAndWait(holder); + } + } + // ── Schema ─────────────────────────────────────────────────────────────── public struct IpcTestSchema : IMemorySchema diff --git a/InterprocessMemory.Tests/LibraryHardeningTests.cs b/InterprocessMemory.Tests/LibraryHardeningTests.cs index a7d075c..4b6d6ea 100644 --- a/InterprocessMemory.Tests/LibraryHardeningTests.cs +++ b/InterprocessMemory.Tests/LibraryHardeningTests.cs @@ -82,10 +82,21 @@ public void WriteLock_CannotBeReleasedByDifferentThread() Assert.That(buffer.TryAcquireWriteLock(TimeSpan.FromSeconds(1)), Is.True); - var invalidRelease = new Thread(buffer.ReleaseWriteLock); + // An unhandled exception on a raw Thread would take the test host down, so capture it. + Exception? releaseError = null; + var invalidRelease = new Thread(() => + { + try + { buffer.ReleaseWriteLock(); } + catch (Exception ex) { releaseError = ex; } + }); invalidRelease.Start(); invalidRelease.Join(); + // The misuse is reported instead of being silently ignored (a silently ignored release is + // what turned `await` inside a lock scope into a lock nobody could ever release). + Assert.That(releaseError, Is.InstanceOf()); + var contender = Task.Run(() => buffer.TryAcquireWriteLock(TimeSpan.FromMilliseconds(50))); Assert.That(contender.Result, Is.False); @@ -94,6 +105,110 @@ public void WriteLock_CannotBeReleasedByDifferentThread() buffer.ReleaseWriteLock(); } + [Test] + public void Dispose_WhileThreadWaitsForWriteLock_ReleasesWaiterInsteadOfCrashing() + { + // Dispose used to unmap the view while a thread was still spinning on the header, which is an + // AccessViolationException: the whole process dies and the exception cannot be caught. + var buffer = new MemoryRegion( + N("DisposeWriteWaiter"), + new MemoryRegionOptions { Capacity = 256 }); + + Assert.That(buffer.TryAcquireWriteLock(TimeSpan.FromSeconds(1)), Is.True); + + using var started = new ManualResetEventSlim(false); + var waiter = Task.Run(() => + { + started.Set(); + return buffer.TryAcquireWriteLock(Timeout.InfiniteTimeSpan); + }); + Assert.That(started.Wait(TimeSpan.FromSeconds(5)), Is.True); + Thread.Sleep(100); // let the waiter reach its spin loop + + buffer.Dispose(); + + var error = Assert.Throws(() => waiter.Wait(TimeSpan.FromSeconds(10))); + Assert.That(error!.InnerException, Is.InstanceOf()); + } + + [Test] + public void Dispose_WhileThreadWaitsForReadLock_ReleasesWaiterInsteadOfCrashing() + { + var buffer = new MemoryRegion( + N("DisposeReadWaiter"), + new MemoryRegionOptions { Capacity = 256 }); + + Assert.That(buffer.TryAcquireWriteLock(TimeSpan.FromSeconds(1)), Is.True); + + using var started = new ManualResetEventSlim(false); + var waiter = Task.Run(() => + { + started.Set(); + return buffer.TryAcquireReadLock(Timeout.InfiniteTimeSpan); + }); + Assert.That(started.Wait(TimeSpan.FromSeconds(5)), Is.True); + Thread.Sleep(100); + + buffer.Dispose(); + + var error = Assert.Throws(() => waiter.Wait(TimeSpan.FromSeconds(10))); + Assert.That(error!.InnerException, Is.InstanceOf()); + } + + [Test] + public void StructuredMemory_WriteLockGuard_DisposedOnAnotherThread_ThrowsAndStaysHeld() + { + string name = N("WriteGuardThread"); + using var memory = StructuredMemory.CreateOrOpen(name, new SimpleSchema()); + using var peer = StructuredMemory.OpenExisting(name, new SimpleSchema()); + + // What `await` inside a lock scope does: the guard is disposed on a different thread. + var guard = memory.AcquireWriteLock(); + Exception? error = null; + Task.Run(() => + { + try + { guard.Dispose(); } + catch (Exception ex) { error = ex; } + }).Wait(); + + Assert.That(error, Is.InstanceOf()); + + // Nothing was released or unbalanced: the lock still belongs to the acquiring thread... + Assert.Throws(() => peer.AcquireWriteLock(TimeSpan.FromMilliseconds(100))); + + // ...which can still release it properly, after which other holders get in. + guard.Dispose(); + using (peer.AcquireWriteLock(TimeSpan.FromSeconds(1))) + { + } + } + + [Test] + public void StructuredMemory_ReadLockGuard_DisposedOnAnotherThread_ThrowsAndStaysHeld() + { + string name = N("ReadGuardThread"); + using var memory = StructuredMemory.CreateOrOpen(name, new SimpleSchema()); + using var peer = StructuredMemory.OpenExisting(name, new SimpleSchema()); + + var guard = memory.AcquireReadLock(); + Exception? error = null; + Task.Run(() => + { + try + { guard.Dispose(); } + catch (Exception ex) { error = ex; } + }).Wait(); + + Assert.That(error, Is.InstanceOf()); + Assert.Throws(() => peer.AcquireWriteLock(TimeSpan.FromMilliseconds(100))); + + guard.Dispose(); + using (peer.AcquireWriteLock(TimeSpan.FromSeconds(1))) + { + } + } + [Test] public void ReadLock_DoubleRelease_DoesNotBreakWriterExclusion() { diff --git a/InterprocessMemory.Tests/TypeLayoutFingerprintTests.cs b/InterprocessMemory.Tests/TypeLayoutFingerprintTests.cs new file mode 100644 index 0000000..d475f3d --- /dev/null +++ b/InterprocessMemory.Tests/TypeLayoutFingerprintTests.cs @@ -0,0 +1,92 @@ +using System.Collections.Generic; +using System.IO; +using NUnit.Framework; +using InterprocessMemory; + +namespace InterprocessMemory.Tests; + +[TestFixture] +public class TypeLayoutFingerprintTests +{ + private static string N(string prefix) => $"Fingerprint_{prefix}_{Guid.NewGuid():N}"; + + [Test] + public void MarshallableTypes_KeepTheirRelease300Fingerprint() + { + // These values were produced by 3.0.0. A different value would make a process built from + // this version reject regions created by a 3.0.0 process (and vice versa) with + // "different format or element type", so they must only ever change together with the + // shared-memory format version. + Assert.That(TypeLayoutFingerprint.Create(), + Is.EqualTo(new TypeLayoutFingerprint(0xD4616E4290E3AB20, 0xEB0DEB962578DCD5))); + Assert.That(TypeLayoutFingerprint.Create(), + Is.EqualTo(new TypeLayoutFingerprint(0xD52D441606CE7C3F, 0x4D50B1F77C44F34C))); + Assert.That(TypeLayoutFingerprint.Create(), + Is.EqualTo(new TypeLayoutFingerprint(0x7EE1FD251B406A46, 0x9427C9B56B8CC942))); + Assert.That(TypeLayoutFingerprint.Create(), + Is.EqualTo(new TypeLayoutFingerprint(0xB7337E9F0D12BB73, 0x2C0F045AB00D96C8))); + } + + [Test] + public void GenericUnmanagedStructs_HaveStableDistinctFingerprints() + { + // Marshal.SizeOf rejects generic types, so (int, int) used to fail with ArgumentException + // in every typed container even though it satisfies the unmanaged constraint. + var tupleA = TypeLayoutFingerprint.Create<(int, int)>(); + var tupleB = TypeLayoutFingerprint.Create<(int, int)>(); + var differentArgument = TypeLayoutFingerprint.Create<(int, uint)>(); + var differentShape = TypeLayoutFingerprint.Create<(int, int, int)>(); + var pair = TypeLayoutFingerprint.Create>(); + + Assert.That(tupleA, Is.EqualTo(tupleB)); + Assert.That(tupleA, Is.Not.EqualTo(differentArgument)); + Assert.That(tupleA, Is.Not.EqualTo(differentShape)); + Assert.That(tupleA, Is.Not.EqualTo(pair)); + } + + [Test] + public void SingleProducerQueue_AcceptsGenericStruct() + { + string name = N("SpscTuple"); + using var producer = SingleProducerQueue<(int, long)>.CreateOrOpen(name, 8); + Assert.That(producer.TryEnqueue((7, 9L)), Is.True); + + using var consumer = SingleProducerQueue<(int, long)>.OpenExisting(name); + Assert.That(consumer.TryDequeue(out (int, long) item), Is.True); + Assert.That(item, Is.EqualTo((7, 9L))); + } + + [Test] + public void ConcurrentQueue_AcceptsGenericStruct() + { + string name = N("MpmcPair"); + using var queue = InterprocessMemory.ConcurrentQueue>.CreateOrOpen(name, 8); + Assert.That(queue.TryEnqueue(new KeyValuePair(3, 4)), Is.True); + + using var reader = InterprocessMemory.ConcurrentQueue>.OpenExisting(name); + Assert.That(reader.TryDequeue(out KeyValuePair item), Is.True); + Assert.That(item.Key, Is.EqualTo(3)); + Assert.That(item.Value, Is.EqualTo(4)); + } + + [Test] + public void SharedArray_AcceptsGenericStruct() + { + string name = N("ArrayTuple"); + using var owner = SharedArray<(int, int)>.CreateOrOpen(name, 4); + owner[2] = (5, 6); + + using var reader = SharedArray<(int, int)>.OpenExisting(name); + Assert.That(reader[2], Is.EqualTo((5, 6))); + } + + [Test] + public void OpeningGenericStructQueueWithDifferentLayout_Throws() + { + // Same size (8 bytes), different element type: only the fingerprint tells them apart. + string name = N("Mismatch"); + using var owner = SingleProducerQueue<(int, int)>.CreateOrOpen(name, 4); + + Assert.Throws(() => SingleProducerQueue<(int, uint)>.OpenExisting(name)); + } +} diff --git a/InterprocessMemory/IMemoryRegion.cs b/InterprocessMemory/IMemoryRegion.cs index ef36713..5dfa840 100644 --- a/InterprocessMemory/IMemoryRegion.cs +++ b/InterprocessMemory/IMemoryRegion.cs @@ -131,13 +131,21 @@ public interface IMemoryRegion : IDisposable Memory GetMemory(long offset, int length); /// - /// Tries to acquire an exclusive write lock with timeout + /// Tries to acquire an exclusive write lock with timeout. + /// The wait is released with if the region is disposed + /// by another thread while waiting. /// bool TryAcquireWriteLock(TimeSpan timeout); /// - /// Releases the write lock + /// Releases the write lock. The lock is owned by the acquiring thread of the acquiring + /// process, so it must be released on that same thread: do not await between + /// acquiring and releasing it. /// + /// + /// The calling thread does not own the write lock, or the lock was taken over (for example by + /// orphan-lock recovery) before it was released. + /// void ReleaseWriteLock(); /// diff --git a/InterprocessMemory/MemoryRegion.cs b/InterprocessMemory/MemoryRegion.cs index e0845b8..f3b47f8 100644 --- a/InterprocessMemory/MemoryRegion.cs +++ b/InterprocessMemory/MemoryRegion.cs @@ -59,15 +59,37 @@ public sealed unsafe class MemoryRegion : IMemoryRegion private volatile int _disposed; + // Threads currently spinning inside TryAcquireWriteLock/TryAcquireReadLock. Those waits can be + // unbounded, so they are the one place where another thread can realistically call Dispose() + // while the shared header is being touched; Dispose() waits for this count to drain before it + // unmaps the view. Read/Write are not tracked: they are short, and two interlocked operations + // per call would cost more than they protect. + private int _activeWaiters; + + // How often a lock waiter re-probes whether the owning process is still alive. + private const long OrphanCheckIntervalMs = 250; + + // Upper bound on how long Dispose() waits for lock waiters to notice the disposed flag. + private static readonly TimeSpan s_waiterDrainTimeout = TimeSpan.FromSeconds(5); + // Cached once per process: stamped into the header at lock acquire so an orphan check // can distinguish "same PID, same process" from "same PID, recycled by the OS for an - // unrelated process". Process.StartTime can throw under restricted permissions (Linux - // containers without /proc, certain Windows ACLs) — in that case we store 0 and the - // orphan check silently falls back to PID-only matching. + // unrelated process". The start time can be unreadable under restricted permissions + // (Linux containers without /proc, certain Windows ACLs) — in that case we store 0 and + // the orphan check silently falls back to PID-only matching. + // + // Windows: Process.StartTime.ToBinary() (the kernel creation time, identical for every observer). + // Linux: the kernel start tick count from /proc//stat. Process.StartTime must NOT be used + // there: every process derives it from its own wall-clock boot-time snapshot, so the value the + // owner records and the value another process computes for the same owner differ by + // milliseconds and would make every live owner look like an impostor. private static readonly long s_processStartTimeBinary = TryCaptureProcessStartTime(); private static long TryCaptureProcessStartTime() { + if (OperatingSystem.IsLinux()) + return TryReadLinuxStartTicks(Environment.ProcessId); + try { using var p = Process.GetCurrentProcess(); @@ -79,6 +101,53 @@ private static long TryCaptureProcessStartTime() } } + /// + /// Reads field 22 (starttime, clock ticks since boot) of /proc/<pid>/stat. + /// Returns 0 when it cannot be read. + /// + private static long TryReadLinuxStartTicks(int pid) + { + try + { + string stat = File.ReadAllText("/proc/" + pid + "/stat"); + + // The command name (field 2) is parenthesised and may itself contain spaces or + // parentheses, so split only what follows the LAST ')'. The first token after it + // is field 3, which makes field 22 index 19. + int commEnd = stat.LastIndexOf(')'); + if (commEnd < 0) + return 0; + + string[] fields = stat.Substring(commEnd + 2).Split(' '); + return fields.Length > 19 && long.TryParse(fields[19], out long ticks) ? ticks : 0; + } + catch + { + return 0; + } + } + + /// + /// True when the process currently using is provably not the one + /// that recorded (PID reuse). + /// + private static bool IsOwnerStartTimeMismatch(Process process, int ownerPid, long storedStartTime) + { + if (OperatingSystem.IsLinux()) + { + // 3.0.0 recorded Process.StartTime.ToBinary() here, which is negative for a local + // DateTime and not comparable across processes. Tick counts are positive. Treat the + // legacy format as "unknown" and keep the PID-only decision. + if (storedStartTime < 0) + return false; + + long currentTicks = TryReadLinuxStartTicks(ownerPid); + return currentTicks != 0 && currentTicks != storedStartTime; + } + + return process.StartTime.ToBinary() != storedStartTime; + } + /// /// Extended header structure with orphan lock detection support. /// WriterLockState and ReaderCount are placed in separate 64-byte cache lines @@ -715,14 +784,30 @@ public bool TryAcquireWriteLock(TimeSpan timeout) ThrowIfDisposed(); TimeoutHelper.Validate(timeout, nameof(timeout)); + EnterWait(); + try + { + return TryAcquireWriteLockCore(timeout); + } + finally + { + ExitWait(); + } + } + + private bool TryAcquireWriteLockCore(TimeSpan timeout) + { var header = (SharedHeader*)_basePtr; var sw = Stopwatch.StartNew(); var spinner = new SpinWait(); - bool orphanCheckDone = false; - bool orphanCheckNearTimeout = false; + long nextOrphanCheckMs = 0; while (true) { + // Dispose() unmaps the header while we may still be spinning on it. It waits for + // registered waiters (EnterWait) to leave, and we leave as soon as we see the flag. + ThrowIfDisposed(); + if (Interlocked.CompareExchange(ref header->WriterLockState, 1, 0) == 0) { bool success = false; @@ -739,6 +824,8 @@ public bool TryAcquireWriteLock(TimeSpan timeout) var readerSpinner = new SpinWait(); while (Volatile.Read(ref header->ReaderCount) > 0) { + ThrowIfDisposed(); + if (TimeoutHelper.HasExpired(sw, timeout)) { return false; // Will release lock in finally @@ -768,25 +855,18 @@ public bool TryAcquireWriteLock(TimeSpan timeout) if (TimeoutHelper.HasExpired(sw, timeout)) return false; - if (_options.EnableOrphanLockDetection) + if (_options.EnableOrphanLockDetection && sw.ElapsedMilliseconds >= nextOrphanCheckMs) { - // Check on first CAS failure; re-check when nearing timeout (≥75% elapsed) - // so a lock that becomes orphaned mid-wait is still recovered before giving up. - bool nearTimeout = TimeoutHelper.IsNearExpiry(sw, timeout, 0.75); + // Check on the first CAS failure and then periodically. The owner may die at any + // point while we wait, and a wait with Timeout.InfiniteTimeSpan has no deadline to + // key a one-off re-check on. The probe costs a process lookup, hence the interval. + nextOrphanCheckMs = sw.ElapsedMilliseconds + OrphanCheckIntervalMs; - if (!orphanCheckDone || (nearTimeout && !orphanCheckNearTimeout)) + if (IsWriteLockOrphaned()) { - if (!orphanCheckDone) - orphanCheckDone = true; - else - orphanCheckNearTimeout = true; - - if (IsWriteLockOrphaned()) - { - _logger?.LogWarning("Detected orphan write lock, attempting recovery"); - TryForceReleaseWriteLock(); - continue; - } + _logger?.LogWarning("Detected orphan write lock, attempting recovery"); + TryForceReleaseWriteLock(); + continue; } } @@ -805,21 +885,28 @@ public void ReleaseWriteLock() long currentThreadId = Environment.CurrentManagedThreadId; int ownerPid = Volatile.Read(ref header->LockOwnerProcessId); long ownerThreadId = Volatile.Read(ref header->LockOwnerThreadId); + // Releasing a lock this thread does not own must be loud. Silently ignoring it (the + // previous behaviour) turned `await` inside a lock scope — the continuation resumes on + // another thread — into a write lock that no process could ever release again. if (ownerPid != currentPid || ownerThreadId != currentThreadId) { _logger?.LogWarning( - "ReleaseWriteLock called from PID {Pid}/thread {ThreadId} but lock owner is PID {OwnerPid}/thread {OwnerThreadId} — ignored", + "ReleaseWriteLock called from PID {Pid}/thread {ThreadId} but lock owner is PID {OwnerPid}/thread {OwnerThreadId}", currentPid, currentThreadId, ownerPid, ownerThreadId); - return; + throw new SynchronizationLockException( + $"The write lock must be released by the thread that acquired it " + + $"(caller PID {currentPid}/thread {currentThreadId}, lock owner PID {ownerPid}/thread {ownerThreadId}). " + + "Do not await inside a lock scope."); } int prev = Interlocked.CompareExchange(ref header->LockOwnerProcessId, 0, currentPid); if (prev != currentPid) { _logger?.LogWarning( - "ReleaseWriteLock called from PID {Pid} but lock owner is {OwnerPid} — ignored", + "ReleaseWriteLock called from PID {Pid} but lock owner is {OwnerPid}", currentPid, prev); - return; + throw new SynchronizationLockException( + "The write lock was taken over (for example by orphan-lock recovery) before it was released."); } header->LockOwnerThreadId = 0; @@ -838,12 +925,28 @@ public bool TryAcquireReadLock(TimeSpan timeout) ThrowIfDisposed(); TimeoutHelper.Validate(timeout, nameof(timeout)); + EnterWait(); + try + { + return TryAcquireReadLockCore(timeout); + } + finally + { + ExitWait(); + } + } + + private bool TryAcquireReadLockCore(TimeSpan timeout) + { var header = (SharedHeader*)_basePtr; var sw = Stopwatch.StartNew(); var spinner = new SpinWait(); while (true) { + // See TryAcquireWriteLockCore: leave promptly once Dispose() has started. + ThrowIfDisposed(); + // Fast path: peek the writer flag without any atomic. If a writer is active, // wait — touching ReaderCount unnecessarily would create cache-line traffic on // the reader-side line and prolong the writer's release-then-drain phase. @@ -924,14 +1027,13 @@ public bool IsWriteLockOrphaned() // PID-reuse defense: even when a process with this PID exists, it might be an // unrelated process that the OS recycled the PID for after the real owner died. - // Compare the captured StartTime; mismatch ⇒ impostor ⇒ orphan. + // Compare the captured start time; mismatch ⇒ impostor ⇒ orphan. long storedStartTime = header->LockOwnerProcessStartTime; if (storedStartTime != 0) { try { - long currentStartTime = process.StartTime.ToBinary(); - if (currentStartTime != storedStartTime) + if (IsOwnerStartTimeMismatch(process, ownerPid, storedStartTime)) { _logger?.LogWarning( "Lock owner PID {Pid} still exists but its StartTime differs (orphan from PID reuse)", @@ -1105,7 +1207,26 @@ private void ThrowIfDisposed() } /// - /// Releases all resources used by this buffer + /// Registers the calling thread as a lock waiter. Increment first, then check the flag: with + /// the full fences of the two interlocked operations, either this thread sees the disposed + /// flag or sees this thread in . + /// + private void EnterWait() + { + Interlocked.Increment(ref _activeWaiters); + if (_disposed != 0) + { + Interlocked.Decrement(ref _activeWaiters); + throw new ObjectDisposedException(nameof(MemoryRegion)); + } + } + + private void ExitWait() => Interlocked.Decrement(ref _activeWaiters); + + /// + /// Releases all resources used by this buffer. Threads blocked in + /// or are released with an + /// before the memory is unmapped. /// public void Dispose() { @@ -1113,6 +1234,15 @@ public void Dispose() return; _logger?.LogDebug("Disposing shared buffer '{Name}'", _name); + + // Waiters observe _disposed on every spin iteration and leave within a few milliseconds. + // Never unmap underneath a thread that is still dereferencing the header: that is an + // AccessViolationException, which terminates the process and cannot be caught. + var drain = Stopwatch.StartNew(); + var drainSpinner = new SpinWait(); + while (Volatile.Read(ref _activeWaiters) > 0 && drain.Elapsed < s_waiterDrainTimeout) + drainSpinner.SpinOnce(); + Cleanup(disposing: true); GC.SuppressFinalize(this); } diff --git a/InterprocessMemory/StructuredMemory.cs b/InterprocessMemory/StructuredMemory.cs index 6d4a65a..248592e 100644 --- a/InterprocessMemory/StructuredMemory.cs +++ b/InterprocessMemory/StructuredMemory.cs @@ -720,6 +720,11 @@ public WriteLock AcquireWriteLock() /// Lock acquisition timeout /// A disposable lock guard that releases the lock on dispose /// Thrown when the lock cannot be acquired within the timeout + /// + /// The lock is owned by the calling thread. Dispose the guard on that same thread; do not + /// await inside the guarded scope, or throws + /// . + /// public WriteLock AcquireWriteLock(TimeSpan timeout) { ThrowIfDisposed(); @@ -759,6 +764,10 @@ public ReadLock AcquireReadLock() /// Lock acquisition timeout /// A disposable lock guard that releases the lock on dispose /// Thrown when the lock cannot be acquired within the timeout + /// + /// Dispose the guard on the thread that acquired it; do not await inside the guarded + /// scope, or throws . + /// public ReadLock AcquireReadLock(TimeSpan timeout) { ThrowIfDisposed(); @@ -1152,24 +1161,46 @@ public struct WriteLock : IDisposable { private IMemoryRegion? _buffer; private Action? _onDispose; + private readonly int _ownerThreadId; internal WriteLock(IMemoryRegion? buffer, Action? onDispose = null) { _buffer = buffer; _onDispose = onDispose; + _ownerThreadId = Environment.CurrentManagedThreadId; } /// - /// Releases the write lock if not already released + /// Releases the write lock if not already released. /// + /// + /// Called on a different thread than the one that acquired the lock, which is what happens + /// when the guarded scope contains an await. Nothing is released in that case. + /// public void Dispose() { + if (_onDispose == null) + return; + + // The lock and its reentrancy depth belong to the acquiring thread. Releasing from + // another thread would leave that thread believing it still holds the lock. + if (Environment.CurrentManagedThreadId != _ownerThreadId) + throw new SynchronizationLockException( + "A write lock guard must be disposed on the thread that acquired it. " + + "Do not await inside a lock scope."); + var onDispose = Interlocked.Exchange(ref _onDispose, null); if (onDispose != null) { - _buffer?.ReleaseWriteLock(); - _buffer = null; - onDispose.Invoke(); + try + { + _buffer?.ReleaseWriteLock(); + } + finally + { + _buffer = null; + onDispose.Invoke(); + } } } } @@ -1186,24 +1217,44 @@ public struct ReadLock : IDisposable { private IMemoryRegion? _buffer; private Action? _onDispose; + private readonly int _ownerThreadId; internal ReadLock(IMemoryRegion? buffer, Action? onDispose = null) { _buffer = buffer; _onDispose = onDispose; + _ownerThreadId = Environment.CurrentManagedThreadId; } /// - /// Releases the read lock if not already released + /// Releases the read lock if not already released. /// + /// + /// Called on a different thread than the one that acquired the lock. Nothing is released + /// in that case. See . + /// public void Dispose() { + if (_onDispose == null) + return; + + if (Environment.CurrentManagedThreadId != _ownerThreadId) + throw new SynchronizationLockException( + "A read lock guard must be disposed on the thread that acquired it. " + + "Do not await inside a lock scope."); + var onDispose = Interlocked.Exchange(ref _onDispose, null); if (onDispose != null) { - _buffer?.ReleaseReadLock(); - _buffer = null; - onDispose.Invoke(); + try + { + _buffer?.ReleaseReadLock(); + } + finally + { + _buffer = null; + onDispose.Invoke(); + } } } } diff --git a/InterprocessMemory/TimeoutHelper.cs b/InterprocessMemory/TimeoutHelper.cs index 6f533be..77a1fe3 100644 --- a/InterprocessMemory/TimeoutHelper.cs +++ b/InterprocessMemory/TimeoutHelper.cs @@ -20,12 +20,5 @@ public static bool HasExpired(Stopwatch stopwatch, TimeSpan timeout) { return timeout != Timeout.InfiniteTimeSpan && stopwatch.Elapsed > timeout; } - - public static bool IsNearExpiry(Stopwatch stopwatch, TimeSpan timeout, double fraction) - { - return timeout != Timeout.InfiniteTimeSpan - && timeout > TimeSpan.Zero - && stopwatch.Elapsed.TotalMilliseconds >= timeout.TotalMilliseconds * fraction; - } } } diff --git a/InterprocessMemory/TypeLayoutFingerprint.cs b/InterprocessMemory/TypeLayoutFingerprint.cs index 5f29fbd..1a11496 100644 --- a/InterprocessMemory/TypeLayoutFingerprint.cs +++ b/InterprocessMemory/TypeLayoutFingerprint.cs @@ -3,6 +3,7 @@ using System.Collections.Generic; using System.Linq; using System.Reflection; +using System.Runtime.CompilerServices; using System.Runtime.InteropServices; using System.Security.Cryptography; using System.Text; @@ -14,7 +15,21 @@ internal readonly record struct TypeLayoutFingerprint(ulong Low, ulong High) public static TypeLayoutFingerprint Create() where T : unmanaged { var descriptor = new StringBuilder(256); - AppendType(descriptor, typeof(T), new HashSet()); + try + { + AppendType(descriptor, typeof(T), new HashSet()); + } + catch (ArgumentException) + { + // Marshal.SizeOf/OffsetOf reject generic types (ValueTuple, KeyValuePair, ...), at any + // nesting depth, although they satisfy the unmanaged constraint. Describe those with + // the managed layout instead. Types the marshaller accepts keep the descriptor above, + // so their fingerprint does not change and processes built from different releases + // can still open the same region. + descriptor.Clear(); + AppendManagedType(descriptor, typeof(T), new HashSet()); + } + byte[] hash = SHA256.HashData(Encoding.UTF8.GetBytes(descriptor.ToString())); return new TypeLayoutFingerprint( BinaryPrimitives.ReadUInt64LittleEndian(hash), @@ -55,5 +70,65 @@ private static void AppendType(StringBuilder target, Type type, HashSet ac active.Remove(type); } + + // Unsafe.SizeOf() is the size the containers actually copy (bool = 1 byte), and it works for + // generic types. Unsafe.SizeOf has no generic constraint, so it can be bound to any Type. + private static readonly MethodInfo s_sizeOf = typeof(Unsafe).GetMethod(nameof(Unsafe.SizeOf))!; + + private static int ManagedSizeOf(Type type) => + (int)s_sizeOf.MakeGenericMethod(type).Invoke(null, null)!; + + // Version-independent type name. Type.FullName of a constructed generic type embeds the + // assembly-qualified name (including Version=) of every type argument, which differs between + // .NET releases and would make processes on different runtimes disagree. + private static void AppendTypeName(StringBuilder target, Type type) + { + target.Append(type.Assembly.GetName().Name).Append(':'); + if (!type.IsConstructedGenericType) + { + target.Append(type.FullName); + return; + } + + target.Append(type.GetGenericTypeDefinition().FullName).Append('['); + Type[] arguments = type.GetGenericArguments(); + for (int i = 0; i < arguments.Length; i++) + { + if (i > 0) + target.Append(','); + AppendTypeName(target, arguments[i]); + } + target.Append(']'); + } + + private static void AppendManagedType(StringBuilder target, Type type, HashSet active) + { + AppendTypeName(target, type); + target.Append('|').Append(ManagedSizeOf(type)); + + var layout = type.StructLayoutAttribute; + target.Append('|').Append((int)(layout?.Value ?? LayoutKind.Auto)) + .Append('|').Append(layout?.Pack ?? 0) + .Append('|').Append(layout?.Size ?? 0); + + if (type.IsPrimitive || type.IsEnum || !active.Add(type)) + return; + + // Declaration order (metadata token), explicit FieldOffset values, the total size and the + // layout kind together pin down the memory layout without Marshal.OffsetOf. + var fields = type + .GetFields(BindingFlags.Instance | BindingFlags.Public | BindingFlags.NonPublic) + .OrderBy(field => field.MetadataToken); + + foreach (var field in fields) + { + int? explicitOffset = field.GetCustomAttribute()?.Value; + target.Append(";f:").Append(field.Name).Append('@') + .Append(explicitOffset?.ToString() ?? "-").Append(':'); + AppendManagedType(target, field.FieldType, active); + } + + active.Remove(type); + } } } diff --git a/README.md b/README.md index 7a79557..dd279d7 100644 --- a/README.md +++ b/README.md @@ -59,7 +59,14 @@ if (memory.TryAcquireWriteLock(TimeSpan.FromSeconds(1))) ``` Lock ownership includes the process ID, managed thread ID, and process start time. A process -waiting for a write lock can recover a lock left behind by a terminated process. +waiting for a write lock keeps checking whether the owner is still alive (also when it waits +with `Timeout.InfiniteTimeSpan`) and recovers a lock left behind by a terminated process. + +The write lock belongs to the thread that acquired it. Release it on that same thread; releasing +from another thread throws `SynchronizationLockException` and leaves the lock held. In particular, +do not `await` between acquiring and releasing a lock, because the continuation may resume on a +different thread. Disposing a region while another thread is waiting for one of its locks releases +that thread with an `ObjectDisposedException`. ## Choosing a data structure @@ -204,6 +211,10 @@ using (memory.AcquireWriteLock()) UTF-8 strings, and arrays to prevent torn reads and writes. Use an explicit lock when several fields form one transaction. +The lock guards returned by `AcquireWriteLock()` and `AcquireReadLock()` must be disposed on the +thread that acquired them. Keep the guarded scope synchronous: an `await` inside it lets the guard +be disposed on another thread, which throws `SynchronizationLockException`. + ## Shared arrays ```csharp From 0b0bb0d55db92e9fac9293eb55cafc4144270953 Mon Sep 17 00:00:00 2001 From: zacala Date: Sat, 3 Oct 2026 22:29:13 +0000 Subject: [PATCH 02/29] Make time-based lock takeover opt-in and add crash-recovery tools * OrphanLockTimeout now defaults to TimeSpan.Zero. The previous default of 30 s let a waiter take the write lock from a healthy process that held it longer than that (a long transaction, a paused debugger), breaking mutual exclusion. Recovery of a lock whose owner process has exited is unaffected and stays on by default. DefaultOrphanLockTimeout keeps its value and is documented as the suggested value when opting in. * Read locks are not attributed to an owner, so a process that dies holding one leaves the shared reader count above zero forever; on Linux the /dev/shm file even outlives every user. This cannot be detected automatically, so give operators explicit tools: - MemoryRegion.ForceResetLocks() / StructuredMemory.ForceResetLocks() clear the write lock and the reader count. - LockOwnerInfo.ReaderCount exposes the count for diagnosis. - MemoryRegion.Remove(name, options) deletes the backing storage (Linux /dev/shm file or an explicit FilePath) so a region left unusable by a crashed creator, or one with an unwanted capacity or element type, can be recreated. The initialization timeout message now points to it. README documents both the new default and the recovery procedure. Tests: cross-process dead-reader scenario (new hold_read_lock worker role), ForceResetLocks for MemoryRegion and StructuredMemory, Remove for Linux regions, file-backed regions and a region stuck in the initializing state. --- InterprocessMemory.TestWorker/Program.cs | 14 ++ InterprocessMemory.Tests/CrossProcessTests.cs | 30 +++- .../LibraryHardeningTests.cs | 130 ++++++++++++++++++ InterprocessMemory/IMemoryRegion.cs | 7 + InterprocessMemory/MemoryRegion.cs | 89 +++++++++++- InterprocessMemory/MemoryRegionOptions.cs | 20 ++- InterprocessMemory/StructuredMemory.cs | 13 ++ README.md | 19 +++ 8 files changed, 312 insertions(+), 10 deletions(-) diff --git a/InterprocessMemory.TestWorker/Program.cs b/InterprocessMemory.TestWorker/Program.cs index 07e5dcd..91d9614 100644 --- a/InterprocessMemory.TestWorker/Program.cs +++ b/InterprocessMemory.TestWorker/Program.cs @@ -19,6 +19,7 @@ /// try_write_lock <name> — try the cross-process write lock for 250 ms /// orphan_write_lock <name> — acquire a write lock and exit without releasing it /// hold_write_lock <name> — acquire a write lock, print "holding", and keep it until killed +/// hold_read_lock <name> — acquire a read lock, print "holding", and keep it until killed /// if (args.Length < 2) { @@ -45,6 +46,7 @@ "try_write_lock" => TryWriteLock(bufferName), "orphan_write_lock" => OrphanWriteLock(bufferName), "hold_write_lock" => HoldWriteLock(bufferName), + "hold_read_lock" => HoldReadLock(bufferName), _ => Error($"Unknown role: {role}") }; @@ -213,6 +215,18 @@ static int HoldWriteLock(string name) return 0; } +static int HoldReadLock(string name) +{ + using var region = MemoryRegion.OpenExisting(name); + if (!region.TryAcquireReadLock(TimeSpan.FromSeconds(5))) + return Error("failed to acquire held test read lock"); + + Console.WriteLine("holding"); + Console.Out.Flush(); + Thread.Sleep(Timeout.Infinite); + return 0; +} + static int Error(string msg) { Console.Error.WriteLine(msg); diff --git a/InterprocessMemory.Tests/CrossProcessTests.cs b/InterprocessMemory.Tests/CrossProcessTests.cs index 46dd7bd..1caaf73 100644 --- a/InterprocessMemory.Tests/CrossProcessTests.cs +++ b/InterprocessMemory.Tests/CrossProcessTests.cs @@ -294,12 +294,12 @@ public void CrossProcess_OrphanWriteLock_IsRecovered() } /// - /// Starts a child that holds the write lock until it is killed, and returns once the child - /// has reported that it holds the lock. + /// Starts a child that holds a lock until it is killed, and returns once the child has reported + /// that it holds the lock. is hold_write_lock or hold_read_lock. /// - private static Process StartLockHolder(string bufferName) + private static Process StartLockHolder(string bufferName, string role = "hold_write_lock") { - Process process = Process.Start(CreateHelperStartInfo("hold_write_lock", bufferName))!; + Process process = Process.Start(CreateHelperStartInfo(role, bufferName))!; string? line = process.StandardOutput.ReadLine(); if (line != "holding") { @@ -412,6 +412,28 @@ public void CrossProcess_FiniteWait_RecoversPromptlyWhenOwnerDies() } } + [Test, Timeout(30000)] + public void CrossProcess_DeadReaderProcess_NeedsForceResetLocks() + { + // Read locks are not attributed to an owner, so a reader that dies leaves the shared count + // above zero and nothing can tell that it is stale. ForceResetLocks is the operator's way out. + string name = GetUniqueName("DeadReader"); + using var region = MemoryRegion.CreateOrOpen(name, 256); + using Process holder = StartLockHolder(name, "hold_read_lock"); + Assert.That(region.GetLockOwnerInfo().ReaderCount, Is.EqualTo(1)); + KillAndWait(holder); + + Assert.That(region.TryAcquireWriteLock(TimeSpan.FromMilliseconds(500)), Is.False, + "the dead reader's lock is still counted"); + Assert.That(region.GetLockOwnerInfo().ReaderCount, Is.EqualTo(1)); + + region.ForceResetLocks(); + + Assert.That(region.GetLockOwnerInfo().ReaderCount, Is.EqualTo(0)); + Assert.That(region.TryAcquireWriteLock(TimeSpan.FromSeconds(1)), Is.True); + region.ReleaseWriteLock(); + } + // ── Schema ─────────────────────────────────────────────────────────────── public struct IpcTestSchema : IMemorySchema diff --git a/InterprocessMemory.Tests/LibraryHardeningTests.cs b/InterprocessMemory.Tests/LibraryHardeningTests.cs index 4b6d6ea..fbfbcf8 100644 --- a/InterprocessMemory.Tests/LibraryHardeningTests.cs +++ b/InterprocessMemory.Tests/LibraryHardeningTests.cs @@ -209,6 +209,136 @@ public void StructuredMemory_ReadLockGuard_DisposedOnAnotherThread_ThrowsAndStay } } + [Test] + public void OrphanLockTimeout_IsDisabledByDefault() + { + // A time limit takes the lock away from a healthy owner that merely holds it for long + // (a long transaction, a paused debugger), so it is opt-in. Dead owners are still recovered. + var options = new MemoryRegionOptions(); + + Assert.That(options.OrphanLockTimeout, Is.EqualTo(TimeSpan.Zero)); + Assert.That(options.EnableOrphanLockDetection, Is.True); + } + + [Test] + public void ForceResetLocks_ClearsReadersAndHeldWriteLock() + { + using var buffer = new MemoryRegion( + N("ForceReset"), + new MemoryRegionOptions { Capacity = 256 }); + + Assert.That(buffer.TryAcquireReadLock(TimeSpan.FromSeconds(1)), Is.True); + Assert.That(buffer.GetLockOwnerInfo().ReaderCount, Is.EqualTo(1)); + Assert.That(buffer.TryAcquireWriteLock(TimeSpan.FromMilliseconds(100)), Is.False); + + buffer.ForceResetLocks(); + + Assert.That(buffer.GetLockOwnerInfo().ReaderCount, Is.EqualTo(0)); + Assert.That(buffer.TryAcquireWriteLock(TimeSpan.FromSeconds(1)), Is.True); + + // Resetting also frees a write lock that is still "held"; its holder learns that on release. + buffer.ForceResetLocks(); + Assert.Throws(() => buffer.ReleaseWriteLock()); + + Assert.That(buffer.TryAcquireWriteLock(TimeSpan.FromSeconds(1)), Is.True); + buffer.ReleaseWriteLock(); + } + + [Test] + public void StructuredMemory_ForceResetLocks_UnblocksWriter() + { + string name = N("StructuredReset"); + using var memory = StructuredMemory.CreateOrOpen(name, new SimpleSchema()); + using var peer = StructuredMemory.OpenExisting(name, new SimpleSchema()); + + // A reader that never comes back, as after a crash. + var staleReader = peer.AcquireReadLock(); + Assert.Throws(() => memory.AcquireWriteLock(TimeSpan.FromMilliseconds(100))); + + memory.ForceResetLocks(); + + using (memory.AcquireWriteLock(TimeSpan.FromSeconds(1))) + { + } + + staleReader.Dispose(); + } + + [Test] + public void Remove_DeletesLinuxRegion_SoItCanBeRecreatedWithDifferentCapacity() + { + if (!OperatingSystem.IsLinux()) + Assert.Ignore("Only Linux keeps regions in /dev/shm after their last user is gone."); + + string name = N("Remove"); + using (MemoryRegion.CreateOrOpen(name, 256)) + { + } + + // The region outlives its users, so a different capacity is rejected... + Assert.Throws(() => MemoryRegion.CreateOrOpen(name, 512)); + + // ...until it is removed. + Assert.That(MemoryRegion.Remove(name), Is.True); + Assert.That(MemoryRegion.Remove(name), Is.False); + + using var fresh = MemoryRegion.CreateOrOpen(name, 512); + Assert.That(fresh.IsOwner, Is.True); + Assert.That(fresh.Capacity, Is.EqualTo(512)); + } + + [Test] + public void Remove_ClearsRegionLeftInitializingByCrashedCreator() + { + if (!OperatingSystem.IsLinux()) + Assert.Ignore("Only Linux keeps regions in /dev/shm after their last user is gone."); + + // "IPMI": the creator died after claiming initialization and before publishing the header, + // which makes every opener wait and then time out. + string name = N("StuckInit"); + using (var file = new FileStream("/dev/shm/" + name, FileMode.CreateNew)) + { + file.SetLength(128 + 256); + file.Write(BitConverter.GetBytes(0x494D5049u)); + } + + Assert.That(MemoryRegion.Remove(name), Is.True); + + using var region = MemoryRegion.CreateOrOpen(name, 256); + Assert.That(region.IsOwner, Is.True); + } + + [Test] + public void Remove_FileBackedRegion_DeletesTheFile() + { + string name = N("RemoveFile"); + string path = Path.Combine(Path.GetTempPath(), name + ".bin"); + var options = new MemoryRegionOptions { FilePath = path }; + try + { + using (MemoryRegion.CreateOrOpen(name, 256, options)) + { + } + + Assert.That(File.Exists(path), Is.True); + Assert.That(MemoryRegion.Remove(name, options), Is.True); + Assert.That(File.Exists(path), Is.False); + Assert.That(MemoryRegion.Remove(name, options), Is.False); + } + finally + { + File.Delete(path); + } + } + + [Test] + public void Remove_UnknownRegion_ReturnsFalse_AndBadNamesAreRejected() + { + Assert.That(MemoryRegion.Remove(N("Nothing")), Is.False); + Assert.Throws(() => MemoryRegion.Remove("a/b")); + Assert.Throws(() => MemoryRegion.Remove("")); + } + [Test] public void ReadLock_DoubleRelease_DoesNotBreakWriterExclusion() { diff --git a/InterprocessMemory/IMemoryRegion.cs b/InterprocessMemory/IMemoryRegion.cs index 5dfa840..22bbdc3 100644 --- a/InterprocessMemory/IMemoryRegion.cs +++ b/InterprocessMemory/IMemoryRegion.cs @@ -74,6 +74,13 @@ public readonly struct LockOwnerInfo /// Gets whether the lock is orphaned (owner process died) /// public bool IsOrphan { get; init; } + + /// + /// Gets the number of read locks currently held across all processes. Read locks are not + /// attributed to an owner, so a process that dies while holding one leaves this count + /// permanently above zero (writers then time out); see . + /// + public int ReaderCount { get; init; } } /// diff --git a/InterprocessMemory/MemoryRegion.cs b/InterprocessMemory/MemoryRegion.cs index f3b47f8..33d3f2b 100644 --- a/InterprocessMemory/MemoryRegion.cs +++ b/InterprocessMemory/MemoryRegion.cs @@ -247,6 +247,54 @@ internal static MemoryRegion OpenExisting( RegionKind regionKind) => new(name, capacityBytes: null, options, createOrOpen: false, regionKind); + /// + /// Deletes the backing storage of a named region so that the next + /// starts from scratch. + /// Use it to get rid of a region that a crash left unusable, for example one whose creator died + /// during initialization, or to change the capacity or element type of an existing region. + /// + /// Linux keeps the region as a file in /dev/shm that outlives its users, so it has to be + /// removed explicitly. Windows named sections are reference counted by the kernel and vanish when + /// the last handle closes; for them (without ) this + /// returns false. + /// + /// + /// Stop every process that uses the region first. Processes that still have it mapped keep + /// working on the removed storage, while later openers get a new, independent region. + /// This applies to every region kind (typed queues, arrays, structured memory), all of which are + /// addressed by the same name. + /// + /// + /// The region name that was passed to CreateOrOpen. + /// + /// Pass the same options (in particular ) that the region + /// was created with when it is file backed. + /// + /// true when backing storage was deleted; false when there was nothing to delete. + public static bool Remove(string name, MemoryRegionOptions? options = null) + { + ValidateFlatName(name); + + string? path = options?.FilePath; + if (string.IsNullOrEmpty(path)) + { + if (OperatingSystem.IsWindows()) + return false; + + if (!OperatingSystem.IsLinux()) + throw new PlatformNotSupportedException( + "MemoryRegion requires Windows or Linux unless MemoryRegionOptions.FilePath is used."); + + path = "/dev/shm/" + name; + } + + if (!File.Exists(path)) + return false; + + File.Delete(path); + return true; + } + internal MemoryRegion(string name, MemoryRegionOptions? options = null) : this( name, @@ -535,7 +583,9 @@ private bool InitializeOrOpen() } if (sw.Elapsed > TimeSpan.FromSeconds(5)) throw new TimeoutException( - "Timed out waiting for shared memory to be initialized by another process"); + "Timed out waiting for shared memory to be initialized by another process. " + + "If that process crashed during initialization, stop every user of the region " + + "and call MemoryRegion.Remove(name)."); Thread.SpinWait(100); } @@ -1127,6 +1177,40 @@ public bool TryForceReleaseWriteLock() return true; } + /// + /// Unconditionally clears the shared write lock and the read-lock count. + /// + /// This is the recovery tool for a lock state that cannot be recovered automatically. Read + /// locks are not attributed to an owner, so when a process dies while holding one the + /// shared reader count stays above zero forever (see ) + /// and every writer times out, including after the region is reopened on Linux, where the + /// backing file outlives its users. + /// + /// + /// Call it only when no process is inside a critical section of this region; resetting locks + /// that are legitimately held lets a writer run concurrently with them. A thread that held + /// a lock when it was reset gets from its release. + /// + /// + public void ForceResetLocks() + { + ThrowIfDisposed(); + + var header = (SharedHeader*)_basePtr; + + _logger?.LogWarning( + "Force resetting locks of '{Name}' (writer state {WriterState}, readers {Readers})", + _name, Volatile.Read(ref header->WriterLockState), Volatile.Read(ref header->ReaderCount)); + + Volatile.Write(ref header->LockOwnerProcessId, 0); + header->LockOwnerThreadId = 0; + header->LockOwnerProcessStartTime = 0; + header->LockAcquiredTimestamp = 0; + Thread.MemoryBarrier(); + Volatile.Write(ref header->WriterLockState, 0); + Interlocked.Exchange(ref header->ReaderCount, 0); + } + /// public LockOwnerInfo GetLockOwnerInfo() { @@ -1139,7 +1223,8 @@ public LockOwnerInfo GetLockOwnerInfo() ProcessId = header->LockOwnerProcessId, ThreadId = header->LockOwnerThreadId, AcquiredTimestamp = header->LockAcquiredTimestamp, - IsOrphan = IsWriteLockOrphaned() + IsOrphan = IsWriteLockOrphaned(), + ReaderCount = Volatile.Read(ref header->ReaderCount) }; } diff --git a/InterprocessMemory/MemoryRegionOptions.cs b/InterprocessMemory/MemoryRegionOptions.cs index 9d685b3..7aa01bb 100644 --- a/InterprocessMemory/MemoryRegionOptions.cs +++ b/InterprocessMemory/MemoryRegionOptions.cs @@ -20,7 +20,9 @@ public sealed class MemoryRegionOptions public static readonly TimeSpan DefaultLockTimeout = TimeSpan.FromSeconds(5); /// - /// Default orphan lock timeout: 30 seconds + /// A reasonable value to pass as when you opt in to + /// time-based lock takeover (30 seconds). It is NOT the default: + /// is (disabled) unless you set it. /// public static readonly TimeSpan DefaultOrphanLockTimeout = TimeSpan.FromSeconds(30); @@ -45,14 +47,24 @@ public sealed class MemoryRegionOptions public string? FilePath { get; set; } /// - /// Gets or sets whether to enable orphan lock detection and recovery + /// Gets or sets whether a waiting process recovers a write lock whose owner process has exited + /// (or whose PID was reused by another process) /// public bool EnableOrphanLockDetection { get; set; } = true; /// - /// Gets or sets the timeout after which a lock is considered orphaned + /// Gets or sets the time after which a write lock held by a process that is still ALIVE is + /// also treated as orphaned, so that a waiter may take it over. + /// (the default) disables this. + /// + /// A lock whose owner process has exited is always recovered while + /// is on. A time limit additionally breaks mutual + /// exclusion for a healthy owner that simply holds the lock for longer than the limit (a long + /// transaction, a paused debugger), so enable it only when every critical section is known to + /// be shorter than the value you choose. + /// /// - public TimeSpan OrphanLockTimeout { get; set; } = DefaultOrphanLockTimeout; + public TimeSpan OrphanLockTimeout { get; set; } = TimeSpan.Zero; /// /// Gets or sets whether to enable checksum verification diff --git a/InterprocessMemory/StructuredMemory.cs b/InterprocessMemory/StructuredMemory.cs index 248592e..b818821 100644 --- a/InterprocessMemory/StructuredMemory.cs +++ b/InterprocessMemory/StructuredMemory.cs @@ -787,6 +787,19 @@ public ReadLock AcquireReadLock(TimeSpan timeout) return new ReadLock(_buffer, DecrementReadLockDepth); } + /// + /// Unconditionally clears the cross-process write lock and read-lock count of this region. + /// A process that dies while holding a read lock leaves the shared reader count above zero + /// forever, which makes every later writer time out and, on Linux, survives reopening the region. + /// Call this only when no process is inside a critical section of the region. + /// See . + /// + public void ForceResetLocks() + { + ThrowIfDisposed(); + ((MemoryRegion)_buffer).ForceResetLocks(); + } + /// /// Checks if a field exists in the schema /// diff --git a/README.md b/README.md index dd279d7..f5d9c88 100644 --- a/README.md +++ b/README.md @@ -62,6 +62,10 @@ Lock ownership includes the process ID, managed thread ID, and process start tim waiting for a write lock keeps checking whether the owner is still alive (also when it waits with `Timeout.InfiniteTimeSpan`) and recovers a lock left behind by a terminated process. +Recovery is based on the owner process being gone. A write lock held by a process that is still +alive is never taken over unless you opt in with `MemoryRegionOptions.OrphanLockTimeout`, which +trades mutual exclusion for liveness when a critical section may run longer than the timeout. + The write lock belongs to the thread that acquired it. Release it on that same thread; releasing from another thread throws `SynchronizationLockException` and leaves the lock held. In particular, do not `await` between acquiring and releasing a lock, because the continuation may resume on a @@ -237,6 +241,21 @@ modifying the existing bytes. Version 3 does not migrate live 2.x regions. Stop every 2.x process, remove the named/file-backed region, and recreate it with version 3. See [MIGRATION.md](MIGRATION.md). +## Recovering after a crash + +Windows named sections disappear when their last handle closes. On Linux a region is a file in +`/dev/shm` that outlives its users, so a crash can leave state behind: + +- A process that dies while holding a **read lock** leaves the shared reader count above zero and + every writer times out. Read locks have no owner, so this cannot be detected automatically. + `GetLockOwnerInfo().ReaderCount` shows the stale count; once no process is inside a critical + section, call `MemoryRegion.ForceResetLocks()` (or `StructuredMemory.ForceResetLocks()`). +- A creator that dies during initialization makes every opener time out. A region with a different + capacity or element type than the one you now want is rejected as well. In both cases stop all + users and call `MemoryRegion.Remove(name)` (pass the same `MemoryRegionOptions` for file-backed + regions); the next `CreateOrOpen` starts from scratch. It works for every data structure because + they are all addressed by the same name. + ## Build and test ```shell From cedf8b4cd29ef06d5a9397371f94ec14aa88c777 Mon Sep 17 00:00:00 2001 From: zacala Date: Sun, 4 Oct 2026 02:20:10 +0000 Subject: [PATCH 03/29] Narrow the Dispose race with a grace period and document the contract Disposing a region while another thread is inside Read/Write or a typed queue unmaps memory that thread is still using, which terminates the process with an AccessViolationException. Measured with three threads busy polling a ConcurrentQueue on a 4-core machine, Dispose killed the process in 12 of 12 runs of 400 races each. Tracking in-flight calls would close the race but is not affordable on the lock-free paths: two interlocked operations per call made a single-thread queue round trip about 9x slower (3.3 ns -> 31 ns), two threads exchanging items about 17x slower (about 70 M -> 4 M msgs/s) and MemoryRegion.Read about 2x slower. Striping the counter per CPU did not help. Instead, MemoryRegion.Dispose now keeps the mapping for MemoryRegion.DisposeGracePeriod (default 10 ms, process-wide, Zero disables) after the instance is marked disposed. New calls already fail with ObjectDisposedException, so the pause lets calls that were past their check finish. Measured: 12 of 12 runs of 400 races survive with the default, and 8 oversubscribed pollers on 4 cores still failed in 1 of 8 runs (0 of 8 at 25 ms). It is a mitigation, not a guarantee, and says so in the XML docs and the README: the only complete protection is to stop and join every thread that uses an instance before disposing it. Tests: the grace period setting, that Dispose honours it, and a busy-poll Dispose race that kills the process without the grace period. --- .../LibraryHardeningTests.cs | 74 +++++++++++++++++++ InterprocessMemory/MemoryRegion.cs | 44 +++++++++++ README.md | 19 +++++ 3 files changed, 137 insertions(+) diff --git a/InterprocessMemory.Tests/LibraryHardeningTests.cs b/InterprocessMemory.Tests/LibraryHardeningTests.cs index fbfbcf8..5b1272c 100644 --- a/InterprocessMemory.Tests/LibraryHardeningTests.cs +++ b/InterprocessMemory.Tests/LibraryHardeningTests.cs @@ -209,6 +209,80 @@ public void StructuredMemory_ReadLockGuard_DisposedOnAnotherThread_ThrowsAndStay } } + [Test] + public void DisposeGracePeriod_DefaultsToTenMilliseconds_AndRejectsNegativeValues() + { + Assert.That(MemoryRegion.DisposeGracePeriod, Is.EqualTo(TimeSpan.FromMilliseconds(10))); + Assert.Throws( + () => MemoryRegion.DisposeGracePeriod = TimeSpan.FromMilliseconds(-1)); + } + + [Test] + public void Dispose_KeepsTheMappingForTheGracePeriod() + { + TimeSpan original = MemoryRegion.DisposeGracePeriod; + try + { + MemoryRegion.DisposeGracePeriod = TimeSpan.FromMilliseconds(100); + var withGrace = new MemoryRegion(N("Grace"), new MemoryRegionOptions { Capacity = 256 }); + var stopwatch = System.Diagnostics.Stopwatch.StartNew(); + withGrace.Dispose(); + Assert.That(stopwatch.ElapsedMilliseconds, Is.GreaterThanOrEqualTo(90)); + + MemoryRegion.DisposeGracePeriod = TimeSpan.Zero; + var withoutGrace = new MemoryRegion(N("NoGrace"), new MemoryRegionOptions { Capacity = 256 }); + stopwatch.Restart(); + withoutGrace.Dispose(); + Assert.That(stopwatch.ElapsedMilliseconds, Is.LessThan(90)); + } + finally + { + MemoryRegion.DisposeGracePeriod = original; + } + } + + [Test, Timeout(120000)] + public void Dispose_WhileThreadsBusyPollTheQueue_DoesNotCrash() + { + // The common shutdown pattern: consumers spin on TryDequeue while another thread disposes. + // Without DisposeGracePeriod this killed the process with an AccessViolationException within a + // few hundred rounds on a 4-core machine. This is a mitigation test, not a proof: a thread that + // is descheduled for longer than the grace period at the wrong moment can still fail. + var random = new Random(42); + int pollerCount = Math.Max(3, Environment.ProcessorCount - 1); + + for (int round = 0; round < 200; round++) + { + string name = N($"BusyPoll{round}"); + var queue = InterprocessMemory.ConcurrentQueue.CreateOrOpen(name, 64); + var pollers = new Task[pollerCount]; + for (int i = 0; i < pollers.Length; i++) + { + pollers[i] = Task.Run(() => + { + try + { + while (true) + { + queue.TryDequeue(out _); + queue.TryEnqueue(1); + } + } + catch (ObjectDisposedException) + { + // Expected: the queue was disposed under the poller. + } + }); + } + + Thread.Sleep(random.Next(0, 3)); + queue.Dispose(); + + Assert.That(Task.WaitAll(pollers, TimeSpan.FromSeconds(10)), Is.True, $"round {round}"); + MemoryRegion.Remove(name); + } + } + [Test] public void OrphanLockTimeout_IsDisabledByDefault() { diff --git a/InterprocessMemory/MemoryRegion.cs b/InterprocessMemory/MemoryRegion.cs index 33d3f2b..6fd7d20 100644 --- a/InterprocessMemory/MemoryRegion.cs +++ b/InterprocessMemory/MemoryRegion.cs @@ -72,6 +72,36 @@ public sealed unsafe class MemoryRegion : IMemoryRegion // Upper bound on how long Dispose() waits for lock waiters to notice the disposed flag. private static readonly TimeSpan s_waiterDrainTimeout = TimeSpan.FromSeconds(5); + // Ticks. See DisposeGracePeriod. + private static long s_disposeGraceTicks = TimeSpan.FromMilliseconds(10).Ticks; + + /// + /// How long keeps the memory mapped after the region has been marked + /// disposed, so that operations on other threads that are already past their disposed check can + /// finish before the view is unmapped (default: 10 ms; disables it). + /// This is a process-wide setting. + /// + /// It is a best-effort mitigation, not a guarantee. The lock-free members (Read, + /// Write, the typed queues, SharedArray) deliberately do no per-call bookkeeping, so + /// cannot know whether another thread is still inside one. A thread that is + /// descheduled for longer than this period at exactly that moment would touch unmapped memory, + /// which terminates the process with an . The only complete + /// protection is to stop and join every thread that uses an instance before disposing it. + /// + /// + /// The value is negative. + public static TimeSpan DisposeGracePeriod + { + get => TimeSpan.FromTicks(Interlocked.Read(ref s_disposeGraceTicks)); + set + { + if (value < TimeSpan.Zero) + throw new ArgumentOutOfRangeException(nameof(value), "DisposeGracePeriod must not be negative."); + + Interlocked.Exchange(ref s_disposeGraceTicks, value.Ticks); + } + } + // Cached once per process: stamped into the header at lock acquire so an orphan check // can distinguish "same PID, same process" from "same PID, recycled by the OS for an // unrelated process". The start time can be unreadable under restricted permissions @@ -1312,6 +1342,12 @@ private void EnterWait() /// Releases all resources used by this buffer. Threads blocked in /// or are released with an /// before the memory is unmapped. + /// + /// Stop and join every thread that uses this instance before disposing it. Other members check + /// the disposed flag but are not tracked, so a thread that is inside one of them while the + /// memory is unmapped crashes the process; narrows that window + /// but does not close it. + /// /// public void Dispose() { @@ -1328,6 +1364,14 @@ public void Dispose() while (Volatile.Read(ref _activeWaiters) > 0 && drain.Elapsed < s_waiterDrainTimeout) drainSpinner.SpinOnce(); + // Calls that passed their disposed check just before the flag was set are still running + // against the mapping and are not tracked (that would cost every call two interlocked + // operations; measured at roughly 9x on a queue round trip). New calls fail fast with + // ObjectDisposedException, so a short pause lets those in-flight calls complete. + long graceTicks = Interlocked.Read(ref s_disposeGraceTicks); + if (graceTicks > 0) + Thread.Sleep(TimeSpan.FromTicks(graceTicks)); + Cleanup(disposing: true); GC.SuppressFinalize(this); } diff --git a/README.md b/README.md index f5d9c88..236292a 100644 --- a/README.md +++ b/README.md @@ -241,6 +241,25 @@ modifying the existing bytes. Version 3 does not migrate live 2.x regions. Stop every 2.x process, remove the named/file-backed region, and recreate it with version 3. See [MIGRATION.md](MIGRATION.md). +## Disposing while other threads are running + +Stop and join every thread that uses an instance before you dispose it. The lock-free members +(`Read`, `Write`, the typed queues, `SharedArray`) read and write through a raw pointer and do no +per-call bookkeeping, because tracking calls costs every one of them two interlocked operations +(roughly 9 times slower for a queue round trip, and about 17 times slower for two threads +exchanging items). A call that is still running when the memory is unmapped terminates the process +with an `AccessViolationException`, which cannot be caught. + +What the library does to reduce the risk: + +- Calls made after `Dispose()` started fail with `ObjectDisposedException`. +- Threads waiting for a lock are released with `ObjectDisposedException`, and `Dispose()` waits for + them before it unmaps anything. +- `Dispose()` keeps the memory mapped for `MemoryRegion.DisposeGracePeriod` (10 ms by default, + process-wide, `TimeSpan.Zero` disables it) so that calls already in flight can finish. This is + best effort: a thread that is descheduled for longer than that at exactly the wrong moment still + hits unmapped memory. + ## Recovering after a crash Windows named sections disappear when their last handle closes. On Linux a region is a file in From 8e88ffca4e9b3d41cc2a0bb9aa93e94fb763ed32 Mon Sep 17 00:00:00 2001 From: zacala Date: Sun, 4 Oct 2026 04:49:30 +0000 Subject: [PATCH 04/29] Fix SharedArray leak on failed open and remove avoidable per-call cost * SharedArray did not dispose its region when the header check failed (another element type or length), so the mapping and, on Linux, its file descriptor stayed open until the finalizer ran. The other containers already clean up; do the same here. * SharedArray and StructuredMemory expose no statistics, yet their region updated its read/write counters with interlocked operations on every access. Turn the counters off for these internal regions: no observable change, SharedArray get 24 -> 18 ns and set 15.6 -> 5.9 ns. * Every automatically locked StructuredMemory call (values wider than 8 bytes, strings, blobs, arrays) allocated 104 bytes: a new delegate for the lock guard (method group conversion) plus a Stopwatch instance inside the lock wait. Cache the delegates per instance and time the lock waits with Stopwatch timestamps. Measured 104 -> 0 bytes per call and Write 311 -> 190 ns. Tests: failed open leaves no extra file descriptor (Linux), and automatically locked access allocates nothing per call. --- .../LibraryHardeningTests.cs | 64 +++++++++++++++++++ InterprocessMemory/MemoryRegion.cs | 19 +++--- InterprocessMemory/SharedArray.cs | 31 ++++++--- InterprocessMemory/StructuredMemory.cs | 22 +++++-- InterprocessMemory/TimeoutHelper.cs | 9 +++ 5 files changed, 123 insertions(+), 22 deletions(-) diff --git a/InterprocessMemory.Tests/LibraryHardeningTests.cs b/InterprocessMemory.Tests/LibraryHardeningTests.cs index 5b1272c..b05d307 100644 --- a/InterprocessMemory.Tests/LibraryHardeningTests.cs +++ b/InterprocessMemory.Tests/LibraryHardeningTests.cs @@ -283,6 +283,70 @@ public void Dispose_WhileThreadsBusyPollTheQueue_DoesNotCrash() } } + private static int CountOpenFilesMatching(string name) + { + return Directory.GetFiles("/proc/self/fd").Count(path => + { + try + { return new FileInfo(path).LinkTarget?.Contains(name) == true; } + catch (IOException) { return false; } + }); + } + + [Test] + public void SharedArray_FailedOpen_ReleasesTheMappingImmediately() + { + if (!OperatingSystem.IsLinux()) + Assert.Ignore("Counts the process's open file descriptors through /proc."); + + // Opening with another element type fails the header check after the region is mapped. + // Without an explicit dispose the file descriptor stayed open until the finalizer ran. + string name = N("ArrayLeak"); + using var owner = SharedArray.CreateOrOpen(name, 4); + int before = CountOpenFilesMatching(name); + + Assert.Throws(() => SharedArray.OpenExisting(name)); + + Assert.That(CountOpenFilesMatching(name), Is.EqualTo(before)); + } + + public struct WideSchema : IMemorySchema + { + public IEnumerable GetFields() + { + yield return FieldDefinition.Scalar("Id"); + yield return FieldDefinition.String("Name", 8); + } + } + + [Test] + public void StructuredMemory_AutoLockedAccess_DoesNotAllocatePerCall() + { + // Values wider than eight bytes and strings take the region lock automatically. Each of those + // calls used to allocate a delegate (about 104 bytes) for the lock guard. + using var memory = StructuredMemory.CreateOrOpen(N("NoAlloc"), new WideSchema()); + Guid id = Guid.NewGuid(); + + for (int i = 0; i < 2000; i++) + { + memory.Write("Id", id); + memory.Read("Id"); + memory.WriteString("Name", "ab"); + } + + const int Rounds = 10_000; + long before = GC.GetAllocatedBytesForCurrentThread(); + for (int i = 0; i < Rounds; i++) + { + memory.Write("Id", id); + memory.Read("Id"); + memory.WriteString("Name", "ab"); + } + long bytesPerCall = (GC.GetAllocatedBytesForCurrentThread() - before) / (Rounds * 3); + + Assert.That(bytesPerCall, Is.LessThan(8)); + } + [Test] public void OrphanLockTimeout_IsDisabledByDefault() { diff --git a/InterprocessMemory/MemoryRegion.cs b/InterprocessMemory/MemoryRegion.cs index 6fd7d20..91393c4 100644 --- a/InterprocessMemory/MemoryRegion.cs +++ b/InterprocessMemory/MemoryRegion.cs @@ -878,7 +878,7 @@ public bool TryAcquireWriteLock(TimeSpan timeout) private bool TryAcquireWriteLockCore(TimeSpan timeout) { var header = (SharedHeader*)_basePtr; - var sw = Stopwatch.StartNew(); + long start = Stopwatch.GetTimestamp(); // not a Stopwatch instance: that would allocate on every lock var spinner = new SpinWait(); long nextOrphanCheckMs = 0; @@ -906,7 +906,7 @@ private bool TryAcquireWriteLockCore(TimeSpan timeout) { ThrowIfDisposed(); - if (TimeoutHelper.HasExpired(sw, timeout)) + if (TimeoutHelper.HasExpired(start, timeout)) { return false; // Will release lock in finally } @@ -932,15 +932,15 @@ private bool TryAcquireWriteLockCore(TimeSpan timeout) } } - if (TimeoutHelper.HasExpired(sw, timeout)) + if (TimeoutHelper.HasExpired(start, timeout)) return false; - if (_options.EnableOrphanLockDetection && sw.ElapsedMilliseconds >= nextOrphanCheckMs) + if (_options.EnableOrphanLockDetection && ElapsedMilliseconds(start) >= nextOrphanCheckMs) { // Check on the first CAS failure and then periodically. The owner may die at any // point while we wait, and a wait with Timeout.InfiniteTimeSpan has no deadline to // key a one-off re-check on. The probe costs a process lookup, hence the interval. - nextOrphanCheckMs = sw.ElapsedMilliseconds + OrphanCheckIntervalMs; + nextOrphanCheckMs = ElapsedMilliseconds(start) + OrphanCheckIntervalMs; if (IsWriteLockOrphaned()) { @@ -1019,7 +1019,7 @@ public bool TryAcquireReadLock(TimeSpan timeout) private bool TryAcquireReadLockCore(TimeSpan timeout) { var header = (SharedHeader*)_basePtr; - var sw = Stopwatch.StartNew(); + long start = Stopwatch.GetTimestamp(); var spinner = new SpinWait(); while (true) @@ -1033,7 +1033,7 @@ private bool TryAcquireReadLockCore(TimeSpan timeout) int writerState = Volatile.Read(ref header->WriterLockState); if (writerState != 0) { - if (TimeoutHelper.HasExpired(sw, timeout)) + if (TimeoutHelper.HasExpired(start, timeout)) return false; spinner.SpinOnce(); continue; @@ -1055,7 +1055,7 @@ private bool TryAcquireReadLockCore(TimeSpan timeout) // twice extra — same penalty as the previous design's CAS-rollback path. Interlocked.Decrement(ref header->ReaderCount); - if (TimeoutHelper.HasExpired(sw, timeout)) + if (TimeoutHelper.HasExpired(start, timeout)) return false; spinner.SpinOnce(); @@ -1338,6 +1338,9 @@ private void EnterWait() private void ExitWait() => Interlocked.Decrement(ref _activeWaiters); + private static long ElapsedMilliseconds(long startTimestamp) => + (long)Stopwatch.GetElapsedTime(startTimestamp).TotalMilliseconds; + /// /// Releases all resources used by this buffer. Threads blocked in /// or are released with an diff --git a/InterprocessMemory/SharedArray.cs b/InterprocessMemory/SharedArray.cs index 87ade64..4caf1a1 100644 --- a/InterprocessMemory/SharedArray.cs +++ b/InterprocessMemory/SharedArray.cs @@ -64,24 +64,39 @@ private SharedArray(string name, int? length, bool createOrOpen) _buffer = MemoryRegion.CreateOrOpen( name, checked(ArrayHeaderSize + dataSize), - options: null, + CreateRegionOptions(), RegionKind.SharedArray); - - if (_buffer.IsOwner) - InitializeHeader(); - else - ValidateAndLoadHeader(expectedLength: _length); } else { _buffer = MemoryRegion.OpenExisting( name, - options: null, + CreateRegionOptions(), RegionKind.SharedArray); - ValidateAndLoadHeader(expectedLength: null); + } + + // The region is already mapped here. Opening an array of another element type or length + // throws from the header check, and without this the mapping and (on Linux) its file + // descriptor would stay open until the finalizer runs. + try + { + if (createOrOpen && _buffer.IsOwner) + InitializeHeader(); + else + ValidateAndLoadHeader(expectedLength: createOrOpen ? _length : null); + } + catch + { + _buffer.Dispose(); + throw; } } + // The array exposes no statistics, so the region's per-call counters would only cost an + // interlocked operation on every element access. + private static MemoryRegionOptions CreateRegionOptions() => + new() { EnableStatistics = false }; + private void InitializeHeader() { Span header = stackalloc byte[ArrayHeaderSize]; diff --git a/InterprocessMemory/StructuredMemory.cs b/InterprocessMemory/StructuredMemory.cs index b818821..6b6cec8 100644 --- a/InterprocessMemory/StructuredMemory.cs +++ b/InterprocessMemory/StructuredMemory.cs @@ -40,6 +40,11 @@ public sealed class StructuredMemory : IDisposable where TSchema : stru private readonly ThreadLocal _writeLockDepth = new(() => 0); private readonly ThreadLocal _readLockDepth = new(() => 0); + // A method group is converted to a new delegate on every use, which allocated on each + // automatically locked read or write. Convert once per instance instead. + private readonly Action _decrementWriteLockDepth; + private readonly Action _decrementReadLockDepth; + /// /// Gets the schema instance defining the memory layout /// @@ -94,6 +99,8 @@ internal StructuredMemory( _schema = schema; _compatibility = compatibility; + _decrementWriteLockDepth = DecrementWriteLockDepth; + _decrementReadLockDepth = DecrementReadLockDepth; _fields = BuildFieldMetadata(schema); if (_fields.Count == 0) @@ -107,15 +114,18 @@ internal StructuredMemory( long totalSize = SchemaHeaderSize + CalculateTotalSize(_fields); + // No statistics are exposed here, so skip the region's per-call counters. + var regionOptions = new MemoryRegionOptions { EnableStatistics = false }; + if (create) { _buffer = MemoryRegion.CreateOrOpen( - name, totalSize, options: null, RegionKind.StructuredMemory); + name, totalSize, regionOptions, RegionKind.StructuredMemory); } else { _buffer = MemoryRegion.OpenExisting( - name, options: null, RegionKind.StructuredMemory); + name, regionOptions, RegionKind.StructuredMemory); if (_buffer.Capacity != totalSize) { _buffer.Dispose(); @@ -734,14 +744,14 @@ public WriteLock AcquireWriteLock(TimeSpan timeout) if (_writeLockDepth.Value > 0) { IncrementWriteLockDepth(); - return new WriteLock(null, DecrementWriteLockDepth); + return new WriteLock(null, _decrementWriteLockDepth); } if (!_buffer.TryAcquireWriteLock(timeout)) throw new TimeoutException($"Failed to acquire write lock within {timeout}"); IncrementWriteLockDepth(); - return new WriteLock(_buffer, DecrementWriteLockDepth); + return new WriteLock(_buffer, _decrementWriteLockDepth); } /// @@ -777,14 +787,14 @@ public ReadLock AcquireReadLock(TimeSpan timeout) if (_readLockDepth.Value > 0 || _writeLockDepth.Value > 0) { IncrementReadLockDepth(); - return new ReadLock(null, DecrementReadLockDepth); + return new ReadLock(null, _decrementReadLockDepth); } if (!_buffer.TryAcquireReadLock(timeout)) throw new TimeoutException($"Failed to acquire read lock within {timeout}"); IncrementReadLockDepth(); - return new ReadLock(_buffer, DecrementReadLockDepth); + return new ReadLock(_buffer, _decrementReadLockDepth); } /// diff --git a/InterprocessMemory/TimeoutHelper.cs b/InterprocessMemory/TimeoutHelper.cs index 77a1fe3..05f4ed2 100644 --- a/InterprocessMemory/TimeoutHelper.cs +++ b/InterprocessMemory/TimeoutHelper.cs @@ -20,5 +20,14 @@ public static bool HasExpired(Stopwatch stopwatch, TimeSpan timeout) { return timeout != Timeout.InfiniteTimeSpan && stopwatch.Elapsed > timeout; } + + /// + /// Same check against a start value. Unlike a + /// instance this does not allocate, which matters on paths that run per call. + /// + public static bool HasExpired(long startTimestamp, TimeSpan timeout) + { + return timeout != Timeout.InfiniteTimeSpan && Stopwatch.GetElapsedTime(startTimestamp) > timeout; + } } } From adec0150f4259598c56c7d2f1c5ad88345093412 Mon Sep 17 00:00:00 2001 From: zacala Date: Sun, 4 Oct 2026 04:55:36 +0000 Subject: [PATCH 05/29] Remove finalizers that release nothing ConcurrentMessageQueue, SingleProducerByteStream, SharedArray and StructuredMemory had finalizers that only set a flag or disposed a MemoryHandle, which wraps an unmanaged pointer and owns nothing. They made every instance finalizable, which costs an extra GC promotion, without freeing anything: when Dispose is never called, the MemoryRegion's own finalizer already unmaps the memory. Also drop GC.SuppressFinalize from SingleProducerQueue and ConcurrentQueue, which never had a finalizer. --- InterprocessMemory/ConcurrentMessageQueue.cs | 18 ++------------- InterprocessMemory/ConcurrentQueue.cs | 1 - InterprocessMemory/SharedArray.cs | 14 +----------- .../SingleProducerByteStream.cs | 22 ++----------------- InterprocessMemory/SingleProducerQueue.cs | 1 - InterprocessMemory/StructuredMemory.cs | 11 ---------- 6 files changed, 5 insertions(+), 62 deletions(-) diff --git a/InterprocessMemory/ConcurrentMessageQueue.cs b/InterprocessMemory/ConcurrentMessageQueue.cs index e8a8d46..867ce06 100644 --- a/InterprocessMemory/ConcurrentMessageQueue.cs +++ b/InterprocessMemory/ConcurrentMessageQueue.cs @@ -633,24 +633,10 @@ public void Dispose() if (Interlocked.Exchange(ref _disposed, 1) != 0) return; + // The MemoryHandle wraps an unmanaged pointer and owns nothing, so there is no finalizer + // here: if Dispose is never called, the MemoryRegion's own finalizer unmaps the memory. _memoryHandle.Dispose(); - // See SingleProducerByteStream: only dispose the managed _buffer on the deterministic - // path. The finalizer below skips it to avoid touching peer objects whose own - // finalizers may have already run. _buffer?.Dispose(); - GC.SuppressFinalize(this); - } - - /// - /// Releases unmanaged resources if Dispose was not called. - /// - ~ConcurrentMessageQueue() - { - if (Interlocked.Exchange(ref _disposed, 1) != 0) - return; - try - { _memoryHandle.Dispose(); } - catch { /* best-effort */ } } [MethodImpl(MethodImplOptions.AggressiveInlining)] diff --git a/InterprocessMemory/ConcurrentQueue.cs b/InterprocessMemory/ConcurrentQueue.cs index 0cceb89..de10486 100644 --- a/InterprocessMemory/ConcurrentQueue.cs +++ b/InterprocessMemory/ConcurrentQueue.cs @@ -343,7 +343,6 @@ public void Dispose() return; _memoryHandle.Dispose(); _region.Dispose(); - GC.SuppressFinalize(this); } } } diff --git a/InterprocessMemory/SharedArray.cs b/InterprocessMemory/SharedArray.cs index 4caf1a1..eb837f0 100644 --- a/InterprocessMemory/SharedArray.cs +++ b/InterprocessMemory/SharedArray.cs @@ -302,20 +302,8 @@ public void Dispose() if (Interlocked.Exchange(ref _disposed, 1) != 0) return; + // No finalizer: if Dispose is never called, the MemoryRegion's own finalizer unmaps the memory. _buffer?.Dispose(); - GC.SuppressFinalize(this); - } - - /// - /// Releases unmanaged resources if Dispose was not called. Does NOT proactively dispose - /// the inner — that has its own finalizer and - /// touching it from here risks running against an already-finalized peer (finalizer - /// order is undefined). The peer's finalizer reclaims its unmanaged handles directly. - /// - ~SharedArray() - { - // Just mark disposed so a racing manual Dispose is a no-op. No managed work here. - Interlocked.Exchange(ref _disposed, 1); } } } diff --git a/InterprocessMemory/SingleProducerByteStream.cs b/InterprocessMemory/SingleProducerByteStream.cs index 2c9b93e..846b488 100644 --- a/InterprocessMemory/SingleProducerByteStream.cs +++ b/InterprocessMemory/SingleProducerByteStream.cs @@ -431,28 +431,10 @@ public void Dispose() if (Interlocked.Exchange(ref _disposed, 1) != 0) return; - // MemoryHandle wraps an unmanaged pointer — safe to dispose anywhere. + // The MemoryHandle wraps an unmanaged pointer and owns nothing, so there is no finalizer + // here: if Dispose is never called, the MemoryRegion's own finalizer unmaps the memory. _memoryHandle.Dispose(); - // _buffer (MemoryRegion) is a managed object with its OWN finalizer. - // We only proactively dispose it on the deterministic path; from our finalizer we - // let the GC handle it to avoid touching a possibly-already-finalized peer. _buffer?.Dispose(); - GC.SuppressFinalize(this); - } - - /// - /// Releases unmanaged resources if Dispose was not called. - /// Skips the managed _buffer.Dispose() — its own finalizer reclaims it. - /// - ~SingleProducerByteStream() - { - // Guard against double-dispose if Dispose() already ran. _disposed is volatile so - // we observe its current value here. - if (Interlocked.Exchange(ref _disposed, 1) != 0) - return; - try - { _memoryHandle.Dispose(); } - catch { /* unmanaged release; best-effort */ } } [MethodImpl(MethodImplOptions.AggressiveInlining)] diff --git a/InterprocessMemory/SingleProducerQueue.cs b/InterprocessMemory/SingleProducerQueue.cs index a8e0284..89cffd5 100644 --- a/InterprocessMemory/SingleProducerQueue.cs +++ b/InterprocessMemory/SingleProducerQueue.cs @@ -250,7 +250,6 @@ public void Dispose() return; _memoryHandle.Dispose(); _region.Dispose(); - GC.SuppressFinalize(this); } } } diff --git a/InterprocessMemory/StructuredMemory.cs b/InterprocessMemory/StructuredMemory.cs index 6b6cec8..c90ddc8 100644 --- a/InterprocessMemory/StructuredMemory.cs +++ b/InterprocessMemory/StructuredMemory.cs @@ -1147,17 +1147,6 @@ public void Dispose() _buffer?.Dispose(); _writeLockDepth.Dispose(); _readLockDepth.Dispose(); - GC.SuppressFinalize(this); - } - - /// - /// Releases unmanaged resources if Dispose was not called - /// - ~StructuredMemory() - { - Interlocked.Exchange(ref _disposed, 1); - // MemoryRegion owns its unmanaged finalizer path. Avoid invoking - // managed Dispose logic from this finalizer. } private static void EnsureScalarField(FieldMetadata metadata) From 576273e315b17ee6ace609cdd1abeab45133b895 Mon Sep 17 00:00:00 2001 From: zacala Date: Sun, 4 Oct 2026 04:57:00 +0000 Subject: [PATCH 06/29] Share one power-of-two rounding helper between the ring buffers SingleProducerQueue, ConcurrentQueue and ConcurrentMessageQueue each carried a private bit-twiddling copy of RoundUpToPowerOf2. Use BitOperations through one internal helper instead. Behaviour is unchanged except that an out-of-range capacity now reports the parameter name "capacity" and the offending value instead of "value", and the capacity-mismatch check computes the rounded value once. --- InterprocessMemory/ConcurrentMessageQueue.cs | 25 ++++++------------- InterprocessMemory/ConcurrentQueue.cs | 26 ++++++-------------- InterprocessMemory/PowerOfTwo.cs | 26 ++++++++++++++++++++ InterprocessMemory/SingleProducerQueue.cs | 26 ++++++-------------- 4 files changed, 50 insertions(+), 53 deletions(-) create mode 100644 InterprocessMemory/PowerOfTwo.cs diff --git a/InterprocessMemory/ConcurrentMessageQueue.cs b/InterprocessMemory/ConcurrentMessageQueue.cs index 867ce06..5d79794 100644 --- a/InterprocessMemory/ConcurrentMessageQueue.cs +++ b/InterprocessMemory/ConcurrentMessageQueue.cs @@ -194,7 +194,7 @@ private ConcurrentMessageQueue( if (createOrOpen) { - _slotCount = RoundUpToPowerOf2(capacity!.Value); + _slotCount = PowerOfTwo.RoundUp(capacity!.Value, "capacity"); _maxMessageSize = maxMessageSize!.Value; _slotTotalSize = RoundUpToMultiple( checked(SlotHeaderSize + _maxMessageSize), 8); @@ -309,10 +309,13 @@ private void ValidateAndLoadBuffer(int? requestedCapacity, int? requestedMaxMess storedMaxMessageSize <= 0 || storedSlotStride != expectedSlotStride) throw new InvalidDataException("The message queue header contains invalid sizing metadata."); - if (requestedCapacity.HasValue && - RoundUpToPowerOf2(requestedCapacity.Value) != storedSlotCount) - throw new InvalidOperationException( - $"Capacity mismatch: expected {RoundUpToPowerOf2(requestedCapacity.Value)}, found {storedSlotCount}"); + if (requestedCapacity.HasValue) + { + int expectedCapacity = PowerOfTwo.RoundUp(requestedCapacity.Value, "capacity"); + if (expectedCapacity != storedSlotCount) + throw new InvalidOperationException( + $"Capacity mismatch: expected {expectedCapacity}, found {storedSlotCount}"); + } if (requestedMaxMessageSize.HasValue && requestedMaxMessageSize.Value != storedMaxMessageSize) throw new InvalidOperationException( @@ -639,18 +642,6 @@ public void Dispose() _buffer?.Dispose(); } - [MethodImpl(MethodImplOptions.AggressiveInlining)] - private static int RoundUpToPowerOf2(int value) - { - value--; - value |= value >> 1; - value |= value >> 2; - value |= value >> 4; - value |= value >> 8; - value |= value >> 16; - return value + 1; - } - [MethodImpl(MethodImplOptions.AggressiveInlining)] private static int RoundUpToMultiple(int value, int multiple) { diff --git a/InterprocessMemory/ConcurrentQueue.cs b/InterprocessMemory/ConcurrentQueue.cs index de10486..8ac8f79 100644 --- a/InterprocessMemory/ConcurrentQueue.cs +++ b/InterprocessMemory/ConcurrentQueue.cs @@ -99,7 +99,7 @@ private ConcurrentQueue( if (createOrOpen) { - _capacity = RoundUpToPowerOf2(capacity!.Value); + _capacity = PowerOfTwo.RoundUp(capacity!.Value, "capacity"); _capacityMask = _capacity - 1; _slotStride = RoundUpToMultiple(checked(SlotHeaderSize + _elementSize), 8); long regionCapacity = checked(HeaderSize + (long)_capacity * _slotStride); @@ -179,10 +179,13 @@ private void ValidateAndLoad(int? requestedCapacity) _header->FingerprintHigh != _fingerprint.High) throw new InvalidDataException("The queue has a different format or element type."); - if (requestedCapacity.HasValue && - RoundUpToPowerOf2(requestedCapacity.Value) != storedCapacity) - throw new InvalidOperationException( - $"Capacity mismatch: expected {RoundUpToPowerOf2(requestedCapacity.Value)}, found {storedCapacity}."); + if (requestedCapacity.HasValue) + { + int expectedCapacity = PowerOfTwo.RoundUp(requestedCapacity.Value, "capacity"); + if (expectedCapacity != storedCapacity) + throw new InvalidOperationException( + $"Capacity mismatch: expected {expectedCapacity}, found {storedCapacity}."); + } long expectedRegionSize = checked(HeaderSize + (long)storedCapacity * storedStride); if (_region.Capacity != expectedRegionSize) @@ -314,19 +317,6 @@ public bool TryDequeue( Volatile.Read(ref _header->FailedDequeues)); } - private static int RoundUpToPowerOf2(int value) - { - if (value > 1 << 30) - throw new ArgumentOutOfRangeException(nameof(value)); - value--; - value |= value >> 1; - value |= value >> 2; - value |= value >> 4; - value |= value >> 8; - value |= value >> 16; - return value + 1; - } - private static int RoundUpToMultiple(int value, int multiple) => checked((value + multiple - 1) / multiple * multiple); diff --git a/InterprocessMemory/PowerOfTwo.cs b/InterprocessMemory/PowerOfTwo.cs new file mode 100644 index 0000000..5a5f392 --- /dev/null +++ b/InterprocessMemory/PowerOfTwo.cs @@ -0,0 +1,26 @@ +using System; +using System.Numerics; + +namespace InterprocessMemory +{ + internal static class PowerOfTwo + { + /// The largest power of two that fits in a positive . + public const int MaxInt32 = 1 << 30; + + /// + /// Rounds a requested item count up to the next power of two, which the ring buffers need so + /// they can wrap with a mask instead of a modulo. + /// + public static int RoundUp(int value, string paramName) + { + if (value <= 0 || value > MaxInt32) + throw new ArgumentOutOfRangeException( + paramName, + value, + $"The value must be between 1 and {MaxInt32} so it can be rounded up to a power of two."); + + return (int)BitOperations.RoundUpToPowerOf2((uint)value); + } + } +} diff --git a/InterprocessMemory/SingleProducerQueue.cs b/InterprocessMemory/SingleProducerQueue.cs index 89cffd5..0b8faf4 100644 --- a/InterprocessMemory/SingleProducerQueue.cs +++ b/InterprocessMemory/SingleProducerQueue.cs @@ -71,7 +71,7 @@ private SingleProducerQueue(string name, int? capacity, bool createOrOpen) if (createOrOpen) { - _capacity = RoundUpToPowerOf2(capacity!.Value); + _capacity = PowerOfTwo.RoundUp(capacity!.Value, "capacity"); _capacityMask = _capacity - 1; long regionCapacity = checked(HeaderSize + (long)_capacity * _elementSize); if (regionCapacity > int.MaxValue) @@ -132,10 +132,13 @@ private void ValidateAndLoad(int? requestedCapacity) _header->FingerprintHigh != _fingerprint.High) throw new InvalidDataException("The queue has a different format or element type."); - if (requestedCapacity.HasValue && - RoundUpToPowerOf2(requestedCapacity.Value) != storedCapacity) - throw new InvalidOperationException( - $"Capacity mismatch: expected {RoundUpToPowerOf2(requestedCapacity.Value)}, found {storedCapacity}."); + if (requestedCapacity.HasValue) + { + int expectedCapacity = PowerOfTwo.RoundUp(requestedCapacity.Value, "capacity"); + if (expectedCapacity != storedCapacity) + throw new InvalidOperationException( + $"Capacity mismatch: expected {expectedCapacity}, found {storedCapacity}."); + } long expectedRegionSize = checked(HeaderSize + (long)storedCapacity * _elementSize); if (_region.Capacity != expectedRegionSize) @@ -224,19 +227,6 @@ public bool TryDequeue( return true; } - private static int RoundUpToPowerOf2(int value) - { - if (value > 1 << 30) - throw new ArgumentOutOfRangeException(nameof(value)); - value--; - value |= value >> 1; - value |= value >> 2; - value |= value >> 4; - value |= value >> 8; - value |= value >> 16; - return value + 1; - } - [MethodImpl(MethodImplOptions.AggressiveInlining)] private void ThrowIfDisposed() { From be510cb3b3696ef2cd0f787791c3032519359c28 Mon Sep 17 00:00:00 2001 From: zacala Date: Sun, 4 Oct 2026 05:02:44 +0000 Subject: [PATCH 07/29] Publish SharedArray and StructuredMemory headers with proper ordering Both types wrote their header fields and then the magic with two plain region writes, and read the whole header in one copy while polling for the magic. MemoryRegion.Write/Read do no fencing, and a single copy gives no ordering between the magic and the bytes after it, so on a weakly ordered CPU (ARM) an opener could see the magic before the fields and fail with a spurious "different format" error. The queues already publish their magic last with a release write. Insert a full fence before the magic is written, and on the reader poll the 4-byte magic alone, then fence and read the fields. This is a no-op on x86, and the ordering cannot be exercised on the x86 machine used here; the existing open/create tests cover the unchanged behaviour. --- InterprocessMemory/SharedArray.cs | 14 ++++++++++++-- InterprocessMemory/StructuredMemory.cs | 12 ++++++++++-- 2 files changed, 22 insertions(+), 4 deletions(-) diff --git a/InterprocessMemory/SharedArray.cs b/InterprocessMemory/SharedArray.cs index eb837f0..f52f12f 100644 --- a/InterprocessMemory/SharedArray.cs +++ b/InterprocessMemory/SharedArray.cs @@ -107,6 +107,10 @@ private void InitializeHeader() BinaryPrimitives.WriteUInt64LittleEndian(header.Slice(16), _fingerprint.Low); BinaryPrimitives.WriteUInt64LittleEndian(header.Slice(24), _fingerprint.High); _buffer.Write(header, 0); + + // Publish the magic last. Region.Write does no fencing, so on a weakly ordered CPU another + // process could otherwise see the magic before the fields it announces. + Thread.MemoryBarrier(); BinaryPrimitives.WriteUInt32LittleEndian(header, ArrayMagic); _buffer.Write(header.Slice(0, sizeof(uint)), 0); } @@ -114,17 +118,23 @@ private void InitializeHeader() private void ValidateAndLoadHeader(int? expectedLength) { Span header = stackalloc byte[ArrayHeaderSize]; + Span magic = stackalloc byte[sizeof(uint)]; var sw = Stopwatch.StartNew(); while (true) { - _buffer.Read(header, 0); - if (BinaryPrimitives.ReadUInt32LittleEndian(header) == ArrayMagic) + _buffer.Read(magic, 0); + if (BinaryPrimitives.ReadUInt32LittleEndian(magic) == ArrayMagic) break; if (sw.Elapsed > TimeSpan.FromSeconds(5)) throw new InvalidDataException("Timed out waiting for the shared-array header."); Thread.SpinWait(100); } + // Read the fields only after the magic has been seen, never in the same copy: a single + // copy gives no ordering between the magic and the bytes that follow it. + Thread.MemoryBarrier(); + _buffer.Read(header, 0); + int version = BinaryPrimitives.ReadInt32LittleEndian(header.Slice(4)); int storedLength = BinaryPrimitives.ReadInt32LittleEndian(header.Slice(8)); int storedElementSize = BinaryPrimitives.ReadInt32LittleEndian(header.Slice(12)); diff --git a/InterprocessMemory/StructuredMemory.cs b/InterprocessMemory/StructuredMemory.cs index c90ddc8..39e6b7a 100644 --- a/InterprocessMemory/StructuredMemory.cs +++ b/InterprocessMemory/StructuredMemory.cs @@ -839,6 +839,9 @@ private void WriteSchemaHeader() BitConverter.TryWriteBytes(header.Slice(12), _schemaHash); _buffer.Write(header, 0); + + // Publish the magic last (see SharedArray.InitializeHeader). + Thread.MemoryBarrier(); BitConverter.TryWriteBytes(header, SchemaMagic); _buffer.Write(header.Slice(0, sizeof(uint)), 0); } @@ -846,11 +849,12 @@ private void WriteSchemaHeader() private void ValidateSchemaCompatibility() { Span header = stackalloc byte[SchemaHeaderSize]; + Span magic = stackalloc byte[sizeof(uint)]; var sw = Stopwatch.StartNew(); while (true) { - _buffer.Read(header, 0); - if (BitConverter.ToUInt32(header) == SchemaMagic) + _buffer.Read(magic, 0); + if (BitConverter.ToUInt32(magic) == SchemaMagic) break; if (sw.Elapsed > TimeSpan.FromSeconds(5)) throw new InvalidDataException( @@ -858,6 +862,10 @@ private void ValidateSchemaCompatibility() Thread.SpinWait(100); } + // Read the fields only after the magic has been seen (see SharedArray.ValidateAndLoadHeader). + Thread.MemoryBarrier(); + _buffer.Read(header, 0); + StoredSchemaVersion = BitConverter.ToInt32(header.Slice(4)); int storedFieldCount = BitConverter.ToInt32(header.Slice(8)); int storedHash = BitConverter.ToInt32(header.Slice(12)); From da75f6a3cb41176821a969abb9f452caa6ad4dc7 Mon Sep 17 00:00:00 2001 From: zacala Date: Sun, 4 Oct 2026 05:16:14 +0000 Subject: [PATCH 08/29] Fix two MPMC stress tests that hang or fail when run on their own * MPMC_BurstTraffic_ShouldHandleSpikes: the consumer gave up after 10,000 empty polls, which can be used up in microseconds when it starts before the producer. The producer then spun forever on a full queue and the test hung (about 45% of isolated runs). The consumer now waits for the full burst, and both sides stop with an error after 30 s instead of hanging. * MPMC_HighContention_ShouldMaintainIntegrity: same give-up heuristic, plus thread-pool starvation. Four spinning producers on Task.Run occupy every pool worker of a small machine, the consumers cannot start for seconds, and the producers give up on the full queue ("stuck at message 256", which is the slot count). It failed 25 of 25 isolated runs in a fresh process and only passed inside the full suite because earlier tests had already grown the pool. Producers and consumers now run on dedicated threads (TaskCreationOptions.LongRunning) and consumers wait for the full count with a deadline. Both pass 15 of 15 isolated runs afterwards. No library change. --- .../ConcurrentMessageQueueTests.cs | 39 +++++++++++-------- .../ExtremeStressTests.cs | 30 +++++++------- 2 files changed, 40 insertions(+), 29 deletions(-) diff --git a/InterprocessMemory.Tests/ConcurrentMessageQueueTests.cs b/InterprocessMemory.Tests/ConcurrentMessageQueueTests.cs index a475f97..20405ef 100644 --- a/InterprocessMemory.Tests/ConcurrentMessageQueueTests.cs +++ b/InterprocessMemory.Tests/ConcurrentMessageQueueTests.cs @@ -370,8 +370,11 @@ public async Task MPMC_HighContention_ShouldMaintainIntegrity() var producersDone = 0; var errors = new ConcurrentBag(); + // The producers and consumers spin, so each gets a dedicated thread. On the shared thread pool + // the four spinning producers occupy every worker of a small machine, the consumers cannot start + // for seconds (the pool adds threads slowly), and the producers give up on the full queue. // Producers - var producers = Enumerable.Range(0, producerCount).Select(producerId => Task.Run(() => + var producers = Enumerable.Range(0, producerCount).Select(producerId => Task.Factory.StartNew(() => { try { @@ -397,36 +400,40 @@ public async Task MPMC_HighContention_ShouldMaintainIntegrity() { Interlocked.Increment(ref producersDone); } - })).ToArray(); + }, TaskCreationOptions.LongRunning)).ToArray(); // Consumers - var consumers = Enumerable.Range(0, consumerCount).Select(consumerId => Task.Run(() => + var consumers = Enumerable.Range(0, consumerCount).Select(consumerId => Task.Factory.StartNew(() => { var readBuffer = new byte[64]; - int emptyReads = 0; + var deadline = System.Diagnostics.Stopwatch.StartNew(); - while (received.Count < totalExpected && emptyReads < 1000) + // Wait for the full count instead of giving up after a number of empty polls: in a fresh + // process the consumers can burn through any such budget before the producers have sent + // anything (JIT, scheduling), and the producers then block forever on a full queue. + while (received.Count < totalExpected) { + if (deadline.Elapsed > TimeSpan.FromSeconds(30)) + { + errors.Add($"Consumer {consumerId} timed out with {received.Count}/{totalExpected} received"); + return; + } + var bytesRead = buffer.TryRead(readBuffer); if (bytesRead >= 4) { received.Add(BitConverter.ToInt32(readBuffer, 0)); - emptyReads = 0; + } + else if (Volatile.Read(ref producersDone) == producerCount && buffer.ApproximateCount == 0) + { + Thread.Sleep(1); } else { - emptyReads++; - if (producersDone == producerCount && buffer.ApproximateCount == 0) - { - Thread.Sleep(1); - } - else - { - Thread.SpinWait(10); - } + Thread.SpinWait(10); } } - })).ToArray(); + }, TaskCreationOptions.LongRunning)).ToArray(); await Task.WhenAll(producers.Concat(consumers)); diff --git a/InterprocessMemory.Tests/ExtremeStressTests.cs b/InterprocessMemory.Tests/ExtremeStressTests.cs index 61795f1..ec77332 100644 --- a/InterprocessMemory.Tests/ExtremeStressTests.cs +++ b/InterprocessMemory.Tests/ExtremeStressTests.cs @@ -127,7 +127,7 @@ public async Task MPMC_BurstTraffic_ShouldHandleSpikes() for (int burst = 0; burst < burstCount; burst++) { var received = new ConcurrentBag(); - var producerDone = false; + var deadline = System.Diagnostics.Stopwatch.StartNew(); // Burst producer var producer = Task.Run(() => @@ -137,30 +137,34 @@ public async Task MPMC_BurstTraffic_ShouldHandleSpikes() { BitConverter.TryWriteBytes(data, burst * messagesPerBurst + i); while (!buffer.TryWrite(data)) + { + if (deadline.Elapsed > TimeSpan.FromSeconds(30)) + { + errors.Add($"Burst {burst}: producer blocked at message {i}"); + return; + } + Thread.SpinWait(1); + } } - producerDone = true; }); - // Consumer + // Consumer: wait for the full burst instead of giving up after a number of empty polls. + // That budget used to run out in microseconds when the consumer started before the + // producer, after which the producer spun forever on a full queue and the test hung. var consumer = Task.Run(() => { var readBuf = new byte[64]; - int emptyCount = 0; - while (received.Count < messagesPerBurst && emptyCount < 10000) + while (received.Count < messagesPerBurst) { + if (deadline.Elapsed > TimeSpan.FromSeconds(30)) + return; + var bytesRead = buffer.TryRead(readBuf); if (bytesRead > 0) - { received.Add(BitConverter.ToInt32(readBuf, 0)); - emptyCount = 0; - } else - { - emptyCount++; - if (producerDone && buffer.ApproximateCount == 0) - break; - } + Thread.SpinWait(1); } }); From 19baa34bab73a98f1dbfabadab7b7fce0a7568bd Mon Sep 17 00:00:00 2001 From: zacala Date: Mon, 5 Oct 2026 14:19:16 +0000 Subject: [PATCH 09/29] Add cross-process locking to SharedArray Elements wider than 8 bytes, or whose size is not a power of two, could be read torn while another process wrote them: about 1% of reads of a 64-byte element in a two-thread stress test (233,928 of 23.7 M). Nothing exposed the shared region lock, so users had no way to prevent it. StructuredMemory already locks such values; SharedArray did not. * Elements of 1, 2, 4 or 8 bytes are copied with one aligned move and keep the lock-free path (a per-T constant, so the JIT drops the branch). Every other size is read and written under the shared region lock by the indexer, CopyTo, CopyFrom and Fill (a Fill takes it once for the range). The torn-read reproduction now shows 0 torn reads. * AcquireReadLock/AcquireWriteLock([timeout]) for any element type, to read or change several elements as one unit. They are reentrant for the calling thread, so the indexer inside a lock does not take it again, and writing while holding only a read lock throws InvalidOperationException, as in StructuredMemory. The guards are ref structs: the lock belongs to one thread, and the compiler now rejects holding a guard in an async method (CS9104), which is the misuse that leaked write locks before. * ForceResetLocks(), like MemoryRegion and StructuredMemory, to clear the lock state a crashed process left behind. * The indexer no longer uses stackalloc (the JIT cannot inline methods that do), reading through a span over a local instead. Behaviour change: arrays of wide or odd-sized elements now take a lock per element access, about ten times fewer reads per second in the stress test, in exchange for correct values. Lock-free element types are unaffected (SharedArray get/set are within measurement noise of before). Processes running a version without this locking do not take part in it. Tests cover torn reads, atomic types bypassing the lock, exclusion between instances, reentrancy, read-to-write upgrade rejection, consistent snapshots of several elements, double dispose of guards, ForceResetLocks, and round trips of 64-byte and 3-byte elements through every access path. --- InterprocessMemory.Tests/SharedArrayTests.cs | 277 ++++++++++++++++ InterprocessMemory/SharedArray.cs | 323 ++++++++++++++++++- README.md | 35 +- 3 files changed, 620 insertions(+), 15 deletions(-) diff --git a/InterprocessMemory.Tests/SharedArrayTests.cs b/InterprocessMemory.Tests/SharedArrayTests.cs index 9e25e10..10976f9 100644 --- a/InterprocessMemory.Tests/SharedArrayTests.cs +++ b/InterprocessMemory.Tests/SharedArrayTests.cs @@ -272,4 +272,281 @@ private struct TestPoint public int X; public int Y; } + + // ── Cross-process locking ──────────────────────────────────────────────── + + public struct Wide64 { public long A, B, C, D, E, F, G, H; } + + // Three bytes: not a power of two, so a single aligned move cannot copy it atomically. + public struct Odd3 { public byte X, Y, Z; } + + private static string LockName(string tag) => $"SharedArrayLock_{tag}_{Guid.NewGuid():N}"; + + private static Wide64 Wide(long v) => new() { A = v, B = v, C = v, D = v, E = v, F = v, G = v, H = v }; + + private static bool IsWhole(Wide64 w) => + w.A == w.B && w.B == w.C && w.C == w.D && w.D == w.E && w.E == w.F && w.F == w.G && w.G == w.H; + + [Test, Timeout(60000)] + public void WideElements_AreNeverObservedTorn_AcrossInstances() + { + // Without the automatic lock about 1% of reads of a 64-byte element were torn in this setup. + string name = LockName("Torn"); + using var writerArray = SharedArray.CreateOrOpen(name, 4); + using var readerArray = SharedArray.OpenExisting(name); + using var stop = new CancellationTokenSource(TimeSpan.FromSeconds(1.5)); + long reads = 0; + long torn = 0; + + var writer = Task.Factory.StartNew(() => + { + long i = 0; + while (!stop.IsCancellationRequested) + writerArray[0] = Wide(++i); + }, TaskCreationOptions.LongRunning); + + var reader = Task.Factory.StartNew(() => + { + while (!stop.IsCancellationRequested) + { + if (!IsWhole(readerArray[0])) + torn++; + reads++; + } + }, TaskCreationOptions.LongRunning); + + Task.WaitAll(writer, reader); + + Assert.That(reads, Is.GreaterThan(0)); + Assert.That(torn, Is.EqualTo(0), $"{torn} of {reads} reads were torn"); + } + + [Test, Timeout(60000)] + public void AtomicElements_BypassTheLock_WideElementsDoNot() + { + string atomicName = LockName("Atomic"); + using var atomic = SharedArray.CreateOrOpen(atomicName, 4); + using var atomicPeer = SharedArray.OpenExisting(atomicName); + + string wideName = LockName("Wide"); + using var wide = SharedArray.CreateOrOpen(wideName, 4); + using var widePeer = SharedArray.OpenExisting(wideName); + + Task blockedRead; + using (atomic.AcquireWriteLock()) + using (wide.AcquireWriteLock()) + { + // A long is copied with one aligned move, so reading it needs no lock and does not wait. + Assert.That(Task.Run(() => atomicPeer[0]).Wait(TimeSpan.FromSeconds(2)), Is.True); + + // A 64-byte struct could be torn, so its reader waits for the writer to finish. + blockedRead = Task.Run(() => widePeer[0]); + Assert.That(blockedRead.Wait(TimeSpan.FromMilliseconds(300)), Is.False); + } + + Assert.That(blockedRead.Wait(TimeSpan.FromSeconds(5)), Is.True); + } + + [Test, Timeout(30000)] + public void ExplicitWriteLock_ExcludesOtherInstances_ForAnyElementType() + { + string name = LockName("Exclude"); + using var owner = SharedArray.CreateOrOpen(name, 4); + using var other = SharedArray.OpenExisting(name); + + using (owner.AcquireWriteLock()) + { + Assert.Throws(() => other.AcquireWriteLock(TimeSpan.FromMilliseconds(100))); + Assert.Throws(() => other.AcquireReadLock(TimeSpan.FromMilliseconds(100))); + } + + using (other.AcquireWriteLock(TimeSpan.FromSeconds(1))) + { + } + + // Readers share the lock. + using (owner.AcquireReadLock(TimeSpan.FromSeconds(1))) + using (other.AcquireReadLock(TimeSpan.FromSeconds(1))) + { + } + } + + [Test, Timeout(30000)] + public void Locks_AreReentrant_AndTheIndexerDoesNotTakeThemAgain() + { + string name = LockName("Reentrant"); + using var array = SharedArray.CreateOrOpen(name, 8); + using var peer = SharedArray.OpenExisting(name); + + using (array.AcquireWriteLock()) + { + array[0] = Wide(1); + Assert.That(IsWhole(array[0]), Is.True); + array.CopyFrom(1, new[] { Wide(2), Wide(3) }); + array.Fill(Wide(9), 3, 2); + + using (array.AcquireWriteLock()) + using (array.AcquireReadLock()) + { + array[7] = Wide(7); + Assert.That(array[7].A, Is.EqualTo(7)); + } + + // The inner guards released only their own level: the outer lock is still exclusive. + Assert.Throws(() => peer.AcquireReadLock(TimeSpan.FromMilliseconds(100))); + } + + using (peer.AcquireWriteLock(TimeSpan.FromSeconds(1))) + { + Assert.That(peer[7].A, Is.EqualTo(7)); + } + } + + [Test, Timeout(30000)] + public void WritingWhileHoldingOnlyAReadLock_Throws() + { + using var narrow = SharedArray.CreateOrOpen(LockName("Upgrade"), 4); + using var wide = SharedArray.CreateOrOpen(LockName("UpgradeWide"), 4); + + using (narrow.AcquireReadLock()) + { + // Upgrading would wait for this very thread to release its read lock. + Assert.Throws(() => narrow.AcquireWriteLock()); + } + + using (wide.AcquireReadLock()) + { + Assert.Throws(() => wide[0] = Wide(1)); + Assert.Throws(() => wide.CopyFrom(0, new[] { Wide(1) })); + Assert.That(IsWhole(wide[0]), Is.True); + } + + // Nothing was left behind by the rejected attempts. + using (narrow.AcquireWriteLock(TimeSpan.FromSeconds(1))) + { + } + + wide[0] = Wide(5); + Assert.That(wide[0].A, Is.EqualTo(5)); + } + + [Test, Timeout(60000)] + public void ExplicitLocks_GiveConsistentSnapshotsOfSeveralElements() + { + // Two longs are each atomic, but a reader can still see one of them from before and the other + // from after an update. The explicit locks make the pair one unit. + string name = LockName("Snapshot"); + using var writerArray = SharedArray.CreateOrOpen(name, 2); + using var readerArray = SharedArray.OpenExisting(name); + using var stop = new CancellationTokenSource(TimeSpan.FromSeconds(1.5)); + long snapshots = 0; + long mismatches = 0; + + var writer = Task.Factory.StartNew(() => + { + long i = 0; + while (!stop.IsCancellationRequested) + { + i++; + using var guard = writerArray.AcquireWriteLock(); + writerArray[0] = i; + writerArray[1] = i; + } + }, TaskCreationOptions.LongRunning); + + var reader = Task.Factory.StartNew(() => + { + while (!stop.IsCancellationRequested) + { + long first; + long second; + using (readerArray.AcquireReadLock()) + { + first = readerArray[0]; + second = readerArray[1]; + } + + if (first != second) + mismatches++; + snapshots++; + } + }, TaskCreationOptions.LongRunning); + + Task.WaitAll(writer, reader); + + Assert.That(snapshots, Is.GreaterThan(0)); + Assert.That(mismatches, Is.EqualTo(0), $"{mismatches} of {snapshots} snapshots were inconsistent"); + } + + [Test] + public void Guards_AreHarmlessWhenDisposedTwice() + { + string name = LockName("DoubleDispose"); + using var array = SharedArray.CreateOrOpen(name, 4); + using var peer = SharedArray.OpenExisting(name); + + var guard = array.AcquireWriteLock(); + guard.Dispose(); + guard.Dispose(); + default(SharedArray.WriteLock).Dispose(); + default(SharedArray.ReadLock).Dispose(); + + // A second release must not have freed anything it did not hold or unbalanced the bookkeeping. + using (peer.AcquireWriteLock(TimeSpan.FromSeconds(1))) + { + } + + using (array.AcquireWriteLock(TimeSpan.FromSeconds(1))) + { + array[0] = 1; + } + } + + [Test] + public void ForceResetLocks_UnblocksWritersAfterALostReader() + { + string name = LockName("Reset"); + using var array = SharedArray.CreateOrOpen(name, 4); + using var peer = SharedArray.OpenExisting(name); + + // A reader that never comes back, as after a crash. + var lostReader = peer.AcquireReadLock(); + Assert.Throws(() => array.AcquireWriteLock(TimeSpan.FromMilliseconds(100))); + + array.ForceResetLocks(); + + using (array.AcquireWriteLock(TimeSpan.FromSeconds(1))) + { + } + + lostReader.Dispose(); + } + + [Test] + public void WideAndOddSizedElements_RoundTripThroughEveryAccessPath() + { + using var wide = SharedArray.CreateOrOpen(LockName("RoundTripWide"), 10); + wide[3] = Wide(33); + Assert.That(wide[3].H, Is.EqualTo(33)); + + wide.CopyFrom(4, new[] { Wide(4), Wide(5) }); + var copy = new Wide64[2]; + wide.CopyTo(4, copy); + Assert.That(copy[0].A, Is.EqualTo(4)); + Assert.That(copy[1].A, Is.EqualTo(5)); + + wide.Fill(Wide(8), 6, 4); + Assert.That(wide[9].A, Is.EqualTo(8)); + Assert.That(wide[5].A, Is.EqualTo(5), "the range before the fill is untouched"); + Assert.That(wide[0].A, Is.EqualTo(0), "an element nobody wrote is still zero"); + + wide.Clear(); + Assert.That(wide[3].A, Is.EqualTo(0)); + + using var odd = SharedArray.CreateOrOpen(LockName("RoundTripOdd"), 5); + odd[2] = new Odd3 { X = 1, Y = 2, Z = 3 }; + Assert.That(odd[2].Z, Is.EqualTo(3)); + odd.Fill(new Odd3 { X = 9, Y = 9, Z = 9 }, 0, 5); + Assert.That(odd[4].Y, Is.EqualTo(9)); + } } diff --git a/InterprocessMemory/SharedArray.cs b/InterprocessMemory/SharedArray.cs index f52f12f..33452cc 100644 --- a/InterprocessMemory/SharedArray.cs +++ b/InterprocessMemory/SharedArray.cs @@ -13,6 +13,18 @@ namespace InterprocessMemory /// High-performance generic shared array with type safety and zero-allocation indexer. /// Provides array-like access to shared memory with compile-time type checking. /// Cross-platform: backed by which supports Windows and Linux. + /// + /// Atomicity. Elements whose size is 1, 2, 4 or 8 bytes are copied with a single aligned + /// move, so another process never observes half of one, and they are accessed without any lock. + /// Every other element size (a 16-byte , a 64-byte struct, ...) could be read + /// torn while another process writes it, so the indexer, , + /// and take the shared region lock for those types. + /// + /// + /// Use / when several elements + /// must be observed or changed together, for any element type. The guards belong to the calling + /// thread, may be nested, and cannot be held across an await (they are ref structs). + /// /// /// Unmanaged value type public sealed class SharedArray : IDisposable where T : unmanaged @@ -27,6 +39,14 @@ public sealed class SharedArray : IDisposable where T : unmanaged private readonly TypeLayoutFingerprint _fingerprint; private volatile int _disposed; + // True for element sizes that one aligned move cannot copy atomically. Per-T constant, so the + // JIT removes the locking branch from the hot path of 1/2/4/8-byte element types. + private static readonly bool s_needsLock = !(Unsafe.SizeOf() is 1 or 2 or 4 or 8); + + // Reentrancy bookkeeping, per thread: a thread that holds the lock must not take it again. + private readonly ThreadLocal _writeLockDepth = new(() => 0); + private readonly ThreadLocal _readLockDepth = new(() => 0); + /// /// Gets the number of elements in the array /// @@ -160,8 +180,13 @@ private void ValidateAndLoadHeader(int? expectedLength) /// /// Gets or sets the element at the specified index. - /// Zero-allocation accessor using direct memory access. + /// Zero-allocation accessor using direct memory access. Element types other than 1, 2, 4 or 8 + /// bytes wide are read and written under the shared region lock (see the class remarks). /// + /// The automatic lock could not be taken within 5 seconds. + /// + /// A wide element is written while this thread holds only a read lock. + /// public T this[int index] { [MethodImpl(MethodImplOptions.AggressiveInlining)] @@ -171,9 +196,7 @@ public T this[int index] if ((uint)index >= (uint)_length) throw new IndexOutOfRangeException(); - Span buffer = stackalloc byte[_elementSize]; - _buffer.Read(buffer, ArrayHeaderSize + (long)index * _elementSize); - return MemoryMarshal.Read(buffer); + return s_needsLock && !IsHoldingAnyLock() ? ReadElementLocked(index) : ReadElement(index); } [MethodImpl(MethodImplOptions.AggressiveInlining)] set @@ -182,14 +205,57 @@ public T this[int index] if ((uint)index >= (uint)_length) throw new IndexOutOfRangeException(); - ReadOnlySpan buffer = MemoryMarshal.AsBytes(MemoryMarshal.CreateReadOnlySpan(ref value, 1)); - _buffer.Write(buffer, ArrayHeaderSize + (long)index * _elementSize); + if (s_needsLock && !IsHoldingWriteLock()) + WriteElementLocked(index, value); + else + WriteElement(index, value); + } + } + + private T ReadElement(int index) + { + T value = default; + _buffer.Read(MemoryMarshal.AsBytes(MemoryMarshal.CreateSpan(ref value, 1)), + ArrayHeaderSize + (long)index * _elementSize); + return value; + } + + private void WriteElement(int index, T value) + { + _buffer.Write(MemoryMarshal.AsBytes(MemoryMarshal.CreateReadOnlySpan(ref value, 1)), + ArrayHeaderSize + (long)index * _elementSize); + } + + private T ReadElementLocked(int index) + { + bool tookRegionLock = EnterRead(MemoryRegionOptions.DefaultLockTimeout); + try + { + return ReadElement(index); + } + finally + { + ExitRead(tookRegionLock); + } + } + + private void WriteElementLocked(int index, T value) + { + bool tookRegionLock = EnterWrite(MemoryRegionOptions.DefaultLockTimeout); + try + { + WriteElement(index, value); + } + finally + { + ExitWrite(tookRegionLock); } } /// /// Copies a range of elements to a span. - /// High-performance batch operation with SIMD optimization. + /// High-performance batch operation with SIMD optimization. For element types other than + /// 1, 2, 4 or 8 bytes wide the range is read under the shared region lock. /// /// Starting index in the array /// Destination span to copy elements to @@ -205,13 +271,30 @@ public void CopyTo(int startIndex, Span destination) if (startIndex < 0 || (long)startIndex + destination.Length > _length) throw new ArgumentOutOfRangeException(nameof(startIndex)); - var byteSpan = MemoryMarshal.AsBytes(destination); - _buffer.Read(byteSpan, ArrayHeaderSize + (long)startIndex * _elementSize); + if (!s_needsLock || IsHoldingAnyLock()) + { + CopyToCore(startIndex, destination); + return; + } + + bool tookRegionLock = EnterRead(MemoryRegionOptions.DefaultLockTimeout); + try + { + CopyToCore(startIndex, destination); + } + finally + { + ExitRead(tookRegionLock); + } } + private void CopyToCore(int startIndex, Span destination) => + _buffer.Read(MemoryMarshal.AsBytes(destination), ArrayHeaderSize + (long)startIndex * _elementSize); + /// /// Copies a span of elements to the array. - /// High-performance batch operation with SIMD optimization. + /// High-performance batch operation with SIMD optimization. For element types other than + /// 1, 2, 4 or 8 bytes wide the range is written under the shared region lock. /// /// Starting index in the array /// Source span to copy elements from @@ -224,13 +307,30 @@ public void CopyFrom(int startIndex, ReadOnlySpan source) if (startIndex < 0 || (long)startIndex + source.Length > _length) throw new ArgumentOutOfRangeException(nameof(startIndex)); - var byteSpan = MemoryMarshal.AsBytes(source); - _buffer.Write(byteSpan, ArrayHeaderSize + (long)startIndex * _elementSize); + if (!s_needsLock || IsHoldingWriteLock()) + { + CopyFromCore(startIndex, source); + return; + } + + bool tookRegionLock = EnterWrite(MemoryRegionOptions.DefaultLockTimeout); + try + { + CopyFromCore(startIndex, source); + } + finally + { + ExitWrite(tookRegionLock); + } } + private void CopyFromCore(int startIndex, ReadOnlySpan source) => + _buffer.Write(MemoryMarshal.AsBytes(source), ArrayHeaderSize + (long)startIndex * _elementSize); + /// /// Fills a range with a value. - /// Optimized for large ranges using vectorization. + /// Optimized for large ranges using vectorization. For element types other than 1, 2, 4 or + /// 8 bytes wide the whole range is written under one shared region lock. /// /// Value to fill with /// Starting index (default: 0) @@ -247,6 +347,25 @@ public void Fill(T value, int startIndex = 0, int count = -1) if (startIndex < 0 || count < 0 || (long)startIndex + count > _length) throw new ArgumentOutOfRangeException(); + if (!s_needsLock || IsHoldingWriteLock()) + { + FillCore(value, startIndex, count); + return; + } + + bool tookRegionLock = EnterWrite(MemoryRegionOptions.DefaultLockTimeout); + try + { + FillCore(value, startIndex, count); + } + finally + { + ExitWrite(tookRegionLock); + } + } + + private void FillCore(T value, int startIndex, int count) + { // Batch fill: create a filled buffer and write in chunks int batchCount = Math.Min(count, 4096); int batchBytes = batchCount * _elementSize; @@ -292,11 +411,185 @@ private void FillBatched(int startIndex, int count, Span batch) while (offset < count) { int batchSize = Math.Min(batch.Length, count - offset); - CopyFrom(startIndex + offset, batch.Slice(0, batchSize)); + CopyFromCore(startIndex + offset, batch.Slice(0, batchSize)); offset += batchSize; } } + /// + /// Acquires the shared write lock (all processes) with the default timeout of 5 seconds. + /// See . + /// + public WriteLock AcquireWriteLock() => AcquireWriteLock(MemoryRegionOptions.DefaultLockTimeout); + + /// + /// Acquires the shared write lock, which excludes every other reader and writer that uses a lock, + /// in all processes, until the returned guard is disposed. Use it to change several elements as one + /// unit. The lock is reentrant for the calling thread; the indexer and the range operations inside + /// the scope do not take it again. + /// + /// How long to wait; waits indefinitely. + /// The lock could not be acquired within the timeout. + /// + /// The calling thread holds only a read lock; upgrading it would deadlock the thread against itself. + /// + public WriteLock AcquireWriteLock(TimeSpan timeout) => new(this, EnterWrite(timeout)); + + /// + /// Acquires the shared read lock (all processes) with the default timeout of 5 seconds. + /// See . + /// + public ReadLock AcquireReadLock() => AcquireReadLock(MemoryRegionOptions.DefaultLockTimeout); + + /// + /// Acquires the shared read lock, which excludes writers but not other readers, in all processes, + /// until the returned guard is disposed. Use it to observe several elements as one consistent + /// snapshot. It is reentrant for the calling thread, also inside a write lock. + /// + /// How long to wait; waits indefinitely. + /// The lock could not be acquired within the timeout. + public ReadLock AcquireReadLock(TimeSpan timeout) => new(this, EnterRead(timeout)); + + /// + /// Unconditionally clears the cross-process write lock and read-lock count of this array's region. + /// A process that dies while holding a lock can leave the reader count above zero forever. Call this + /// only when no process is inside a critical section of the array. See + /// . + /// + public void ForceResetLocks() + { + ThrowIfDisposed(); + ((MemoryRegion)_buffer).ForceResetLocks(); + } + + [MethodImpl(MethodImplOptions.AggressiveInlining)] + private bool IsHoldingAnyLock() => _writeLockDepth.Value > 0 || _readLockDepth.Value > 0; + + [MethodImpl(MethodImplOptions.AggressiveInlining)] + private bool IsHoldingWriteLock() => _writeLockDepth.Value > 0; + + /// Returns true when the region lock was taken, false for a reentrant acquisition. + private bool EnterWrite(TimeSpan timeout) + { + ThrowIfDisposed(); + TimeoutHelper.Validate(timeout, nameof(timeout)); + + if (_writeLockDepth.Value == 0 && _readLockDepth.Value > 0) + throw new InvalidOperationException( + "Cannot take the write lock while holding only a read lock. Release the read lock first, " + + "or take the write lock before reading and writing together."); + + bool tookRegionLock = false; + if (_writeLockDepth.Value == 0) + { + if (!_buffer.TryAcquireWriteLock(timeout)) + throw new TimeoutException($"Failed to acquire write lock within {timeout}"); + tookRegionLock = true; + } + + _writeLockDepth.Value++; + return tookRegionLock; + } + + private void ExitWrite(bool tookRegionLock) + { + try + { + if (tookRegionLock) + _buffer.ReleaseWriteLock(); + } + finally + { + _writeLockDepth.Value--; + } + } + + /// Returns true when the region lock was taken, false for a reentrant acquisition. + private bool EnterRead(TimeSpan timeout) + { + ThrowIfDisposed(); + TimeoutHelper.Validate(timeout, nameof(timeout)); + + bool tookRegionLock = false; + if (_readLockDepth.Value == 0 && _writeLockDepth.Value == 0) + { + if (!_buffer.TryAcquireReadLock(timeout)) + throw new TimeoutException($"Failed to acquire read lock within {timeout}"); + tookRegionLock = true; + } + + _readLockDepth.Value++; + return tookRegionLock; + } + + private void ExitRead(bool tookRegionLock) + { + try + { + if (tookRegionLock) + _buffer.ReleaseReadLock(); + } + finally + { + _readLockDepth.Value--; + } + } + + /// + /// Guard returned by ; disposing it releases the lock. It is a ref + /// struct on purpose: the lock belongs to one thread, and the compiler therefore rejects holding the + /// guard across an await or handing it to another thread. + /// + public ref struct WriteLock + { + private SharedArray? _owner; + private readonly bool _tookRegionLock; + + internal WriteLock(SharedArray owner, bool tookRegionLock) + { + _owner = owner; + _tookRegionLock = tookRegionLock; + } + + /// Releases the write lock; disposing more than once has no further effect. + public void Dispose() + { + SharedArray? owner = _owner; + if (owner is null) + return; + + _owner = null; + owner.ExitWrite(_tookRegionLock); + } + } + + /// + /// Guard returned by ; disposing it releases the lock. A ref struct + /// for the same reason as . + /// + public ref struct ReadLock + { + private SharedArray? _owner; + private readonly bool _tookRegionLock; + + internal ReadLock(SharedArray owner, bool tookRegionLock) + { + _owner = owner; + _tookRegionLock = tookRegionLock; + } + + /// Releases the read lock; disposing more than once has no further effect. + public void Dispose() + { + SharedArray? owner = _owner; + if (owner is null) + return; + + _owner = null; + owner.ExitRead(_tookRegionLock); + } + } + [MethodImpl(MethodImplOptions.AggressiveInlining)] private void ThrowIfDisposed() { @@ -314,6 +607,8 @@ public void Dispose() // No finalizer: if Dispose is never called, the MemoryRegion's own finalizer unmaps the memory. _buffer?.Dispose(); + _writeLockDepth.Dispose(); + _readLockDepth.Dispose(); } } } diff --git a/README.md b/README.md index 236292a..fe1a03b 100644 --- a/README.md +++ b/README.md @@ -232,6 +232,39 @@ Console.WriteLine(reader[0]); The array header validates the element type fingerprint and restores its length for openers. +Elements that are 1, 2, 4 or 8 bytes wide are copied with a single aligned move, so another process +never sees half of one, and they are read and written without any lock. Every other element size +(a `Guid`, a `Vector3`, a 64-byte struct, ...) could be read torn while another process writes it, +so `SharedArray` takes the shared region lock for those types in the indexer, `CopyTo`, `CopyFrom` +and `Fill` (the whole range of a `Fill` under one lock). That is correct but slower per access; for +many elements take the lock once yourself. + +Use the explicit locks when several elements must be read or changed together. This applies to +any element type, because two atomic elements can still be seen from different updates: + +```csharp +using (array.AcquireWriteLock()) +{ + array[0] = x; + array[1] = y; // other processes see both changes or neither +} + +using (reader.AcquireReadLock()) +{ + long first = reader[0]; + long second = reader[1]; // a consistent pair +} +``` + +The locks are shared by every process and thread that uses a lock, reentrant for the calling thread +(the indexer inside a lock does not take it again), and time out with `TimeoutException`. Taking the +write lock while the thread holds only a read lock throws `InvalidOperationException`. The guards are +`ref struct`s, because a lock belongs to one thread and must never be held across an `await`: the +compiler rejects a guard as a `using` resource in an `async` method (error CS9104), so do the locked +work in a small synchronous method and call that. `array.ForceResetLocks()` clears a lock state left +behind by a crash (see below). Processes still running a version without +this locking do not take part in it. + ## Version 3 format Every region has a version 3 magic value, format version, and data-structure kind. Opening a @@ -268,7 +301,7 @@ Windows named sections disappear when their last handle closes. On Linux a regio - A process that dies while holding a **read lock** leaves the shared reader count above zero and every writer times out. Read locks have no owner, so this cannot be detected automatically. `GetLockOwnerInfo().ReaderCount` shows the stale count; once no process is inside a critical - section, call `MemoryRegion.ForceResetLocks()` (or `StructuredMemory.ForceResetLocks()`). + section, call `MemoryRegion.ForceResetLocks()` (or the same method on `StructuredMemory` or `SharedArray`). - A creator that dies during initialization makes every opener time out. A region with a different capacity or element type than the one you now want is rejected as well. In both cases stop all users and call `MemoryRegion.Remove(name)` (pass the same `MemoryRegionOptions` for file-backed From b0b52ddae77d7e9bde5818102055a63152eeb3f5 Mon Sep 17 00:00:00 2001 From: zacala Date: Mon, 5 Oct 2026 14:33:47 +0000 Subject: [PATCH 10/29] Add Linux/Windows CI matrix Build and test on ubuntu-latest and windows-latest for pushes to main and for pull requests. The gating test step skips tests tagged TimingSensitive (a scheduling-fairness assertion) and LongRunning (the [Explicit] soak tests); the timing-sensitive ones run in a separate informational step so a slow shared runner does not fail the build. TRX results are uploaded for every run. --- .github/workflows/ci.yml | 95 +++++++++++++++++++ .../ConcurrencyStabilityTests.cs | 5 + .../ExtremeStressTests.cs | 2 + 3 files changed, 102 insertions(+) create mode 100644 .github/workflows/ci.yml diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..d9ce129 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,95 @@ +name: CI + +on: + push: + branches: [main] + pull_request: + workflow_dispatch: + +permissions: + contents: read + +# A newer push to the same branch or pull request supersedes the run in progress. +concurrency: + group: ci-${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: true + +env: + DOTNET_NOLOGO: true + DOTNET_CLI_TELEMETRY_OPTOUT: true + +jobs: + test: + name: Build and test (${{ matrix.os }}) + runs-on: ${{ matrix.os }} + timeout-minutes: 30 + + strategy: + # Report both platforms even when one fails: the library keeps its state differently on each + # (Windows named sections vanish with their last handle, Linux files in /dev/shm persist). + fail-fast: false + matrix: + os: [ubuntu-latest, windows-latest] + + defaults: + run: + # One shell on both platforms keeps quoting of the filter expressions identical. + shell: bash + + steps: + - name: Check out + uses: actions/checkout@v4 + + - name: Set up .NET + uses: actions/setup-dotnet@v4 + with: + dotnet-version: 8.0.x + + - name: Cache NuGet packages + uses: actions/cache@v4 + with: + path: ~/.nuget/packages + key: nuget-${{ runner.os }}-${{ hashFiles('**/*.csproj') }} + restore-keys: nuget-${{ runner.os }}- + + - name: Restore + run: dotnet restore InterprocessMemory.sln + + # Builds every project, including the benchmarks and the child-process test worker that the + # cross-process tests start. Packing on every build is not needed here. + - name: Build + run: dotnet build InterprocessMemory.sln --configuration Release --no-restore -p:GeneratePackageOnBuild=false + + # --blame names the running test when the test host dies, which is what an + # AccessViolationException from a memory-mapped region looks like; the hang timeout stops a + # stuck test from holding the runner until the job timeout. + - name: Test + run: > + dotnet test InterprocessMemory.Tests/InterprocessMemory.Tests.csproj + --configuration Release --no-build + --filter "Category!=TimingSensitive&Category!=LongRunning" + --blame-hang-timeout 10m --blame-hang-dump-type none + --logger "trx;LogFileName=test-results.trx" + --results-directory TestResults + + # Tests marked LongRunning (minutes each) are explicit-only and are left to manual runs. + # + # Tests whose outcome depends on the core count and the scheduler. They still run and show up + # in the log and the artifact, but a failure here does not fail the job. + - name: Test (timing sensitive, informational) + continue-on-error: true + run: > + dotnet test InterprocessMemory.Tests/InterprocessMemory.Tests.csproj + --configuration Release --no-build + --filter "Category=TimingSensitive" + --blame-hang-timeout 5m --blame-hang-dump-type none + --logger "trx;LogFileName=timing-sensitive-results.trx" + --results-directory TestResults + + - name: Upload test results + if: always() + uses: actions/upload-artifact@v4 + with: + name: test-results-${{ matrix.os }} + path: TestResults + if-no-files-found: ignore diff --git a/InterprocessMemory.Tests/ConcurrencyStabilityTests.cs b/InterprocessMemory.Tests/ConcurrencyStabilityTests.cs index 43e1f39..0ad16c7 100644 --- a/InterprocessMemory.Tests/ConcurrencyStabilityTests.cs +++ b/InterprocessMemory.Tests/ConcurrencyStabilityTests.cs @@ -285,6 +285,7 @@ public async Task StabilityA_OrphanCheckUnderLegitimateLockUse_NoFalsePositive() [Test] [Timeout(150000)] [Explicit("Long-running test — 2 min sustained SPSC")] + [Category("LongRunning")] public async Task Stability_SPSC_2Minutes_OrderPreserved() { // The existing Stability_MPMC_2Minutes_Continuous covers MPMC. SPSC has different @@ -350,6 +351,7 @@ public async Task Stability_SPSC_2Minutes_OrderPreserved() [Test] [Timeout(150000)] [Explicit("Long-running test — 2 min sustained Strict mixed access")] + [Category("LongRunning")] public async Task Stability_Strict_2Minutes_MixedAccessNoLockLeak() { // Strict's reentrant lock + auto-lock on >8-byte types is intricate. A long run with @@ -493,7 +495,10 @@ public async Task Fairness_WriterUnderModerateReaders_MakesProgress() // ── MPMC producer fairness ─────────────────────────────────────────────── + // The ratio depends on the core count and the scheduler (eight spinning producers share the + // thread pool), so CI runs this separately and does not fail the build on it. [Test] + [Category("TimingSensitive")] [Timeout(30000)] public async Task Fairness_Mpmc_ProducersGetReasonableShare() { diff --git a/InterprocessMemory.Tests/ExtremeStressTests.cs b/InterprocessMemory.Tests/ExtremeStressTests.cs index ec77332..d693625 100644 --- a/InterprocessMemory.Tests/ExtremeStressTests.cs +++ b/InterprocessMemory.Tests/ExtremeStressTests.cs @@ -37,6 +37,7 @@ private string GetUniqueName(string prefix) [Test] [Timeout(180000)] [Explicit("Long-running stress test")] + [Category("LongRunning")] public async Task MPMC_16Producers_16Consumers_1MillionMessages() { using var buffer = new ConcurrentMessageQueue(GetUniqueName("MPMC_16x16"), slotCount: 4096, slotSize: 128); @@ -691,6 +692,7 @@ await Task.Run(() => [Test] [Timeout(150000)] // 2 minutes plus teardown/assertion headroom [Explicit("Long-running test")] + [Category("LongRunning")] public async Task Stability_MPMC_2Minutes_Continuous() { using var buffer = new ConcurrentMessageQueue(GetUniqueName("Stability_2min"), slotCount: 2048, slotSize: 128); From cdafe2dac3594d8b99f466e18bb45124b11d63a7 Mon Sep 17 00:00:00 2001 From: zacala Date: Mon, 5 Oct 2026 14:38:41 +0000 Subject: [PATCH 11/29] Bring XML docs and code comments up to date Document the thread rule for the write lock, the opt-in OrphanLockTimeout, read-lock recovery and the Dispose contract on IMemoryRegion, MemoryRegion, StructuredMemory and every container's Dispose. Correct the header comment on the recorded start time (platform specific encoding) and drop comments that described how the code used to behave. Comments only, no code changes. --- InterprocessMemory/ConcurrentMessageQueue.cs | 4 +- InterprocessMemory/ConcurrentQueue.cs | 5 ++ InterprocessMemory/IMemoryRegion.cs | 18 +++-- InterprocessMemory/MemoryRegion.cs | 67 +++++++++++-------- InterprocessMemory/SharedArray.cs | 4 +- .../SingleProducerByteStream.cs | 4 +- InterprocessMemory/SingleProducerQueue.cs | 5 ++ InterprocessMemory/StructuredMemory.cs | 16 ++++- 8 files changed, 83 insertions(+), 40 deletions(-) diff --git a/InterprocessMemory/ConcurrentMessageQueue.cs b/InterprocessMemory/ConcurrentMessageQueue.cs index 5d79794..f64bccc 100644 --- a/InterprocessMemory/ConcurrentMessageQueue.cs +++ b/InterprocessMemory/ConcurrentMessageQueue.cs @@ -629,7 +629,9 @@ private void ThrowIfDisposed() } /// - /// Releases all resources used by this buffer + /// Releases the underlying memory region. Stop and join every thread that uses this instance first: + /// calls that do not take a lock are not tracked, so one that is still running while the memory is + /// unmapped terminates the process (see ). /// public void Dispose() { diff --git a/InterprocessMemory/ConcurrentQueue.cs b/InterprocessMemory/ConcurrentQueue.cs index 8ac8f79..2fdfd79 100644 --- a/InterprocessMemory/ConcurrentQueue.cs +++ b/InterprocessMemory/ConcurrentQueue.cs @@ -327,6 +327,11 @@ private void ThrowIfDisposed() throw new ObjectDisposedException(nameof(ConcurrentQueue)); } + /// + /// Releases the underlying memory region. Stop and join every thread that uses this instance first: + /// calls that do not take a lock are not tracked, so one that is still running while the memory is + /// unmapped terminates the process (see ). + /// public void Dispose() { if (Interlocked.Exchange(ref _disposed, 1) != 0) diff --git a/InterprocessMemory/IMemoryRegion.cs b/InterprocessMemory/IMemoryRegion.cs index 22bbdc3..ca3a9fe 100644 --- a/InterprocessMemory/IMemoryRegion.cs +++ b/InterprocessMemory/IMemoryRegion.cs @@ -71,7 +71,9 @@ public readonly struct LockOwnerInfo public long AcquiredTimestamp { get; init; } /// - /// Gets whether the lock is orphaned (owner process died) + /// Gets whether the write lock is orphaned: its owner process has exited (or its PID was reused + /// by another process), or, when is enabled, + /// it has been held for longer than that limit /// public bool IsOrphan { get; init; } @@ -156,22 +158,28 @@ public interface IMemoryRegion : IDisposable void ReleaseWriteLock(); /// - /// Tries to acquire a shared read lock with timeout + /// Tries to acquire a shared read lock with timeout. + /// The wait is released with if the region is disposed + /// by another thread while waiting. /// bool TryAcquireReadLock(TimeSpan timeout); /// - /// Releases the read lock + /// Releases the read lock. Read locks are counted, not owned: a process that exits while + /// holding one leaves the shared count raised, which cannot be detected automatically + /// (see ). /// void ReleaseReadLock(); /// - /// Checks if the current write lock is orphaned (owner process died) + /// Checks if the current write lock is orphaned: its owner process has exited or its PID was + /// reused, or the opt-in has elapsed /// bool IsWriteLockOrphaned(); /// - /// Forces release of an orphaned write lock + /// Forces release of an orphaned write lock. Waiting lock acquisitions perform this + /// recovery themselves; call it directly only to clear the lock without waiting for it. /// /// True if lock was orphaned and released bool TryForceReleaseWriteLock(); diff --git a/InterprocessMemory/MemoryRegion.cs b/InterprocessMemory/MemoryRegion.cs index 91393c4..e9b409c 100644 --- a/InterprocessMemory/MemoryRegion.cs +++ b/InterprocessMemory/MemoryRegion.cs @@ -20,6 +20,15 @@ namespace InterprocessMemory /// /dev/shm (tmpfs) wrapped by MemoryMappedFile.CreateFromFile. Both paths /// yield the same raw pointer after construction, so the hot path (Read/Write/locks/SIMD) /// is byte-for-byte identical and there is no per-call OS dispatch. + /// + /// Read and Write are lock-free and do not take the region lock; use + /// / when several bytes must be + /// observed atomically. The write lock belongs to the acquiring thread and must be released on it. + /// A waiter recovers a write lock whose owner process has exited; recovery of read locks left by a + /// dead process is manual (, ). + /// + /// Stop and join every thread that uses an instance before disposing it; see + /// and . /// public sealed unsafe class MemoryRegion : IMemoryRegion { @@ -205,11 +214,13 @@ private struct SharedHeader [FieldOffset(40)] public long LockAcquiredTimestamp; // PID reuse defense: if a lock-holding process dies and the OS recycles its PID for an // unrelated process, GetProcessById would find the new process alive and skip orphan - // recovery — leaving the lock permanently held. By recording the owner's Process.StartTime - // at acquire and comparing on the orphan check, we detect the impostor. Stored as - // DateTime.ToBinary() (signed long). Value 0 means "not recorded" — older binaries that - // didn't write it, or hosts where StartTime is unreadable (permission denied); in that - // case orphan detection falls back to the PID-only check, preserving prior behavior. + // recovery — leaving the lock permanently held. The owner records its process start time + // at acquire and the orphan check compares it, which exposes the impostor. The encoding is + // platform specific (see s_processStartTimeBinary): Windows stores Process.StartTime.ToBinary(), + // Linux stores the /proc//stat start tick count. Value 0 means "not recorded" (the start + // time was unreadable, e.g. no /proc or an ACL denial); a negative value is the 3.0.0 Linux + // encoding, which cannot be compared across processes. In both cases the orphan check + // falls back to PID-only matching. [FieldOffset(48)] public long LockOwnerProcessStartTime; // bytes 56–63: implicit padding @@ -561,8 +572,8 @@ private bool InitializeOrOpen() // Phase 2: Winner writes all header fields, then promotes Magic to MagicNumber // via a release-store. Losers spin until they observe MagicNumber and // only then read the rest of the header, guaranteeing they never see - // partially-initialized state (the previous code could read Magic= - // MagicNumber but Capacity=0, falsely triggering a capacity mismatch). + // partially-initialized state (e.g. Magic=MagicNumber but Capacity=0, + // which would falsely report a capacity mismatch). uint prev = _createOrOpen ? Interlocked.CompareExchange(ref header->Magic, SharedHeader.MagicInitializing, 0) : Volatile.Read(ref header->Magic); @@ -690,8 +701,8 @@ public int Write(ReadOnlySpan source, long offset) byte* destPtr = GetDataPtr() + offset; // Use SIMD when length is at least one vector. On modern x86-64, unaligned SIMD - // via ReadUnaligned/WriteUnaligned has essentially no penalty, so gating on alignment - // (as the previous IsAligned check did) only hurt small writes by forcing the scalar fallback. + // via ReadUnaligned/WriteUnaligned has essentially no penalty, so the copy is not gated + // on alignment (that would only force small writes onto the scalar fallback). if (_options.EnableSimd && source.Length >= Vector.Count) { WriteSimd(source, destPtr); @@ -763,8 +774,8 @@ public int Read(Span destination, long offset) byte* srcPtr = GetDataPtr() + offset; - // Same rationale as Write — alignment gating was costing performance for small reads - // without buying anything on modern hardware that handles unaligned SIMD natively. + // Same rationale as Write: no alignment gating, because modern hardware handles + // unaligned SIMD natively. if (_options.EnableSimd && destination.Length >= Vector.Count) { ReadSimd(destination, srcPtr); @@ -808,9 +819,9 @@ private static void ReadSimd(Span destination, byte* srcPtr) } } - // IsAligned helper was removed in favor of unconditional SIMD for length >= Vector.Count. - // _options.Alignment is still validated as a power of 2 for forward compatibility and - // for consumers that may use it for their own offset calculations. + // SIMD is used unconditionally for length >= Vector.Count. _options.Alignment is + // validated as a power of 2 but does not influence the copy; it exists for consumers that + // want to align their own offsets. /// public ValueTask WriteAsync(ReadOnlyMemory source, long offset, @@ -965,9 +976,9 @@ public void ReleaseWriteLock() long currentThreadId = Environment.CurrentManagedThreadId; int ownerPid = Volatile.Read(ref header->LockOwnerProcessId); long ownerThreadId = Volatile.Read(ref header->LockOwnerThreadId); - // Releasing a lock this thread does not own must be loud. Silently ignoring it (the - // previous behaviour) turned `await` inside a lock scope — the continuation resumes on - // another thread — into a write lock that no process could ever release again. + // Releasing a lock this thread does not own must be loud. Silently ignoring it would turn + // `await` inside a lock scope — the continuation resumes on another thread — into a write + // lock that no process could ever release again. if (ownerPid != currentPid || ownerThreadId != currentThreadId) { _logger?.LogWarning( @@ -1039,20 +1050,17 @@ private bool TryAcquireReadLockCore(TimeSpan timeout) continue; } - // Optimistic claim: unconditional Interlocked.Increment instead of the previous - // read-CAS-recheck dance. Two wins under reader contention: - // 1. No CAS retry loop when N readers race — every one of them succeeds on - // the first atomic (`lock inc` is a single µop on x86 vs cmpxchg). - // 2. Cleaner code path. The brief window where we "claim" the reader slot - // before re-checking the writer is identical to the old code's CAS-then- - // recheck window — no new race introduced. + // Optimistic claim: an unconditional Interlocked.Increment, then re-check the writer. + // There is no CAS retry loop when N readers race — every one of them succeeds on + // the first atomic (`lock inc` is a single µop on x86 vs cmpxchg). The only window is + // between the claim and the writer re-check, handled by the rollback below. Interlocked.Increment(ref header->ReaderCount); if (Volatile.Read(ref header->WriterLockState) == 0) return true; // A writer acquired between our reader check and our increment. Roll back. // The writer's drain loop will briefly see ReaderCount > 0 and spin once or - // twice extra — same penalty as the previous design's CAS-rollback path. + // twice extra. Interlocked.Decrement(ref header->ReaderCount); if (TimeoutHelper.HasExpired(start, timeout)) @@ -1123,9 +1131,9 @@ public bool IsWriteLockOrphaned() } catch { - // StartTime can throw under restricted permissions (e.g., Linux container - // without /proc, Windows ACL). Fall through to timestamp-based detection - // — degraded but no worse than pre-feature behavior. + // The start time can be unreadable under restricted permissions (e.g., Linux + // container without /proc, Windows ACL). Fall back to the PID-only decision + // (and the optional OrphanLockTimeout check below). } } } @@ -1140,7 +1148,8 @@ public bool IsWriteLockOrphaned() return true; } - // Check timeout-based orphan detection + // The owner process is alive. Only the opt-in time limit (OrphanLockTimeout, disabled + // by default) may still declare the lock orphaned. if (_options.OrphanLockTimeout > TimeSpan.Zero) { long acquiredTimestamp = header->LockAcquiredTimestamp; diff --git a/InterprocessMemory/SharedArray.cs b/InterprocessMemory/SharedArray.cs index 33452cc..f0eb605 100644 --- a/InterprocessMemory/SharedArray.cs +++ b/InterprocessMemory/SharedArray.cs @@ -598,7 +598,9 @@ private void ThrowIfDisposed() } /// - /// Releases all resources used by this array + /// Releases the underlying memory region. Stop and join every thread that uses this instance first: + /// calls that do not take a lock are not tracked, so one that is still running while the memory is + /// unmapped terminates the process (see ). /// public void Dispose() { diff --git a/InterprocessMemory/SingleProducerByteStream.cs b/InterprocessMemory/SingleProducerByteStream.cs index 846b488..8b86055 100644 --- a/InterprocessMemory/SingleProducerByteStream.cs +++ b/InterprocessMemory/SingleProducerByteStream.cs @@ -424,7 +424,9 @@ private void ThrowIfDisposed() } /// - /// Releases all resources used by this buffer + /// Releases the underlying memory region. Stop and join every thread that uses this instance first: + /// calls that do not take a lock are not tracked, so one that is still running while the memory is + /// unmapped terminates the process (see ). /// public void Dispose() { diff --git a/InterprocessMemory/SingleProducerQueue.cs b/InterprocessMemory/SingleProducerQueue.cs index 0b8faf4..7a0a635 100644 --- a/InterprocessMemory/SingleProducerQueue.cs +++ b/InterprocessMemory/SingleProducerQueue.cs @@ -234,6 +234,11 @@ private void ThrowIfDisposed() throw new ObjectDisposedException(nameof(SingleProducerQueue)); } + /// + /// Releases the underlying memory region. Stop and join every thread that uses this instance first: + /// calls that do not take a lock are not tracked, so one that is still running while the memory is + /// unmapped terminates the process (see ). + /// public void Dispose() { if (Interlocked.Exchange(ref _disposed, 1) != 0) diff --git a/InterprocessMemory/StructuredMemory.cs b/InterprocessMemory/StructuredMemory.cs index 39e6b7a..ac576f2 100644 --- a/InterprocessMemory/StructuredMemory.cs +++ b/InterprocessMemory/StructuredMemory.cs @@ -14,6 +14,14 @@ namespace InterprocessMemory /// All fields are declared upfront with fixed types, positions, and sizes. /// Provides zero-allocation access with full type safety. /// Cross-platform via the underlying (Windows + Linux). + /// + /// Values wider than eight bytes, strings, blobs and arrays are read and written under the shared + /// region lock automatically. Use / + /// to group several fields into one transaction. The guards belong to the calling thread: dispose + /// them on the thread that acquired them (do not await inside the guarded scope), otherwise + /// is thrown. clears a lock + /// state left behind by a crashed process. + /// /// public sealed class StructuredMemory : IDisposable where TSchema : struct, IMemorySchema { @@ -21,8 +29,8 @@ public sealed class StructuredMemory : IDisposable where TSchema : stru private const uint SchemaMagic = 0x53504D49; // "IMPS" // x86-64 guarantees atomic load/store for aligned values up to 8 bytes (MOV instruction). // Types wider than this threshold require automatic locking to prevent torn reads/writes. - // Note: ARM64 supports 16-byte atomics (ldp/stp) but .NET on Windows ARM64 uses TSO - // emulation, so 8 bytes is the safe cross-platform limit for this Windows-only library. + // Note: ARM64 supports 16-byte atomics (ldp/stp) but not every supported platform guarantees + // them (e.g. x64 emulation on Windows ARM64), so 8 bytes is the safe cross-platform limit. private const int AtomicThreshold = 8; private const int MaxStackAllocBytes = 1024; // Max bytes for stackalloc (prevent stack overflow) @@ -1145,7 +1153,9 @@ private void ThrowIfHoldingReadLock() } /// - /// Releases all resources used by this shared memory region + /// Releases the underlying memory region. Stop and join every thread that uses this instance first: + /// calls that do not take a lock are not tracked, so one that is still running while the memory is + /// unmapped terminates the process (see ). /// public void Dispose() { From 93dd60dd16ef28d870ce1773f0e5e7256bbe29cd Mon Sep 17 00:00:00 2001 From: zacala Date: Mon, 5 Oct 2026 14:38:41 +0000 Subject: [PATCH 12/29] Update README and MIGRATION, add CHANGELOG Describe the CI matrix and the TimingSensitive/LongRunning test categories, point MIGRATION at MemoryRegion.Remove, and list the behavior changes, fixes and additions since 3.0.0 in a new CHANGELOG. --- CHANGELOG.md | 54 ++++++++++++++++++++++++++++++++++++++++++++++++++++ MIGRATION.md | 4 +++- README.md | 13 ++++++++++++- 3 files changed, 69 insertions(+), 2 deletions(-) create mode 100644 CHANGELOG.md diff --git a/CHANGELOG.md b/CHANGELOG.md new file mode 100644 index 0000000..15651be --- /dev/null +++ b/CHANGELOG.md @@ -0,0 +1,54 @@ +# Changelog + +Changes since the 3.0.0 release. Migrating from 2.x: see [MIGRATION.md](MIGRATION.md). + +## Unreleased + +### Behavior changes + +- `MemoryRegion.ReleaseWriteLock` throws `SynchronizationLockException` when the calling thread does + not own the lock (it used to be ignored). The `StructuredMemory` lock guards do the same when they + are disposed on another thread, which is what an `await` inside a lock scope causes. +- `MemoryRegionOptions.OrphanLockTimeout` defaults to `TimeSpan.Zero` (disabled) instead of 30 seconds. + A lock whose owner process has exited is still recovered by default; a lock held by a live process is + taken over only if you set a timeout. `DefaultOrphanLockTimeout` keeps its value as a suggestion. +- `Dispose()` of a `MemoryRegion`, and of every container that owns one, takes at least + `MemoryRegion.DisposeGracePeriod` (10 ms) longer. +- `SharedArray` takes the shared region lock per element access for element sizes other than + 1, 2, 4 and 8 bytes, so values are no longer torn. These accesses are slower; take the lock once with + `AcquireReadLock()` / `AcquireWriteLock()` for bulk work. +- An out-of-range queue `capacity` is reported with the parameter name `capacity`. + +### Fixed + +- Linux: a live write-lock owner was reported as an orphan and its lock was taken over, because + `Process.StartTime` differs between observers. The owner is now identified by its `/proc//stat` + start tick count. +- Waiting for a lock with `Timeout.InfiniteTimeSpan` now recovers from an owner that dies later + (the owner is probed every 250 ms). +- Disposing a region while another thread waits for one of its locks no longer crashes the process; + the waiter fails with `ObjectDisposedException`. +- Generic unmanaged structs (`ValueTuple`, `KeyValuePair<,>`) work in all typed containers. + Fingerprints of types that already worked are unchanged, so existing regions stay compatible. +- `SharedArray` disposes its region when opening fails because of a different element type or length. +- `SharedArray` and `StructuredMemory` publish their header magic after the other fields, which + prevents a spurious format error on weakly ordered CPUs. +- `Dispose()` racing a call that is still running on another thread is mitigated by + `DisposeGracePeriod`. This is best effort; stop and join threads before disposing. + +### Added + +- `MemoryRegion.Remove(name, options)` deletes the backing storage of a region left unusable by a crash. +- `MemoryRegion.ForceResetLocks()` and `StructuredMemory.ForceResetLocks()` / + `SharedArray.ForceResetLocks()` clear lock state left behind by a crashed process. +- `LockOwnerInfo.ReaderCount` for diagnosing a stale reader count. +- `MemoryRegion.DisposeGracePeriod` (static, process-wide). +- `SharedArray.AcquireReadLock()` / `AcquireWriteLock()` with timeout overloads, returning + reentrant `ref struct` guards. +- GitHub Actions workflow that builds and tests on Linux and Windows. + +### Removed + +- Finalizers on `ConcurrentMessageQueue`, `SingleProducerByteStream`, `SharedArray` and + `StructuredMemory`. The `MemoryRegion` finalizer still unmaps the memory when `Dispose` is never + called. diff --git a/MIGRATION.md b/MIGRATION.md index c394069..3d5d941 100644 --- a/MIGRATION.md +++ b/MIGRATION.md @@ -94,7 +94,9 @@ Version 3 rejects version 2 headers and never overwrites them. 1. Stop all processes using the 2.x region. 2. Preserve data externally if it must survive the upgrade. -3. Remove the explicit backing file or stale Linux `/dev/shm` entry when applicable. +3. Remove the explicit backing file or stale Linux `/dev/shm` entry when applicable. A version 3 + process can do this with `MemoryRegion.Remove(name)` (pass the same `MemoryRegionOptions` for a + file-backed region). 4. Start the version 3 creating process. 5. Start remaining processes with `OpenExisting`. diff --git a/README.md b/README.md index fe1a03b..90b2c85 100644 --- a/README.md +++ b/README.md @@ -312,9 +312,20 @@ Windows named sections disappear when their last handle closes. On Linux a regio ```shell dotnet restore InterprocessMemory.sln -dotnet test InterprocessMemory.sln dotnet build InterprocessMemory.sln --configuration Release +dotnet test InterprocessMemory.sln --configuration Release ``` The test suite includes real child-process transfer, multi-process typed MPMC delivery, cross-process lock exclusion, and orphan-lock recovery. + +[GitHub Actions](.github/workflows/ci.yml) builds and tests on Linux and Windows for every pull +request and every push to `main`. Two groups of tests are kept out of the blocking test step (use +`--filter` to select or exclude them locally): + +- `Category=TimingSensitive`: assertions that depend on thread scheduling, such as a fairness ratio. + CI runs them in a separate, non-blocking step because a busy shared runner can fail them. +- `Category=LongRunning`: the `[Explicit]` soak tests, which NUnit never runs unless asked to. CI does + not run them. + +Changes since the last release are listed in [CHANGELOG.md](CHANGELOG.md). From edb3742a326602c80afbd60d2875a0df428cc8a2 Mon Sep 17 00:00:00 2001 From: zacala Date: Mon, 5 Oct 2026 14:42:40 +0000 Subject: [PATCH 13/29] Run the 64-reader/4-writer stress test on dedicated threads Opt8_OptimisticReader_64Readers_4Writers_AllProgress started 68 loops that spin until a token is cancelled with Task.Run. With few cores they occupy every thread-pool worker, the pool adds a thread only every ~500 ms, and the writers queued behind the readers start late or the cancellation timer callback waits behind them. The test then fails with a starved writer or runs into its 30 s timeout (seen on a GitHub Linux runner). It fails the same way on the code before the lock changes. LongRunning gives every loop its own thread. Passes 6 of 6 runs each pinned to 1, 2 and 4 CPUs; before, it failed 6 of 6 runs on 2 CPUs. --- .../ConcurrencyStabilityTests.cs | 13 +++++++++---- 1 file changed, 9 insertions(+), 4 deletions(-) diff --git a/InterprocessMemory.Tests/ConcurrencyStabilityTests.cs b/InterprocessMemory.Tests/ConcurrencyStabilityTests.cs index 0ad16c7..fe4e11b 100644 --- a/InterprocessMemory.Tests/ConcurrencyStabilityTests.cs +++ b/InterprocessMemory.Tests/ConcurrencyStabilityTests.cs @@ -120,7 +120,12 @@ public async Task Opt8_OptimisticReader_64Readers_4Writers_AllProgress() var writeSuccess = new int[writerCount]; var readSuccess = 0L; - var writers = Enumerable.Range(0, writerCount).Select(w => Task.Run(() => + // Every loop spins until the token is cancelled, so each one gets its own thread + // (LongRunning). On the thread pool, 68 spinning tasks occupy every worker on a machine + // with few cores, the pool adds a thread only every ~500 ms, and the writers queued behind + // the readers start late or the cancellation timer callback itself waits behind them (the + // test then fails with a starved writer or runs into its timeout). + var writers = Enumerable.Range(0, writerCount).Select(w => Task.Factory.StartNew(() => { while (!cts.Token.IsCancellationRequested) { @@ -132,9 +137,9 @@ public async Task Opt8_OptimisticReader_64Readers_4Writers_AllProgress() // Brief pause so we're not pegging the lock continuously Thread.SpinWait(50); } - })).ToArray(); + }, TaskCreationOptions.LongRunning)).ToArray(); - var readers = Enumerable.Range(0, readerCount).Select(_ => Task.Run(() => + var readers = Enumerable.Range(0, readerCount).Select(_ => Task.Factory.StartNew(() => { while (!cts.Token.IsCancellationRequested) { @@ -144,7 +149,7 @@ public async Task Opt8_OptimisticReader_64Readers_4Writers_AllProgress() finally { buf.ReleaseReadLock(); } } } - })).ToArray(); + }, TaskCreationOptions.LongRunning)).ToArray(); await Task.WhenAll(writers.Concat(readers)); From b896dc55b50587552aa0540b80e9b932165e9fcb Mon Sep 17 00:00:00 2001 From: zacala Date: Mon, 5 Oct 2026 14:42:40 +0000 Subject: [PATCH 14/29] Always run the timing-sensitive CI step Without a condition it was skipped when the blocking test step failed, so a failing run showed nothing about the timing-sensitive tests. It cannot fail the job (continue-on-error), so running it unconditionally is safe. --- .github/workflows/ci.yml | 1 + 1 file changed, 1 insertion(+) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index d9ce129..30434d5 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -77,6 +77,7 @@ jobs: # Tests whose outcome depends on the core count and the scheduler. They still run and show up # in the log and the artifact, but a failure here does not fail the job. - name: Test (timing sensitive, informational) + if: ${{ !cancelled() }} continue-on-error: true run: > dotnet test InterprocessMemory.Tests/InterprocessMemory.Tests.csproj From 3805dba77f3932bdbfbb7ff9818f94c11dd5266d Mon Sep 17 00:00:00 2001 From: zacala Date: Tue, 6 Oct 2026 10:58:53 +0000 Subject: [PATCH 15/29] Read and write 1, 2, 4 and 8 byte elements with one typed access SharedArray and StructuredMemory copied small values through Span.CopyTo. For a length of two bytes that is a one byte store followed by a two byte store, so a reader in another process saw 0x0000 or 0xFFFF between 0x00FF and 0xFF00 (about 1.5% of reads in a two thread test; 3 and 6 byte values and small arrays in StructuredMemory were torn in the same way, because only values wider than 8 bytes were locked). AtomicAccess performs one Volatile load or store of exactly the element width. SharedArray uses it for single elements (indexer, one-element CopyTo/CopyFrom/Fill). StructuredMemory uses it for 1, 2, 4 and 8 byte scalars and locks every other size and every array. The new tests fail on the previous code (83,265 of 5.3 M reads torn for ushort, 72,633 of 2.6 M for the StructuredMemory fields) and pass now. --- CHANGELOG.md | 5 + InterprocessMemory.Tests/SharedArrayTests.cs | 96 ++++++++++++++++ .../StructuredMemoryTests.cs | 107 ++++++++++++++++++ InterprocessMemory/AtomicAccess.cs | 81 +++++++++++++ InterprocessMemory/SharedArray.cs | 54 +++++++-- InterprocessMemory/StructuredMemory.cs | 52 +++++---- README.md | 14 ++- 7 files changed, 376 insertions(+), 33 deletions(-) create mode 100644 InterprocessMemory/AtomicAccess.cs diff --git a/CHANGELOG.md b/CHANGELOG.md index 15651be..66c631f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -30,6 +30,11 @@ Changes since the 3.0.0 release. Migrating from 2.x: see [MIGRATION.md](MIGRATIO the waiter fails with `ObjectDisposedException`. - Generic unmanaged structs (`ValueTuple`, `KeyValuePair<,>`) work in all typed containers. Fingerprints of types that already worked are unchanged, so existing regions stay compatible. +- Two byte `SharedArray` elements could be read half written by another process (a two byte copy is a + one byte store plus a two byte store). Elements of 1, 2, 4 and 8 bytes are now read and written with one + typed load/store. `StructuredMemory` had the same defect for 2 byte scalars and, because it only locked + values wider than 8 bytes, also for 3, 5, 6 and 7 byte values and for small arrays: only 1, 2, 4 and 8 + byte scalars are lock-free now, everything else (including every array) takes the shared lock. - `SharedArray` disposes its region when opening fails because of a different element type or length. - `SharedArray` and `StructuredMemory` publish their header magic after the other fields, which prevents a spurious format error on weakly ordered CPUs. diff --git a/InterprocessMemory.Tests/SharedArrayTests.cs b/InterprocessMemory.Tests/SharedArrayTests.cs index 10976f9..b0eb4cf 100644 --- a/InterprocessMemory.Tests/SharedArrayTests.cs +++ b/InterprocessMemory.Tests/SharedArrayTests.cs @@ -321,6 +321,102 @@ public void WideElements_AreNeverObservedTorn_AcrossInstances() Assert.That(torn, Is.EqualTo(0), $"{torn} of {reads} reads were torn"); } + // Two bytes: the width that used to be copied as a one byte store plus a two byte store. + public struct Pair2 { public byte X, Y; } + + private enum ElementPath { Indexer, CopyRange, Fill } + + // The writer alternates two values whose halves differ, the reader (a second instance, as another + // process would be) must only ever see one of the two. Before the element was written with a single + // store, a reader saw 0x0000 or 0xFFFF between 0x00FF and 0xFF00 about once in 200 reads. + private static void AssertNeverTorn(string tag, T first, T second, ElementPath path) + where T : unmanaged + { + string name = LockName($"Atomic{tag}{path}"); + using var writerArray = SharedArray.CreateOrOpen(name, 4); + using var readerArray = SharedArray.OpenExisting(name); + writerArray[0] = first; + using var stop = new CancellationTokenSource(TimeSpan.FromSeconds(0.6)); + long reads = 0; + long torn = 0; + + var writer = Task.Factory.StartNew(() => + { + T[] one = new T[1]; + while (!stop.IsCancellationRequested) + { + switch (path) + { + case ElementPath.Indexer: + writerArray[0] = second; + writerArray[0] = first; + break; + case ElementPath.CopyRange: + one[0] = second; + writerArray.CopyFrom(0, one); + one[0] = first; + writerArray.CopyFrom(0, one); + break; + default: + writerArray.Fill(second, 0, 1); + writerArray.Fill(first, 0, 1); + break; + } + } + }, TaskCreationOptions.LongRunning); + + var reader = Task.Factory.StartNew(() => + { + T[] one = new T[1]; + var expectedFirst = System.Runtime.InteropServices.MemoryMarshal.AsBytes(new[] { first }.AsSpan()).ToArray(); + var expectedSecond = System.Runtime.InteropServices.MemoryMarshal.AsBytes(new[] { second }.AsSpan()).ToArray(); + while (!stop.IsCancellationRequested) + { + T value; + if (path == ElementPath.Indexer) + { + value = readerArray[0]; + } + else + { + readerArray.CopyTo(0, one); + value = one[0]; + } + + var bytes = System.Runtime.InteropServices.MemoryMarshal.AsBytes(new[] { value }.AsSpan()); + if (!bytes.SequenceEqual(expectedFirst) && !bytes.SequenceEqual(expectedSecond)) + torn++; + reads++; + } + }, TaskCreationOptions.LongRunning); + + Task.WaitAll(writer, reader); + + Assert.That(reads, Is.GreaterThan(0)); + Assert.That(torn, Is.EqualTo(0), $"{typeof(T).Name} via {path}: {torn} of {reads} reads were torn"); + } + + [Test, Timeout(60000)] + public void TwoByteElements_AreNeverObservedTorn() + { + foreach (ElementPath path in Enum.GetValues()) + { + AssertNeverTorn("U16", (ushort)0x00FF, (ushort)0xFF00, path); + AssertNeverTorn("Pair2", new Pair2 { X = 0x00, Y = 0xFF }, new Pair2 { X = 0xFF, Y = 0x00 }, path); + } + } + + [Test, Timeout(60000)] + public void OtherAtomicWidths_AreNeverObservedTorn() + { + foreach (ElementPath path in Enum.GetValues()) + { + AssertNeverTorn("U8", (byte)0x0F, (byte)0xF0, path); + AssertNeverTorn("U32", 0x00FF00FFu, 0xFF00FF00u, path); + AssertNeverTorn("U64", 0x00FF00FF00FF00FFul, 0xFF00FF00FF00FF00ul, path); + } + } + [Test, Timeout(60000)] public void AtomicElements_BypassTheLock_WideElementsDoNot() { diff --git a/InterprocessMemory.Tests/StructuredMemoryTests.cs b/InterprocessMemory.Tests/StructuredMemoryTests.cs index 2d9cdd2..039de79 100644 --- a/InterprocessMemory.Tests/StructuredMemoryTests.cs +++ b/InterprocessMemory.Tests/StructuredMemoryTests.cs @@ -2119,4 +2119,111 @@ public void StringWithNullChar_ShouldTruncate() } #endregion + + #region Small values are never observed half written + + public struct Rgb3 { public byte R, G, B; } + + public struct Tag6 { public ushort A, B, C; } + + public struct SmallSchema : IMemorySchema + { + public IEnumerable GetFields() + { + yield return FieldDefinition.Scalar("Short"); + yield return FieldDefinition.Struct("Rgb"); + yield return FieldDefinition.Struct("Tag"); + yield return FieldDefinition.Array("Bytes", 6); + } + } + + [Test, Timeout(60000)] + public void SmallValues_AreNeverObservedHalfWritten_AcrossInstances() + { + // A two byte copy used to be a one byte store plus a two byte store, and 3, 5, 6 and 7 byte + // values and short arrays were written without the lock although they need several stores. + string name = $"{TestBufferName}_Small_{Guid.NewGuid():N}"; + var schema = new SmallSchema(); + using var writerMemory = StructuredMemory.CreateOrOpen(name, schema); + using var readerMemory = StructuredMemory.OpenExisting(name, schema); + using var stop = new CancellationTokenSource(TimeSpan.FromSeconds(1.2)); + + void WriteAll(byte pattern) + { + writerMemory.Write("Short", pattern == 0 ? (short)0 : (short)-1); + writerMemory.Write("Rgb", new Rgb3 { R = pattern, G = pattern, B = pattern }); + writerMemory.Write("Tag", new Tag6 { A = pattern, B = pattern, C = pattern }); + writerMemory.WriteArray("Bytes", new byte[] { pattern, pattern, pattern, pattern, pattern, pattern }); + } + + WriteAll(0); + long reads = 0; + var torn = new System.Collections.Concurrent.ConcurrentBag(); + + var writer = Task.Factory.StartNew(() => + { + while (!stop.IsCancellationRequested) + { + WriteAll(0xFF); + WriteAll(0); + } + }, TaskCreationOptions.LongRunning); + + var reader = Task.Factory.StartNew(() => + { + var bytes = new byte[6]; + while (!stop.IsCancellationRequested) + { + short s = readerMemory.Read("Short"); + if (s != 0 && s != -1) torn.Add($"Short={s:X4}"); + + var rgb = readerMemory.Read("Rgb"); + if (rgb.R != rgb.G || rgb.G != rgb.B) torn.Add($"Rgb={rgb.R:X2}{rgb.G:X2}{rgb.B:X2}"); + + var tag = readerMemory.Read("Tag"); + if (tag.A != tag.B || tag.B != tag.C) torn.Add($"Tag={tag.A:X}/{tag.B:X}/{tag.C:X}"); + + readerMemory.ReadArray("Bytes", bytes); + if (bytes.Any(b => b != bytes[0])) torn.Add("Bytes=" + Convert.ToHexString(bytes)); + + reads++; + } + }, TaskCreationOptions.LongRunning); + + Task.WaitAll(writer, reader); + + Assert.That(reads, Is.GreaterThan(0)); + Assert.That(torn, Is.Empty, $"{torn.Count} torn reads in {reads}, e.g. {torn.FirstOrDefault()}"); + } + + [Test] + public void AtomicSizedScalars_RoundTripWithoutTheLock() + { + string name = $"{TestBufferName}_NoLock_{Guid.NewGuid():N}"; + var schema = new SmallSchema(); + using var memory = StructuredMemory.CreateOrOpen(name, schema); + using var peer = StructuredMemory.OpenExisting(name, schema); + + // The scalar of a power-of-two size is one typed load/store: it does not wait for a writer. + using (memory.AcquireWriteLock()) + { + var read = Task.Run(() => { peer.Write("Short", (short)1234); return peer.Read("Short"); }); + Assert.That(read.Wait(TimeSpan.FromSeconds(2)), Is.True); + Assert.That(read.Result, Is.EqualTo((short)1234)); + } + + // A three byte struct needs several stores, so its reader waits for the writer to finish. + memory.Write("Rgb", new Rgb3 { R = 1, G = 2, B = 3 }); + Task blocked; + using (memory.AcquireWriteLock()) + { + blocked = Task.Run(() => peer.Read("Rgb")); + Assert.That(blocked.Wait(TimeSpan.FromMilliseconds(300)), Is.False); + } + + Assert.That(blocked.Wait(TimeSpan.FromSeconds(10)), Is.True); + Assert.That(blocked.Result.B, Is.EqualTo((byte)3)); + } + + #endregion } diff --git a/InterprocessMemory/AtomicAccess.cs b/InterprocessMemory/AtomicAccess.cs new file mode 100644 index 0000000..45cda94 --- /dev/null +++ b/InterprocessMemory/AtomicAccess.cs @@ -0,0 +1,81 @@ +using System; +using System.Buffers; +using System.Runtime.CompilerServices; +using System.Threading; + +namespace InterprocessMemory +{ + /// + /// Single-copy atomic access to 1, 2, 4 and 8 byte values in shared memory. + /// + /// and copy through + /// Span<byte>.CopyTo, whose small-size path is not one store for every length: a two byte + /// copy is a one byte store followed by a two byte store, so another process can observe a + /// half-written value. A typed volatile access is always one load or store of exactly the + /// element width (and, for 8 bytes on a 32-bit process, an interlocked one), which is what the + /// lock-free element paths of and + /// need. + /// + /// The caller guarantees that the location is aligned to the value size. + /// + internal static unsafe class AtomicAccess + { + /// True for the sizes that and support. + [MethodImpl(MethodImplOptions.AggressiveInlining)] + public static bool IsAtomicSize(int size) => size is 1 or 2 or 4 or 8; + + /// + /// Returns the address of in the data area of the region. The mapping + /// does not move, so the pointer stays valid until the region is disposed; callers must not use + /// it afterwards (the same contract as ). + /// + public static byte* GetPointer(IMemoryRegion region, long offset) + { + using MemoryHandle handle = region.GetMemory(offset, 1).Pin(); + return (byte*)handle.Pointer; + } + + [MethodImpl(MethodImplOptions.AggressiveInlining)] + public static T Read(byte* location) where T : unmanaged + { + // sizeof(T) is a constant per instantiation, so the JIT keeps exactly one of these branches. + if (sizeof(T) == 1) + { + byte value = Volatile.Read(ref *location); + return Unsafe.As(ref value); + } + if (sizeof(T) == 2) + { + ushort value = Volatile.Read(ref *(ushort*)location); + return Unsafe.As(ref value); + } + if (sizeof(T) == 4) + { + uint value = Volatile.Read(ref *(uint*)location); + return Unsafe.As(ref value); + } + if (sizeof(T) == 8) + { + ulong value = Volatile.Read(ref *(ulong*)location); + return Unsafe.As(ref value); + } + + throw new NotSupportedException($"Atomic access needs a 1, 2, 4 or 8 byte type, not {sizeof(T)} bytes."); + } + + [MethodImpl(MethodImplOptions.AggressiveInlining)] + public static void Write(byte* location, T value) where T : unmanaged + { + if (sizeof(T) == 1) + Volatile.Write(ref *location, Unsafe.As(ref value)); + else if (sizeof(T) == 2) + Volatile.Write(ref *(ushort*)location, Unsafe.As(ref value)); + else if (sizeof(T) == 4) + Volatile.Write(ref *(uint*)location, Unsafe.As(ref value)); + else if (sizeof(T) == 8) + Volatile.Write(ref *(ulong*)location, Unsafe.As(ref value)); + else + throw new NotSupportedException($"Atomic access needs a 1, 2, 4 or 8 byte type, not {sizeof(T)} bytes."); + } + } +} diff --git a/InterprocessMemory/SharedArray.cs b/InterprocessMemory/SharedArray.cs index f0eb605..24cc14c 100644 --- a/InterprocessMemory/SharedArray.cs +++ b/InterprocessMemory/SharedArray.cs @@ -14,11 +14,14 @@ namespace InterprocessMemory /// Provides array-like access to shared memory with compile-time type checking. /// Cross-platform: backed by which supports Windows and Linux. /// - /// Atomicity. Elements whose size is 1, 2, 4 or 8 bytes are copied with a single aligned - /// move, so another process never observes half of one, and they are accessed without any lock. - /// Every other element size (a 16-byte , a 64-byte struct, ...) could be read - /// torn while another process writes it, so the indexer, , - /// and take the shared region lock for those types. + /// Atomicity. An element whose size is 1, 2, 4 or 8 bytes is read and written with one + /// aligned load or store (the indexer, and / of a single + /// element), so another process never observes half of one, and no lock is taken. A range of + /// several elements is a plain memory copy: each element is intact in practice, but the range as + /// a whole is not a snapshot, so take a lock for that. Every other element size (a 16-byte + /// , a 64-byte struct, ...) could be read torn while another process writes it, + /// so the indexer, , and take the + /// shared region lock for those types. /// /// /// Use / when several elements @@ -27,13 +30,17 @@ namespace InterprocessMemory /// /// /// Unmanaged value type - public sealed class SharedArray : IDisposable where T : unmanaged + public sealed unsafe class SharedArray : IDisposable where T : unmanaged { private readonly IMemoryRegion _buffer; private const int ArrayHeaderSize = 64; private const uint ArrayMagic = 0x59415249; // "IRAY" private const int FormatVersion = 3; + // Address of element 0. Valid until Dispose (the mapping does not move); used for the + // lock-free single-element path, see AtomicAccess. + private readonly byte* _elements; + private int _length; private readonly int _elementSize; private readonly TypeLayoutFingerprint _fingerprint; @@ -110,6 +117,8 @@ private SharedArray(string name, int? length, bool createOrOpen) _buffer.Dispose(); throw; } + + _elements = AtomicAccess.GetPointer(_buffer, ArrayHeaderSize); } // The array exposes no statistics, so the region's per-call counters would only cost an @@ -214,6 +223,11 @@ public T this[int index] private T ReadElement(int index) { + // 1, 2, 4 and 8 byte elements: one typed load. Going through Span.CopyTo is not enough, + // a two byte copy is a one byte store plus a two byte store and can be seen half done. + if (!s_needsLock) + return AtomicAccess.Read(_elements + (long)index * _elementSize); + T value = default; _buffer.Read(MemoryMarshal.AsBytes(MemoryMarshal.CreateSpan(ref value, 1)), ArrayHeaderSize + (long)index * _elementSize); @@ -222,6 +236,12 @@ private T ReadElement(int index) private void WriteElement(int index, T value) { + if (!s_needsLock) + { + AtomicAccess.Write(_elements + (long)index * _elementSize, value); + return; + } + _buffer.Write(MemoryMarshal.AsBytes(MemoryMarshal.CreateReadOnlySpan(ref value, 1)), ArrayHeaderSize + (long)index * _elementSize); } @@ -288,8 +308,18 @@ public void CopyTo(int startIndex, Span destination) } } - private void CopyToCore(int startIndex, Span destination) => + private void CopyToCore(int startIndex, Span destination) + { + // A single element must stay one load (see ReadElement); longer ranges are a plain copy + // and are not atomic across elements. + if (!s_needsLock && destination.Length == 1) + { + destination[0] = AtomicAccess.Read(_elements + (long)startIndex * _elementSize); + return; + } + _buffer.Read(MemoryMarshal.AsBytes(destination), ArrayHeaderSize + (long)startIndex * _elementSize); + } /// /// Copies a span of elements to the array. @@ -324,8 +354,16 @@ public void CopyFrom(int startIndex, ReadOnlySpan source) } } - private void CopyFromCore(int startIndex, ReadOnlySpan source) => + private void CopyFromCore(int startIndex, ReadOnlySpan source) + { + if (!s_needsLock && source.Length == 1) + { + AtomicAccess.Write(_elements + (long)startIndex * _elementSize, source[0]); + return; + } + _buffer.Write(MemoryMarshal.AsBytes(source), ArrayHeaderSize + (long)startIndex * _elementSize); + } /// /// Fills a range with a value. diff --git a/InterprocessMemory/StructuredMemory.cs b/InterprocessMemory/StructuredMemory.cs index ac576f2..e39db98 100644 --- a/InterprocessMemory/StructuredMemory.cs +++ b/InterprocessMemory/StructuredMemory.cs @@ -23,21 +23,24 @@ namespace InterprocessMemory /// state left behind by a crashed process. /// /// - public sealed class StructuredMemory : IDisposable where TSchema : struct, IMemorySchema + public sealed unsafe class StructuredMemory : IDisposable where TSchema : struct, IMemorySchema { private const int SchemaHeaderSize = 64; // Reserved for schema metadata private const uint SchemaMagic = 0x53504D49; // "IMPS" - // x86-64 guarantees atomic load/store for aligned values up to 8 bytes (MOV instruction). - // Types wider than this threshold require automatic locking to prevent torn reads/writes. - // Note: ARM64 supports 16-byte atomics (ldp/stp) but not every supported platform guarantees - // them (e.g. x64 emulation on Windows ARM64), so 8 bytes is the safe cross-platform limit. - private const int AtomicThreshold = 8; + // An aligned load or store of 1, 2, 4 or 8 bytes is atomic on every supported platform, and + // AtomicAccess performs exactly one. Every other scalar size (3, 5, 6, 7 and anything wider than + // 8 bytes) as well as arrays, strings and blobs (several stores) go through the shared lock. + // ARM64 has 16-byte atomics (ldp/stp), but not every supported platform guarantees them + // (e.g. x64 emulation on Windows ARM64), so 8 bytes is the safe cross-platform limit. private const int MaxStackAllocBytes = 1024; // Max bytes for stackalloc (prevent stack overflow) // Cached TimeSpan to avoid repeated allocations private static readonly TimeSpan DefaultLockTimeout = TimeSpan.FromSeconds(5); private readonly IMemoryRegion _buffer; + // Address of the first byte after the region header, valid until Dispose; used for the + // lock-free scalar path (AtomicAccess). + private readonly byte* _dataBase; private readonly TSchema _schema; private readonly Dictionary _fields; private readonly SchemaCompatibility _compatibility; @@ -161,11 +164,13 @@ internal StructuredMemory( _buffer.Dispose(); throw; } + + _dataBase = AtomicAccess.GetPointer(_buffer, 0); } /// /// Writes a strictly-typed value to a named field. - /// For types larger than 8 bytes (non-atomic), automatic locking is applied. + /// Values that are not 1, 2, 4 or 8 bytes wide are written under the shared lock automatically. /// /// /// Thrown when the caller holds only a read lock and attempts to write a non-atomic value @@ -182,7 +187,7 @@ public void Write(string fieldName, T value) where T : unmanaged EnsureScalarField(metadata); ValidateFieldType(metadata); - bool isNonAtomic = Unsafe.SizeOf() > AtomicThreshold; + bool isNonAtomic = !AtomicAccess.IsAtomicSize(Unsafe.SizeOf()); if (isNonAtomic && !IsHoldingWriteLock()) { ThrowIfHoldingReadLock(); @@ -198,6 +203,14 @@ public void Write(string fieldName, T value) where T : unmanaged [MethodImpl(MethodImplOptions.AggressiveInlining)] private void WriteInternal(T value, FieldMetadata metadata) where T : unmanaged { + if (AtomicAccess.IsAtomicSize(sizeof(T))) + { + // One typed store. Span.CopyTo is not enough: a two byte copy is a one byte store plus a + // two byte store, which another process can see half done. + AtomicAccess.Write(_dataBase + SchemaHeaderSize + metadata.Offset, value); + return; + } + // MemoryMarshal.CreateSpan + AsBytes avoids a stackalloc by reinterpreting // the local variable directly as bytes without any extra copy. _buffer.Write(MemoryMarshal.AsBytes(MemoryMarshal.CreateSpan(ref value, 1)), @@ -206,7 +219,7 @@ private void WriteInternal(T value, FieldMetadata metadata) where T : unmanag /// /// Reads a strictly-typed value from a named field. - /// For types larger than 8 bytes (non-atomic), automatic locking is applied. + /// Values that are not 1, 2, 4 or 8 bytes wide are read under the shared lock automatically. /// [MethodImpl(MethodImplOptions.AggressiveInlining)] public T Read(string fieldName) where T : unmanaged @@ -219,8 +232,8 @@ public T Read(string fieldName) where T : unmanaged EnsureScalarField(metadata); ValidateFieldType(metadata); - // Auto-lock for non-atomic types (>8 bytes) to prevent torn reads - bool needsAutoLock = Unsafe.SizeOf() > AtomicThreshold && !IsHoldingAnyLock(); + // Auto-lock for sizes that one load cannot cover, to prevent torn reads + bool needsAutoLock = !AtomicAccess.IsAtomicSize(Unsafe.SizeOf()) && !IsHoldingAnyLock(); if (needsAutoLock) { using var _ = AcquireReadLock(); @@ -235,6 +248,9 @@ public T Read(string fieldName) where T : unmanaged [MethodImpl(MethodImplOptions.AggressiveInlining)] private T ReadInternal(FieldMetadata metadata) where T : unmanaged { + if (AtomicAccess.IsAtomicSize(sizeof(T))) + return AtomicAccess.Read(_dataBase + SchemaHeaderSize + metadata.Offset); + T value = default; _buffer.Read(MemoryMarshal.AsBytes(MemoryMarshal.CreateSpan(ref value, 1)), SchemaHeaderSize + metadata.Offset); @@ -243,7 +259,7 @@ private T ReadInternal(FieldMetadata metadata) where T : unmanaged /// /// Writes an array to a fixed-size array field. - /// For arrays larger than 8 bytes total, automatic locking is applied. + /// Automatic locking is applied: the elements and the zeroed tail are separate stores. /// public void WriteArray(string fieldName, ReadOnlySpan values) where T : unmanaged { @@ -263,10 +279,10 @@ public void WriteArray(string fieldName, ReadOnlySpan values) where T : un nameof(values)); var bytes = MemoryMarshal.AsBytes(values); - bool isNonAtomic = metadata.Size > AtomicThreshold; - // Auto-lock for non-atomic operations (>8 bytes) - if (isNonAtomic && !IsHoldingWriteLock()) + // Always lock: the data and the zeroed tail are separate stores, so even a field of a few + // bytes could be observed half written. + if (!IsHoldingWriteLock()) { ThrowIfHoldingReadLock(); using var _ = AcquireWriteLock(); @@ -307,7 +323,7 @@ private void ClearBytes(long offset, int count) /// /// Reads an array from a fixed-size array field. - /// For arrays larger than 8 bytes total, automatic locking is applied. + /// Automatic locking is applied. /// public void ReadArray(string fieldName, Span destination) where T : unmanaged { @@ -328,9 +344,7 @@ public void ReadArray(string fieldName, Span destination) where T : unmana var bytes = MemoryMarshal.AsBytes(destination); - // Auto-lock for non-atomic operations (>8 bytes) - using var _ = (bytes.Length > AtomicThreshold && !IsHoldingAnyLock()) - ? AcquireReadLock() : default; + using var _ = !IsHoldingAnyLock() ? AcquireReadLock() : default; _buffer.Read(bytes, SchemaHeaderSize + metadata.Offset); } diff --git a/README.md b/README.md index 90b2c85..10f8d67 100644 --- a/README.md +++ b/README.md @@ -232,12 +232,14 @@ Console.WriteLine(reader[0]); The array header validates the element type fingerprint and restores its length for openers. -Elements that are 1, 2, 4 or 8 bytes wide are copied with a single aligned move, so another process -never sees half of one, and they are read and written without any lock. Every other element size -(a `Guid`, a `Vector3`, a 64-byte struct, ...) could be read torn while another process writes it, -so `SharedArray` takes the shared region lock for those types in the indexer, `CopyTo`, `CopyFrom` -and `Fill` (the whole range of a `Fill` under one lock). That is correct but slower per access; for -many elements take the lock once yourself. +An element that is 1, 2, 4 or 8 bytes wide is read and written with one aligned load or store, so +another process never sees half of one, and no lock is taken. A range of several elements (`CopyTo`, +`CopyFrom`, `Fill`) is a plain memory copy; take a lock when it must be a consistent snapshot. + +Every other element size (a `Guid`, a `Vector3`, a 64-byte struct, ...) could be read torn while +another process writes it, so `SharedArray` takes the shared region lock for those types in the +indexer, `CopyTo`, `CopyFrom` and `Fill` (the whole range of a `Fill` under one lock). That is correct +but slower per access; for many elements take the lock once yourself. Use the explicit locks when several elements must be read or changed together. This applies to any element type, because two atomic elements can still be seen from different updates: From 1056b8ad73ef3767fa45c50e9454cb716ef51445 Mon Sep 17 00:00:00 2001 From: zacala Date: Tue, 6 Oct 2026 16:05:23 +0000 Subject: [PATCH 16/29] Recover write locks that have no owner, and let readers recover dead writers The write lock is taken with a CAS on WriterLockState and the owner's pid is stored right after; release clears the pid and then the state. A process killed in either window leaves the lock held with pid 0, and the orphan check returns false for pid 0, so no waiter could ever recover it (14 of 60 kills of a process looping on lock/unlock). A waiter now clears a lock that has had no owner at every probe for two seconds. Readers did not probe the owner at all and waited out their whole timeout behind a writer that had been killed. They now share the writer's probe, the first one after one interval so a reader that waits for a live writer for microseconds pays nothing. The tests fail on the previous code and pass now. --- CHANGELOG.md | 5 + InterprocessMemory.Tests/CrossProcessTests.cs | 87 +++++++++++ InterprocessMemory/MemoryRegion.cs | 138 ++++++++++++++---- README.md | 6 +- 4 files changed, 205 insertions(+), 31 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 66c631f..7a85bf7 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -26,6 +26,11 @@ Changes since the 3.0.0 release. Migrating from 2.x: see [MIGRATION.md](MIGRATIO start tick count. - Waiting for a lock with `Timeout.InfiniteTimeSpan` now recovers from an owner that dies later (the owner is probed every 250 ms). +- A write lock whose owner was killed between taking the lock and recording its pid (or between clearing the + pid and releasing the lock) stayed held forever, because the orphan check needs a pid. A waiter now + clears a lock that has had no owner for two seconds. +- A waiting reader now recovers a write lock whose owner process died, like a waiting writer does. It used + to wait for its whole timeout. - Disposing a region while another thread waits for one of its locks no longer crashes the process; the waiter fails with `ObjectDisposedException`. - Generic unmanaged structs (`ValueTuple`, `KeyValuePair<,>`) work in all typed containers. diff --git a/InterprocessMemory.Tests/CrossProcessTests.cs b/InterprocessMemory.Tests/CrossProcessTests.cs index 1caaf73..c1c5ef8 100644 --- a/InterprocessMemory.Tests/CrossProcessTests.cs +++ b/InterprocessMemory.Tests/CrossProcessTests.cs @@ -2,6 +2,7 @@ using System.Collections.Generic; using System.Diagnostics; using System.IO; +using System.IO.MemoryMappedFiles; using System.Text; using System.Threading; using NUnit.Framework; @@ -412,6 +413,92 @@ public void CrossProcess_FiniteWait_RecoversPromptlyWhenOwnerDies() } } + [Test, Timeout(40000)] + public void CrossProcess_ReadWait_RecoversWhenWriterProcessDies() + { + // Only a waiting writer used to recover a lock whose owner had died; a reader (a dashboard that + // only reads) waited out its whole timeout behind a writer that no longer existed. + string name = GetUniqueName("ReaderBehindDeadWriter"); + using var region = MemoryRegion.CreateOrOpen(name, 256); + using Process holder = StartLockHolder(name); + try + { + Task reader = Task.Run(() => + { + bool acquired = region.TryAcquireReadLock(TimeSpan.FromSeconds(30)); + if (acquired) + region.ReleaseReadLock(); + return acquired; + }); + + Thread.Sleep(500); + KillAndWait(holder); + + Assert.That(reader.Wait(TimeSpan.FromSeconds(5)), Is.True, + "a reader must notice that the writer is gone, not wait for its whole timeout"); + Assert.That(reader.Result, Is.True); + } + finally + { + KillAndWait(holder); + } + } + + /// + /// Leaves the lock the way a process killed between "state = 1" and "owner = pid" does: held, but + /// with no owner recorded. Writes the region header through a second mapping of the /dev/shm file. + /// + private static void HoldLockWithNoOwner(string name) + { + if (!OperatingSystem.IsLinux()) + Assert.Ignore("Edits the header through the /dev/shm file, which only Linux has."); + + using var file = new FileStream("/dev/shm/" + name, FileMode.Open, FileAccess.ReadWrite, FileShare.ReadWrite); + using var map = MemoryMappedFile.CreateFromFile(file, null, 0, MemoryMappedFileAccess.ReadWrite, + HandleInheritability.None, leaveOpen: true); + using var view = map.CreateViewAccessor(0, 128); + view.Write(28, 0); // LockOwnerProcessId + view.Write(32, 0L); // LockOwnerThreadId + view.Write(24, 1); // WriterLockState + view.Flush(); + } + + [Test, Timeout(40000)] + public void CrossProcess_LockHeldWithNoOwner_IsReleasedByAWritingWaiter() + { + // A kill between the CAS that takes the lock and the store of the owner's pid (or between + // clearing the owner and clearing the state) leaves WriterLockState = 1 and pid = 0. The owner + // checks cannot see that, so the lock stayed held forever. + string name = GetUniqueName("NoOwnerWriter"); + using var region = MemoryRegion.CreateOrOpen(name, 256); + HoldLockWithNoOwner(name); + Assert.That(region.TryAcquireWriteLock(TimeSpan.FromMilliseconds(300)), Is.False, + "the lock is held and the owner check does not recognise it as an orphan"); + + var sw = Stopwatch.StartNew(); + Assert.That(region.TryAcquireWriteLock(TimeSpan.FromSeconds(10)), Is.True); + sw.Stop(); + region.ReleaseWriteLock(); + + Assert.That(sw.ElapsedMilliseconds, Is.GreaterThan(1500), + "a lock that merely looks ownerless for an instant must not be taken at once"); + } + + [Test, Timeout(40000)] + public void CrossProcess_LockHeldWithNoOwner_IsReleasedByAReadingWaiter() + { + string name = GetUniqueName("NoOwnerReader"); + using var region = MemoryRegion.CreateOrOpen(name, 256); + HoldLockWithNoOwner(name); + + var sw = Stopwatch.StartNew(); + Assert.That(region.TryAcquireReadLock(TimeSpan.FromSeconds(10)), Is.True); + sw.Stop(); + region.ReleaseReadLock(); + + Assert.That(sw.ElapsedMilliseconds, Is.GreaterThan(1500)); + } + [Test, Timeout(30000)] public void CrossProcess_DeadReaderProcess_NeedsForceResetLocks() { diff --git a/InterprocessMemory/MemoryRegion.cs b/InterprocessMemory/MemoryRegion.cs index e9b409c..34e539c 100644 --- a/InterprocessMemory/MemoryRegion.cs +++ b/InterprocessMemory/MemoryRegion.cs @@ -78,6 +78,13 @@ public sealed unsafe class MemoryRegion : IMemoryRegion // How often a lock waiter re-probes whether the owning process is still alive. private const long OrphanCheckIntervalMs = 250; + // A held write lock with no owner recorded is normal for a few nanoseconds: the owner sets + // WriterLockState first and writes its identity right after, and clears the identity before it + // clears the state. A process killed in one of those two windows leaves the lock held with nobody + // to attribute it to, which the owner checks cannot see. Waiters clear such a lock after it has + // looked like this at every probe for this long. + private const long OwnerlessLockGraceMs = 2000; + // Upper bound on how long Dispose() waits for lock waiters to notice the disposed flag. private static readonly TimeSpan s_waiterDrainTimeout = TimeSpan.FromSeconds(5); @@ -892,6 +899,7 @@ private bool TryAcquireWriteLockCore(TimeSpan timeout) long start = Stopwatch.GetTimestamp(); // not a Stopwatch instance: that would allocate on every lock var spinner = new SpinWait(); long nextOrphanCheckMs = 0; + long ownerlessSinceMs = -1; while (true) { @@ -946,20 +954,11 @@ private bool TryAcquireWriteLockCore(TimeSpan timeout) if (TimeoutHelper.HasExpired(start, timeout)) return false; - if (_options.EnableOrphanLockDetection && ElapsedMilliseconds(start) >= nextOrphanCheckMs) - { - // Check on the first CAS failure and then periodically. The owner may die at any - // point while we wait, and a wait with Timeout.InfiniteTimeSpan has no deadline to - // key a one-off re-check on. The probe costs a process lookup, hence the interval. - nextOrphanCheckMs = ElapsedMilliseconds(start) + OrphanCheckIntervalMs; - - if (IsWriteLockOrphaned()) - { - _logger?.LogWarning("Detected orphan write lock, attempting recovery"); - TryForceReleaseWriteLock(); - continue; - } - } + // Check on the first CAS failure and then periodically. The owner may die at any point + // while we wait, and a wait with Timeout.InfiniteTimeSpan has no deadline to key a + // one-off re-check on. The probe costs a process lookup, hence the interval. + if (TryRecoverStaleWriteLock(start, ref nextOrphanCheckMs, ref ownerlessSinceMs)) + continue; spinner.SpinOnce(); } @@ -1032,6 +1031,10 @@ private bool TryAcquireReadLockCore(TimeSpan timeout) var header = (SharedHeader*)_basePtr; long start = Stopwatch.GetTimestamp(); var spinner = new SpinWait(); + // A reader usually waits for a live writer for microseconds, so the first probe comes after a + // full interval instead of at once. A writer that died is still noticed within it. + long nextOrphanCheckMs = OrphanCheckIntervalMs; + long ownerlessSinceMs = -1; while (true) { @@ -1046,6 +1049,11 @@ private bool TryAcquireReadLockCore(TimeSpan timeout) { if (TimeoutHelper.HasExpired(start, timeout)) return false; + + // A reader must not wait out its whole timeout behind a writer that no longer exists. + if (TryRecoverStaleWriteLock(start, ref nextOrphanCheckMs, ref ownerlessSinceMs)) + continue; + spinner.SpinOnce(); continue; } @@ -1170,6 +1178,92 @@ public bool IsWriteLockOrphaned() return false; } + /// + /// Called by a waiter that finds the write lock held. Every + /// it releases the lock when its owner process is gone, or when the lock is held with no owner at + /// all (see ). Returns true when it released the lock, so that + /// the caller retries at once. + /// + private bool TryRecoverStaleWriteLock(long start, ref long nextCheckMs, ref long ownerlessSinceMs) + { + if (!_options.EnableOrphanLockDetection) + return false; + + long nowMs = ElapsedMilliseconds(start); + if (nowMs < nextCheckMs) + return false; + + nextCheckMs = nowMs + OrphanCheckIntervalMs; + + if (IsWriteLockOrphaned()) + { + _logger?.LogWarning("Detected orphan write lock, attempting recovery"); + TryForceReleaseWriteLock(); + return true; + } + + return TryReleaseOwnerlessWriteLock(nowMs, ref ownerlessSinceMs); + } + + private bool TryReleaseOwnerlessWriteLock(long nowMs, ref long ownerlessSinceMs) + { + var header = (SharedHeader*)_basePtr; + + if (Volatile.Read(ref header->WriterLockState) == 0 || + Volatile.Read(ref header->LockOwnerProcessId) != 0) + { + ownerlessSinceMs = -1; + return false; + } + + if (ownerlessSinceMs < 0) + { + ownerlessSinceMs = nowMs; + return false; + } + + if (nowMs - ownerlessSinceMs < OwnerlessLockGraceMs) + return false; + + // Only the exact state that was observed is cleared: a lock that was released and taken again + // meanwhile has an owner recorded within nanoseconds of its CAS and is left alone. + if (Volatile.Read(ref header->LockOwnerProcessId) != 0 || + Interlocked.CompareExchange(ref header->WriterLockState, 0, 1) != 1) + { + ownerlessSinceMs = -1; + return false; + } + + header->LockOwnerThreadId = 0; + header->LockOwnerProcessStartTime = 0; + header->LockAcquiredTimestamp = 0; + ownerlessSinceMs = -1; + + _logger?.LogWarning( + "Released a write lock that was held with no owner recorded for {Grace} ms " + + "(its owner was killed while taking or releasing it)", OwnerlessLockGraceMs); + RaiseOrphanLockDetected(); + return true; + } + + private void RaiseOrphanLockDetected() + { + if (!_options.EnableEvents) + return; + + try + { + OnOrphanLockDetected?.Invoke(this, new MemoryRegionEventArgs + { + EventType = MemoryRegionEventType.OrphanLockDetected + }); + } + catch (Exception ex) + { + _logger?.LogWarning(ex, "Event handler threw an exception during OnOrphanLockDetected"); + } + } + /// public bool TryForceReleaseWriteLock() { @@ -1198,21 +1292,7 @@ public bool TryForceReleaseWriteLock() Thread.MemoryBarrier(); Volatile.Write(ref header->WriterLockState, 0); - if (_options.EnableEvents) - { - try - { - OnOrphanLockDetected?.Invoke(this, new MemoryRegionEventArgs - { - EventType = MemoryRegionEventType.OrphanLockDetected - }); - } - catch (Exception ex) - { - _logger?.LogWarning(ex, "Event handler threw an exception during OnOrphanLockDetected"); - } - } - + RaiseOrphanLockDetected(); return true; } diff --git a/README.md b/README.md index 10f8d67..c4bfb9c 100644 --- a/README.md +++ b/README.md @@ -59,8 +59,10 @@ if (memory.TryAcquireWriteLock(TimeSpan.FromSeconds(1))) ``` Lock ownership includes the process ID, managed thread ID, and process start time. A process -waiting for a write lock keeps checking whether the owner is still alive (also when it waits -with `Timeout.InfiniteTimeSpan`) and recovers a lock left behind by a terminated process. +waiting for a write lock or a read lock keeps checking whether the writer that holds the lock is +still alive (also when it waits with `Timeout.InfiniteTimeSpan`) and recovers a lock left behind by a +terminated process. A process killed exactly while it takes or releases the lock can leave it held +with no owner recorded; a waiter clears such a lock once it has looked like that for two seconds. Recovery is based on the owner process being gone. A write lock held by a process that is still alive is never taken over unless you opt in with `MemoryRegionOptions.OrphanLockTimeout`, which From 67334f8b65ea6b9bfb645e6bc015ecf9968fd96e Mon Sep 17 00:00:00 2001 From: zacala Date: Tue, 6 Oct 2026 16:06:20 +0000 Subject: [PATCH 17/29] Give the MPMC queues at least two slots The per-slot sequence is write+1 once an item is published and read+capacity once its slot is free again. With one slot those are the same number, so the second enqueue saw a full queue as empty, overwrote the item and left the queue unable to deliver anything (enqueue(1) and enqueue(2) both returned true, every dequeue after that false). ConcurrentQueue and ConcurrentMessageQueue now raise a requested capacity of 1 to 2 and reject a stored capacity below 2. --- CHANGELOG.md | 4 ++ .../LibraryHardeningTests.cs | 58 +++++++++++++++++++ InterprocessMemory/ConcurrentMessageQueue.cs | 9 ++- InterprocessMemory/ConcurrentQueue.cs | 11 +++- InterprocessMemory/PowerOfTwo.cs | 8 ++- README.md | 3 +- 6 files changed, 83 insertions(+), 10 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 7a85bf7..b054c53 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -33,6 +33,10 @@ Changes since the 3.0.0 release. Migrating from 2.x: see [MIGRATION.md](MIGRATIO to wait for its whole timeout. - Disposing a region while another thread waits for one of its locks no longer crashes the process; the waiter fails with `ObjectDisposedException`. +- `ConcurrentQueue` and `ConcurrentMessageQueue` with a capacity of 1 overwrote the stored item when a second + one was enqueued and then never delivered anything again. They need two slots, so a requested capacity + of 1 is now raised to 2, and an existing region that stored a capacity of 1 is rejected as invalid + (remove it with `MemoryRegion.Remove`). - Generic unmanaged structs (`ValueTuple`, `KeyValuePair<,>`) work in all typed containers. Fingerprints of types that already worked are unchanged, so existing regions stay compatible. - Two byte `SharedArray` elements could be read half written by another process (a two byte copy is a diff --git a/InterprocessMemory.Tests/LibraryHardeningTests.cs b/InterprocessMemory.Tests/LibraryHardeningTests.cs index b05d307..d0f51ca 100644 --- a/InterprocessMemory.Tests/LibraryHardeningTests.cs +++ b/InterprocessMemory.Tests/LibraryHardeningTests.cs @@ -643,6 +643,64 @@ public IEnumerable GetFields() } } + [Test, Timeout(30000)] + public void ConcurrentQueue_CapacityOne_IsRaisedToTwoAndStillReportsFull() + { + // With one slot the sequence that marks a slot published (write + 1) equals the one that marks it + // free again (read + capacity), so the second enqueue overwrote the first item and the queue then + // never produced anything again. + using var queue = ConcurrentQueue.CreateOrOpen(N("MpmcCap1"), 1); + + Assert.That(queue.Capacity, Is.EqualTo(2)); + Assert.That(queue.TryEnqueue(10), Is.True); + Assert.That(queue.TryEnqueue(20), Is.True); + Assert.That(queue.TryEnqueue(30), Is.False, "the queue is full, it must not overwrite item 10"); + + Assert.That(queue.TryDequeue(out int first), Is.True); + Assert.That(first, Is.EqualTo(10)); + Assert.That(queue.TryDequeue(out int second), Is.True); + Assert.That(second, Is.EqualTo(20)); + Assert.That(queue.TryDequeue(out _), Is.False); + + for (int i = 0; i < 10; i++) + { + Assert.That(queue.TryEnqueue(i), Is.True); + Assert.That(queue.TryDequeue(out int value), Is.True); + Assert.That(value, Is.EqualTo(i)); + } + } + + [Test, Timeout(30000)] + public void ConcurrentMessageQueue_CapacityOne_IsRaisedToTwoAndStillReportsFull() + { + using var queue = ConcurrentMessageQueue.CreateOrOpen(N("MqCap1"), 1, 16); + var a = new byte[] { 1 }; + var b = new byte[] { 2 }; + var buffer = new byte[16]; + + Assert.That(queue.Capacity, Is.EqualTo(2)); + Assert.That(queue.TryEnqueue(a), Is.True); + Assert.That(queue.TryEnqueue(b), Is.True); + Assert.That(queue.TryEnqueue(new byte[] { 3 }), Is.False, "the queue is full, it must not overwrite message 1"); + + Assert.That(queue.TryDequeue(buffer, out int length), Is.True); + Assert.That(length, Is.EqualTo(1)); + Assert.That(buffer[0], Is.EqualTo(1)); + Assert.That(queue.TryDequeue(buffer, out length), Is.True); + Assert.That(buffer[0], Is.EqualTo(2)); + Assert.That(queue.TryDequeue(buffer, out _), Is.False); + } + + [Test, Timeout(30000)] + public void ConcurrentQueue_ReopenedWithCapacityOne_MatchesTheRaisedCapacity() + { + string name = N("MpmcCap1Reopen"); + using var owner = ConcurrentQueue.CreateOrOpen(name, 1); + using var reopened = ConcurrentQueue.CreateOrOpen(name, 1); + + Assert.That(reopened.Capacity, Is.EqualTo(2)); + } + public struct BlobSchema : IMemorySchema { public const string Data = "Data"; diff --git a/InterprocessMemory/ConcurrentMessageQueue.cs b/InterprocessMemory/ConcurrentMessageQueue.cs index f64bccc..4efb155 100644 --- a/InterprocessMemory/ConcurrentMessageQueue.cs +++ b/InterprocessMemory/ConcurrentMessageQueue.cs @@ -57,6 +57,9 @@ private struct Slot } private const int HeaderSize = 384; // 6 cache lines for false-sharing prevention + + // See ConcurrentQueue: with one slot the published and the free sequence are equal. + private const int MinimumCapacity = 2; private const int SlotHeaderSize = 16; private const int FormatVersion = 3; private const long HeaderMagic = 0x514D434D504953; @@ -194,7 +197,7 @@ private ConcurrentMessageQueue( if (createOrOpen) { - _slotCount = PowerOfTwo.RoundUp(capacity!.Value, "capacity"); + _slotCount = PowerOfTwo.RoundUp(capacity!.Value, "capacity", MinimumCapacity); _maxMessageSize = maxMessageSize!.Value; _slotTotalSize = RoundUpToMultiple( checked(SlotHeaderSize + _maxMessageSize), 8); @@ -305,13 +308,13 @@ private void ValidateAndLoadBuffer(int? requestedCapacity, int? requestedMaxMess ex); } - if (storedSlotCount <= 0 || (storedSlotCount & (storedSlotCount - 1)) != 0 || + if (storedSlotCount < MinimumCapacity || (storedSlotCount & (storedSlotCount - 1)) != 0 || storedMaxMessageSize <= 0 || storedSlotStride != expectedSlotStride) throw new InvalidDataException("The message queue header contains invalid sizing metadata."); if (requestedCapacity.HasValue) { - int expectedCapacity = PowerOfTwo.RoundUp(requestedCapacity.Value, "capacity"); + int expectedCapacity = PowerOfTwo.RoundUp(requestedCapacity.Value, "capacity", MinimumCapacity); if (expectedCapacity != storedSlotCount) throw new InvalidOperationException( $"Capacity mismatch: expected {expectedCapacity}, found {storedSlotCount}"); diff --git a/InterprocessMemory/ConcurrentQueue.cs b/InterprocessMemory/ConcurrentQueue.cs index 2fdfd79..de1b7c4 100644 --- a/InterprocessMemory/ConcurrentQueue.cs +++ b/InterprocessMemory/ConcurrentQueue.cs @@ -38,6 +38,11 @@ internal struct ConcurrentQueueSlot public sealed unsafe class ConcurrentQueue : IDisposable where T : unmanaged { private const int HeaderSize = 320; + + // A slot's sequence is write+1 once it is published and read+capacity once it is free again. + // With a single slot those two values are the same, so a full queue looks empty to the next + // producer, which overwrites the item and then stalls the queue for good. + private const int MinimumCapacity = 2; private const int SlotHeaderSize = 8; private const int FormatVersion = 3; private const long HeaderMagic = 0x5143504D504953; @@ -99,7 +104,7 @@ private ConcurrentQueue( if (createOrOpen) { - _capacity = PowerOfTwo.RoundUp(capacity!.Value, "capacity"); + _capacity = PowerOfTwo.RoundUp(capacity!.Value, "capacity", MinimumCapacity); _capacityMask = _capacity - 1; _slotStride = RoundUpToMultiple(checked(SlotHeaderSize + _elementSize), 8); long regionCapacity = checked(HeaderSize + (long)_capacity * _slotStride); @@ -171,7 +176,7 @@ private void ValidateAndLoad(int? requestedCapacity) int storedCapacity = _header->Capacity; int storedStride = _header->SlotStride; if (_header->Version != FormatVersion || - storedCapacity <= 0 || + storedCapacity < MinimumCapacity || (storedCapacity & (storedCapacity - 1)) != 0 || _header->ElementSize != _elementSize || storedStride < SlotHeaderSize + _elementSize || @@ -181,7 +186,7 @@ private void ValidateAndLoad(int? requestedCapacity) if (requestedCapacity.HasValue) { - int expectedCapacity = PowerOfTwo.RoundUp(requestedCapacity.Value, "capacity"); + int expectedCapacity = PowerOfTwo.RoundUp(requestedCapacity.Value, "capacity", MinimumCapacity); if (expectedCapacity != storedCapacity) throw new InvalidOperationException( $"Capacity mismatch: expected {expectedCapacity}, found {storedCapacity}."); diff --git a/InterprocessMemory/PowerOfTwo.cs b/InterprocessMemory/PowerOfTwo.cs index 5a5f392..79e856e 100644 --- a/InterprocessMemory/PowerOfTwo.cs +++ b/InterprocessMemory/PowerOfTwo.cs @@ -10,9 +10,11 @@ internal static class PowerOfTwo /// /// Rounds a requested item count up to the next power of two, which the ring buffers need so - /// they can wrap with a mask instead of a modulo. + /// they can wrap with a mask instead of a modulo. Counts below are + /// raised to it (the sequence-numbered MPMC queues cannot tell a full slot from an empty one + /// with a single slot, so they pass 2). /// - public static int RoundUp(int value, string paramName) + public static int RoundUp(int value, string paramName, int minimum = 1) { if (value <= 0 || value > MaxInt32) throw new ArgumentOutOfRangeException( @@ -20,7 +22,7 @@ public static int RoundUp(int value, string paramName) value, $"The value must be between 1 and {MaxInt32} so it can be rounded up to a power of two."); - return (int)BitOperations.RoundUpToPowerOf2((uint)value); + return (int)BitOperations.RoundUpToPowerOf2((uint)Math.Max(value, minimum)); } } } diff --git a/README.md b/README.md index c4bfb9c..deeace9 100644 --- a/README.md +++ b/README.md @@ -129,7 +129,8 @@ if (consumer.TryDequeue(out SensorSample sample)) } ``` -Queue capacity is an item count, not bytes, and is rounded up to the next power of two. +Queue capacity is an item count, not bytes, and is rounded up to the next power of two. The two +multi-producer queues need at least two slots, so a requested capacity of 1 becomes 2. `Capacity` reports the resulting number of slots. The shared header records `sizeof(T)` and a deterministic fingerprint of the type name, From 0604f8441d9a4c6e0f5f7c4e87c28b41eedc4465 Mon Sep 17 00:00:00 2001 From: zacala Date: Tue, 6 Oct 2026 16:07:22 +0000 Subject: [PATCH 18/29] Check the free space before creating a region On Linux a new region is a sparse file in /dev/shm. SetLength succeeds even when the filesystem cannot hold it, and the process is killed with SIGBUS (not catchable) by the first write that does not fit. Docker's default /dev/shm is 64 MB, so this is easy to hit. The creator now compares the size with the free space of the filesystem and throws IOException with a hint about --shm-size; the explicit FilePath mode gets the same check for the bytes it has to add. A filesystem that cannot report its free space is not blocked. --- CHANGELOG.md | 3 ++ .../LibraryHardeningTests.cs | 27 +++++++++++++ InterprocessMemory/MemoryRegion.cs | 38 +++++++++++++++++++ README.md | 4 ++ 4 files changed, 72 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index b054c53..b091687 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -37,6 +37,9 @@ Changes since the 3.0.0 release. Migrating from 2.x: see [MIGRATION.md](MIGRATIO one was enqueued and then never delivered anything again. They need two slots, so a requested capacity of 1 is now raised to 2, and an existing region that stored a capacity of 1 is rejected as invalid (remove it with `MemoryRegion.Remove`). +- Linux: creating a region larger than the free space of `/dev/shm` (Docker's default is 64 MB) succeeded and + the process was killed with an uncatchable `SIGBUS` at the first write that did not fit. `CreateOrOpen` + now throws `IOException` up front; a file-backed region (`MemoryRegionOptions.FilePath`) is checked too. - Generic unmanaged structs (`ValueTuple`, `KeyValuePair<,>`) work in all typed containers. Fingerprints of types that already worked are unchanged, so existing regions stay compatible. - Two byte `SharedArray` elements could be read half written by another process (a two byte copy is a diff --git a/InterprocessMemory.Tests/LibraryHardeningTests.cs b/InterprocessMemory.Tests/LibraryHardeningTests.cs index d0f51ca..523fe31 100644 --- a/InterprocessMemory.Tests/LibraryHardeningTests.cs +++ b/InterprocessMemory.Tests/LibraryHardeningTests.cs @@ -701,6 +701,33 @@ public void ConcurrentQueue_ReopenedWithCapacityOne_MatchesTheRaisedCapacity() Assert.That(reopened.Capacity, Is.EqualTo(2)); } + [Test] + public void EnsureFreeSpace_ThrowsWhenTheFilesystemCannotHoldTheRegion() + { + string directory = Path.GetTempPath(); + + Assert.DoesNotThrow(() => MemoryRegion.EnsureFreeSpace(directory, 1024, "small")); + var ex = Assert.Throws(() => MemoryRegion.EnsureFreeSpace(directory, long.MaxValue, "huge")); + Assert.That(ex!.Message, Does.Contain("huge")); + Assert.That(ex.Message, Does.Contain("--shm-size")); + + // A path the OS cannot describe must not block creating the region. + Assert.DoesNotThrow(() => MemoryRegion.EnsureFreeSpace("\0:/does-not-exist", long.MaxValue, "unknown")); + } + + [Test, Timeout(30000)] + public void CreateOrOpen_WithMoreThanTheDevShmCanHold_ThrowsInsteadOfCrashingLater() + { + if (!OperatingSystem.IsLinux()) + Assert.Ignore("Only the Linux /dev/shm backing can be oversubscribed."); + + // A sparse tmpfs file of this size is created without error, and the process dies with SIGBUS + // at the first write that does not fit. 4 TiB is far beyond any /dev/shm. + string name = N("TooBig"); + Assert.Throws(() => MemoryRegion.CreateOrOpen(name, 4L << 40)); + Assert.That(File.Exists("/dev/shm/" + name), Is.False, "the failed region must not leave a file behind"); + } + public struct BlobSchema : IMemorySchema { public const string Data = "Data"; diff --git a/InterprocessMemory/MemoryRegion.cs b/InterprocessMemory/MemoryRegion.cs index 34e539c..dde2643 100644 --- a/InterprocessMemory/MemoryRegion.cs +++ b/InterprocessMemory/MemoryRegion.cs @@ -446,6 +446,14 @@ private void Initialize() /// private void CreateMmfFromExplicitFilePath(long totalSize) { + if (_createOrOpen) + { + string fullPath = Path.GetFullPath(_options.FilePath!); + long existing = File.Exists(fullPath) ? new FileInfo(fullPath).Length : 0; + if (existing < totalSize) + EnsureFreeSpace(Path.GetDirectoryName(fullPath) ?? fullPath, totalSize - existing, _name); + } + string? mapName = OperatingSystem.IsWindows() ? _name : null; FileMode mode = _createOrOpen ? FileMode.OpenOrCreate : FileMode.Open; _mmf = MemoryMappedFile.CreateFromFile( @@ -510,6 +518,10 @@ private void CreateMmfFromLinuxDevShm(long totalSize) // file: if construction fails between here and AcquirePointer, Cleanup will // unlink the file so the next caller doesn't trip over a half-initialized blob. _createdBackingFile = true; + + // SetLength only reserves address space in tmpfs. If the filesystem cannot hold the region, + // the first write past what fits kills the process with SIGBUS, which cannot be caught. + EnsureFreeSpace("/dev/shm", totalSize, _name); _backingFile.SetLength(totalSize); } else if (_createOrOpen && _backingFile.Length != totalSize) @@ -533,6 +545,32 @@ private void CreateMmfFromLinuxDevShm(long totalSize) _backingFile = null; // ownership transferred — don't double-dispose } + /// + /// Throws when the filesystem that will hold a new region has fewer than + /// bytes available. Best effort: a filesystem that cannot report + /// its free space is not blocked, and another process can still fill it between this check and the + /// first write. + /// + internal static void EnsureFreeSpace(string directory, long requiredBytes, string regionName) + { + long free; + try + { + free = new DriveInfo(directory).AvailableFreeSpace; + } + catch (Exception ex) when (ex is ArgumentException or IOException or UnauthorizedAccessException + or NotSupportedException) + { + return; + } + + if (free < requiredBytes) + throw new IOException( + $"Not enough space for shared memory '{regionName}': it needs {requiredBytes} bytes but " + + $"'{directory}' has {free} available. In a container, raise the limit of /dev/shm " + + "(docker run --shm-size, or a larger memory-backed emptyDir in Kubernetes)."); + } + /// /// Validates the cross-platform flat identifier used for the named region. /// diff --git a/README.md b/README.md index deeace9..f2a2b1a 100644 --- a/README.md +++ b/README.md @@ -93,6 +93,10 @@ The `name` must be the same non-empty flat identifier in every process. Path sep control characters, NUL, and UTF-8 names longer than 255 bytes are rejected consistently on Windows and Linux. +On Linux a region lives in `/dev/shm`, which Docker limits to 64 MB by default. Creating a region that +does not fit throws `IOException` (raise the limit with `--shm-size`, or a larger memory-backed +`emptyDir` in Kubernetes) instead of killing the process with `SIGBUS` on a later write. + ## Typed queues Typed queues accept only fixed-size `unmanaged` values. They copy the value directly to a slot; From 80960975c88df77f990ff48135aefd574a07ec09 Mon Sep 17 00:00:00 2001 From: zacala Date: Tue, 6 Oct 2026 16:09:26 +0000 Subject: [PATCH 19/29] Do not decide from a pid that the lock owner is gone across PID namespaces Containers that share /dev/shm have different PID namespaces. A waiter looked the owner's pid up in its own namespace, did not find it (or found an unrelated process) and either took the lock from a live owner (14 ms in a test with unshare --pid) or never recovered a dead one. The owner records the inode of /proc/self/ns/pid in the reserved part of the region header, before its pid and cleared with the other owner fields. A waiter whose namespace differs from the recorded one skips the pid-based liveness check; the opt-in OrphanLockTimeout still applies. Locks without a recorded namespace (earlier versions, other platforms) keep the pid-only decision. --- CHANGELOG.md | 4 + InterprocessMemory.Tests/CrossProcessTests.cs | 87 +++++++++++++++ InterprocessMemory/MemoryRegion.cs | 100 ++++++++++++++---- README.md | 4 + 4 files changed, 173 insertions(+), 22 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index b091687..c7f8340 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -29,6 +29,10 @@ Changes since the 3.0.0 release. Migrating from 2.x: see [MIGRATION.md](MIGRATIO - A write lock whose owner was killed between taking the lock and recording its pid (or between clearing the pid and releasing the lock) stayed held forever, because the orphan check needs a pid. A waiter now clears a lock that has had no owner for two seconds. +- Linux: processes in different PID namespaces that share `/dev/shm` (containers) took each other's locks, because a + waiter looked the owner's pid up in its own namespace and did not find it. The owner now records its PID + namespace (the inode of `/proc/self/ns/pid`, in the reserved part of the header) and a waiter in another + namespace no longer declares the owner dead from its pid. - A waiting reader now recovers a write lock whose owner process died, like a waiting writer does. It used to wait for its whole timeout. - Disposing a region while another thread waits for one of its locks no longer crashes the process; diff --git a/InterprocessMemory.Tests/CrossProcessTests.cs b/InterprocessMemory.Tests/CrossProcessTests.cs index c1c5ef8..011e600 100644 --- a/InterprocessMemory.Tests/CrossProcessTests.cs +++ b/InterprocessMemory.Tests/CrossProcessTests.cs @@ -499,6 +499,93 @@ public void CrossProcess_LockHeldWithNoOwner_IsReleasedByAReadingWaiter() Assert.That(sw.ElapsedMilliseconds, Is.GreaterThan(1500)); } + /// + /// Makes the write lock look as if a process with in the PID namespace + /// held it (the pid must not exist here), by editing the region header + /// through a second mapping of the /dev/shm file. + /// + private static void HoldLockAs(string name, int pid, long pidNamespace) + { + if (!OperatingSystem.IsLinux()) + Assert.Ignore("Edits the header through the /dev/shm file, which only Linux has."); + if (MemoryRegion.CurrentPidNamespace == 0) + Assert.Ignore("The PID namespace of this process cannot be read (no /proc)."); + + using var file = new FileStream("/dev/shm/" + name, FileMode.Open, FileAccess.ReadWrite, FileShare.ReadWrite); + using var map = MemoryMappedFile.CreateFromFile(file, null, 0, MemoryMappedFileAccess.ReadWrite, + HandleInheritability.None, leaveOpen: true); + using var view = map.CreateViewAccessor(0, 128); + view.Write(88, pidNamespace); // LockOwnerPidNamespace + view.Write(48, 0L); // LockOwnerProcessStartTime: unknown, so the pid decides + view.Write(32, 1L); // LockOwnerThreadId + view.Write(28, pid); // LockOwnerProcessId + view.Write(24, 1); // WriterLockState + view.Flush(); + } + + private const int PidThatIsNotRunning = 4_000_001; + + [Test, Timeout(30000)] + public void CrossProcess_OwnerInAnotherPidNamespace_IsNotTakenOver() + { + // Two containers that share /dev/shm have different PID namespaces. The owner's pid is looked up + // in the waiter's namespace, where it does not exist (or is somebody else), so the waiter used to + // decide that a live owner was dead and took its lock. + string name = GetUniqueName("OtherPidNamespace"); + using var region = MemoryRegion.CreateOrOpen(name, 256); + HoldLockAs(name, PidThatIsNotRunning, MemoryRegion.CurrentPidNamespace + 1); + + Assert.That(region.IsWriteLockOrphaned(), Is.False); + Assert.That(region.GetLockOwnerInfo().IsOrphan, Is.False); + Assert.That(region.TryAcquireWriteLock(TimeSpan.FromMilliseconds(1500)), Is.False, + "the owner lives in another PID namespace, its lock is not ours to take"); + Assert.That(region.TryAcquireReadLock(TimeSpan.FromMilliseconds(600)), Is.False); + } + + [Test, Timeout(30000)] + public void CrossProcess_OwnerInTheSamePidNamespace_IsStillRecovered() + { + string name = GetUniqueName("SamePidNamespace"); + using var region = MemoryRegion.CreateOrOpen(name, 256); + HoldLockAs(name, PidThatIsNotRunning, MemoryRegion.CurrentPidNamespace); + + Assert.That(region.IsWriteLockOrphaned(), Is.True); + Assert.That(region.TryAcquireWriteLock(TimeSpan.FromSeconds(5)), Is.True); + region.ReleaseWriteLock(); + } + + [Test, Timeout(30000)] + public void CrossProcess_OwnerWithoutRecordedPidNamespace_IsStillRecovered() + { + // Locks taken by earlier versions record no namespace; they keep the pid-only decision. + string name = GetUniqueName("UnknownPidNamespace"); + using var region = MemoryRegion.CreateOrOpen(name, 256); + HoldLockAs(name, PidThatIsNotRunning, 0); + + Assert.That(region.IsWriteLockOrphaned(), Is.True); + Assert.That(region.TryAcquireWriteLock(TimeSpan.FromSeconds(5)), Is.True); + region.ReleaseWriteLock(); + } + + [Test, Timeout(30000)] + public void CrossProcess_LiveOwner_RecordsItsPidNamespaceAndClearsItOnRelease() + { + if (!OperatingSystem.IsLinux() || MemoryRegion.CurrentPidNamespace == 0) + Assert.Ignore("Needs the /dev/shm file and a readable /proc/self/ns/pid."); + + string name = GetUniqueName("RecordedNamespace"); + using var region = MemoryRegion.CreateOrOpen(name, 256); + using var file = new FileStream("/dev/shm/" + name, FileMode.Open, FileAccess.ReadWrite, FileShare.ReadWrite); + using var map = MemoryMappedFile.CreateFromFile(file, null, 0, MemoryMappedFileAccess.Read, + HandleInheritability.None, leaveOpen: true); + using var view = map.CreateViewAccessor(0, 128, MemoryMappedFileAccess.Read); + + Assert.That(region.TryAcquireWriteLock(TimeSpan.FromSeconds(1)), Is.True); + Assert.That(view.ReadInt64(88), Is.EqualTo(MemoryRegion.CurrentPidNamespace)); + region.ReleaseWriteLock(); + Assert.That(view.ReadInt64(88), Is.EqualTo(0L), "a stale namespace would mislead the next owner's waiters"); + } + [Test, Timeout(30000)] public void CrossProcess_DeadReaderProcess_NeedsForceResetLocks() { diff --git a/InterprocessMemory/MemoryRegion.cs b/InterprocessMemory/MemoryRegion.cs index dde2643..d382e17 100644 --- a/InterprocessMemory/MemoryRegion.cs +++ b/InterprocessMemory/MemoryRegion.cs @@ -131,6 +131,37 @@ public static TimeSpan DisposeGracePeriod // milliseconds and would make every live owner look like an impostor. private static readonly long s_processStartTimeBinary = TryCaptureProcessStartTime(); + private static readonly long s_pidNamespace = TryReadLinuxPidNamespace(); + + /// Identifier of this process's PID namespace, 0 when it cannot be determined (not Linux). + internal static long CurrentPidNamespace => s_pidNamespace; + + // /proc/self/ns/pid is a symlink whose target looks like "pid:[4026531836]". + private static long TryReadLinuxPidNamespace() + { + if (!OperatingSystem.IsLinux()) + return 0; + + try + { + string? target = new FileInfo("/proc/self/ns/pid").LinkTarget; + int open = target?.IndexOf('[') ?? -1; + int close = target?.IndexOf(']') ?? -1; + if (target != null && open >= 0 && close > open && + long.TryParse(target.AsSpan(open + 1, close - open - 1), out long id)) + return id; + } + catch (Exception ex) when (ex is IOException or UnauthorizedAccessException or NotSupportedException) + { + // No /proc, or a restricted one: leave it unknown and keep the pid-only behaviour. + } + + return 0; + } + + private static bool IsOtherPidNamespace(long ownerNamespace) => + ownerNamespace != 0 && s_pidNamespace != 0 && ownerNamespace != s_pidNamespace; + private static long TryCaptureProcessStartTime() { if (OperatingSystem.IsLinux()) @@ -237,7 +268,12 @@ private struct SharedHeader [FieldOffset(72)] public long ChecksumOffset; [FieldOffset(80)] public int ChecksumLength; [FieldOffset(84)] public int RegionKind; - // bytes 88–127: reserved + // PID namespace (inode of /proc/self/ns/pid) of the process that holds the write lock, 0 when + // unknown. A pid only identifies a process inside its own namespace, so waiters in another + // namespace (two containers sharing /dev/shm) must not use LockOwnerProcessId to decide that + // the owner is gone. It is written before the pid and cleared with the other owner fields. + [FieldOffset(88)] public long LockOwnerPidNamespace; + // bytes 96–127: reserved } private const long HeaderSize = SharedHeader.Size; @@ -633,6 +669,7 @@ private bool InitializeOrOpen() header->LockOwnerProcessId = 0; header->LockOwnerThreadId = 0; header->LockOwnerProcessStartTime = 0; + header->LockOwnerPidNamespace = 0; header->LockAcquiredTimestamp = 0; header->DataChecksum = 0; header->ChecksumOffset = 0; @@ -952,7 +989,8 @@ private bool TryAcquireWriteLockCore(TimeSpan timeout) { // Record lock ownership for orphan detection. StartTime defeats PID-reuse // attacks on the orphan check (see IsWriteLockOrphaned). - header->LockOwnerProcessId = Environment.ProcessId; + header->LockOwnerPidNamespace = s_pidNamespace; + Volatile.Write(ref header->LockOwnerProcessId, Environment.ProcessId); header->LockOwnerThreadId = Environment.CurrentManagedThreadId; header->LockOwnerProcessStartTime = s_processStartTimeBinary; header->LockAcquiredTimestamp = Stopwatch.GetTimestamp(); @@ -983,6 +1021,7 @@ private bool TryAcquireWriteLockCore(TimeSpan timeout) header->LockOwnerProcessId = 0; header->LockOwnerThreadId = 0; header->LockOwnerProcessStartTime = 0; + header->LockOwnerPidNamespace = 0; header->LockAcquiredTimestamp = 0; Interlocked.Exchange(ref header->WriterLockState, 0); } @@ -1039,6 +1078,7 @@ public void ReleaseWriteLock() header->LockOwnerThreadId = 0; header->LockOwnerProcessStartTime = 0; + header->LockOwnerPidNamespace = 0; header->LockAcquiredTimestamp = 0; Thread.MemoryBarrier(); @@ -1152,7 +1192,39 @@ public bool IsWriteLockOrphaned() if (ownerPid == 0) return false; - // Check if process is still alive + // A pid recorded in another PID namespace says nothing about whether the owner is alive here. + if (!IsOtherPidNamespace(Volatile.Read(ref header->LockOwnerPidNamespace)) && + OwnerProcessIsGone(header, ownerPid)) + return true; + + // The owner process is alive. Only the opt-in time limit (OrphanLockTimeout, disabled + // by default) may still declare the lock orphaned. + if (_options.OrphanLockTimeout > TimeSpan.Zero) + { + long acquiredTimestamp = header->LockAcquiredTimestamp; + if (acquiredTimestamp > 0) + { + long elapsed = Stopwatch.GetTimestamp() - acquiredTimestamp; + double elapsedMs = elapsed * 1000.0 / Stopwatch.Frequency; + + if (elapsedMs > _options.OrphanLockTimeout.TotalMilliseconds) + { + _logger?.LogWarning("Lock held for {Elapsed}ms exceeds timeout {Timeout}ms", + elapsedMs, _options.OrphanLockTimeout.TotalMilliseconds); + return true; + } + } + } + + return false; + } + + /// + /// True when the process that recorded itself as the lock owner no longer exists: it has exited, or + /// its pid now belongs to a process that started later (pid reuse). + /// + private bool OwnerProcessIsGone(SharedHeader* header, int ownerPid) + { try { using var process = Process.GetProcessById(ownerPid); @@ -1194,25 +1266,6 @@ public bool IsWriteLockOrphaned() return true; } - // The owner process is alive. Only the opt-in time limit (OrphanLockTimeout, disabled - // by default) may still declare the lock orphaned. - if (_options.OrphanLockTimeout > TimeSpan.Zero) - { - long acquiredTimestamp = header->LockAcquiredTimestamp; - if (acquiredTimestamp > 0) - { - long elapsed = Stopwatch.GetTimestamp() - acquiredTimestamp; - double elapsedMs = elapsed * 1000.0 / Stopwatch.Frequency; - - if (elapsedMs > _options.OrphanLockTimeout.TotalMilliseconds) - { - _logger?.LogWarning("Lock held for {Elapsed}ms exceeds timeout {Timeout}ms", - elapsedMs, _options.OrphanLockTimeout.TotalMilliseconds); - return true; - } - } - } - return false; } @@ -1274,6 +1327,7 @@ private bool TryReleaseOwnerlessWriteLock(long nowMs, ref long ownerlessSinceMs) header->LockOwnerThreadId = 0; header->LockOwnerProcessStartTime = 0; + header->LockOwnerPidNamespace = 0; header->LockAcquiredTimestamp = 0; ownerlessSinceMs = -1; @@ -1326,6 +1380,7 @@ public bool TryForceReleaseWriteLock() header->LockOwnerThreadId = 0; header->LockOwnerProcessStartTime = 0; + header->LockOwnerPidNamespace = 0; header->LockAcquiredTimestamp = 0; Thread.MemoryBarrier(); Volatile.Write(ref header->WriterLockState, 0); @@ -1362,6 +1417,7 @@ public void ForceResetLocks() Volatile.Write(ref header->LockOwnerProcessId, 0); header->LockOwnerThreadId = 0; header->LockOwnerProcessStartTime = 0; + header->LockOwnerPidNamespace = 0; header->LockAcquiredTimestamp = 0; Thread.MemoryBarrier(); Volatile.Write(ref header->WriterLockState, 0); diff --git a/README.md b/README.md index f2a2b1a..1dcf540 100644 --- a/README.md +++ b/README.md @@ -64,6 +64,10 @@ still alive (also when it waits with `Timeout.InfiniteTimeSpan`) and recovers a terminated process. A process killed exactly while it takes or releases the lock can leave it held with no owner recorded; a waiter clears such a lock once it has looked like that for two seconds. +A process id only means something inside its own PID namespace. On Linux the owner records its namespace +in the region header, and a waiter in a different one (two containers sharing `/dev/shm`) never decides +from the pid alone that the owner is gone; only the opt-in `OrphanLockTimeout` can take such a lock over. + Recovery is based on the owner process being gone. A write lock held by a process that is still alive is never taken over unless you opt in with `MemoryRegionOptions.OrphanLockTimeout`, which trades mutual exclusion for liveness when a critical section may run longer than the timeout. From 4e4123abf702dc785763d332df77f493ff4ad5b0 Mon Sep 17 00:00:00 2001 From: zacala Date: Tue, 6 Oct 2026 16:11:09 +0000 Subject: [PATCH 20/29] Fix three small defects found in review - StructuredMemory.AcquireWriteLock with a read guard held on the thread set the writer flag and waited for the thread's own read lock, which blocked every other process until the timeout (forever with an infinite timeout). It throws InvalidOperationException immediately now, as the automatic write lock and SharedArray already did. - SingleProducerByteStream.Available and .Used had no disposed check and read the unmapped header. - The timeout overloads of ConcurrentQueue and ConcurrentMessageQueue incremented the failure counters on every poll. A call counts as one failure when it gives up. Polling also uses a timestamp instead of a Stopwatch allocation. --- CHANGELOG.md | 8 +++ .../LibraryHardeningTests.cs | 72 +++++++++++++++++++ InterprocessMemory/ConcurrentMessageQueue.cs | 33 ++++++--- InterprocessMemory/ConcurrentQueue.cs | 32 ++++++--- .../SingleProducerByteStream.cs | 18 ++++- InterprocessMemory/StructuredMemory.cs | 8 +++ 6 files changed, 148 insertions(+), 23 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index c7f8340..a4e4cbc 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -33,6 +33,14 @@ Changes since the 3.0.0 release. Migrating from 2.x: see [MIGRATION.md](MIGRATIO waiter looked the owner's pid up in its own namespace and did not find it. The owner now records its PID namespace (the inode of `/proc/self/ns/pid`, in the reserved part of the header) and a waiter in another namespace no longer declares the owner dead from its pid. +- `StructuredMemory.AcquireWriteLock()` called while the thread holds a read guard set the writer flag and + waited for that thread's own read lock, blocking every other process until the timeout. It now throws + `InvalidOperationException` at once, like the automatic write lock and `SharedArray` do. +- `SingleProducerByteStream.Available` and `.Used` read the unmapped header after `Dispose()`; they throw + `ObjectDisposedException` like the other members. +- The timeout overloads of `ConcurrentQueue` and `ConcurrentMessageQueue` counted a failed enqueue or + dequeue on every poll (about 1,800 for 4 s of waiting). A call now counts once, when it gives up, and + not at all when it succeeds after waiting. They also no longer allocate a `Stopwatch` per call. - A waiting reader now recovers a write lock whose owner process died, like a waiting writer does. It used to wait for its whole timeout. - Disposing a region while another thread waits for one of its locks no longer crashes the process; diff --git a/InterprocessMemory.Tests/LibraryHardeningTests.cs b/InterprocessMemory.Tests/LibraryHardeningTests.cs index 523fe31..d3b4987 100644 --- a/InterprocessMemory.Tests/LibraryHardeningTests.cs +++ b/InterprocessMemory.Tests/LibraryHardeningTests.cs @@ -728,6 +728,78 @@ public void CreateOrOpen_WithMoreThanTheDevShmCanHold_ThrowsInsteadOfCrashingLat Assert.That(File.Exists("/dev/shm/" + name), Is.False, "the failed region must not leave a file behind"); } + [Test, Timeout(30000)] + public void StructuredMemory_ExplicitWriteLock_WhileHoldingOnlyAReadLock_ThrowsInsteadOfBlockingEveryone() + { + // The automatic write lock already refused this. The explicit one set the writer flag and then + // waited for this very thread's read lock to go away: every other reader and writer queued behind + // it until the timeout, or for good with an infinite timeout. + string name = N("ExplicitUpgrade"); + using var memory = StructuredMemory.CreateOrOpen(name, new SimpleSchema()); + + using (memory.AcquireReadLock()) + { + var sw = System.Diagnostics.Stopwatch.StartNew(); + Assert.Throws(() => memory.AcquireWriteLock(TimeSpan.FromSeconds(10))); + Assert.That(sw.ElapsedMilliseconds, Is.LessThan(2000), "it must fail at once, not wait for the timeout"); + } + + using (memory.AcquireWriteLock(TimeSpan.FromSeconds(1))) + { + } + } + + [Test] + public void SingleProducerByteStream_AvailableAndUsed_AfterDispose_Throw() + { + // Both properties read the unmapped header, which ends the process, instead of failing like the + // other members do. + var stream = SingleProducerByteStream.CreateOrOpen(N("SpscDisposedProps"), 1024); + stream.Dispose(); + + Assert.Throws(() => _ = stream.Available); + Assert.Throws(() => _ = stream.Used); + } + + [Test, Timeout(30000)] + public void ConcurrentQueue_AWaitingCall_CountsAsOneFailureNotOnePerPoll() + { + using var queue = ConcurrentQueue.CreateOrOpen(N("MpmcFailCount"), 4); + + Assert.That(queue.TryDequeue(out _, TimeSpan.FromMilliseconds(300)), Is.False); + Assert.That(queue.GetStatistics().FailedDequeues, Is.EqualTo(1)); + + for (int i = 0; i < 4; i++) + Assert.That(queue.TryEnqueue(i), Is.True); + Assert.That(queue.TryEnqueue(99, TimeSpan.FromMilliseconds(300)), Is.False); + Assert.That(queue.GetStatistics().FailedEnqueues, Is.EqualTo(1)); + + // A call that waits and then succeeds is not a failure. + for (int i = 0; i < 4; i++) + Assert.That(queue.TryDequeue(out _), Is.True); + Task waiter = Task.Run(() => queue.TryDequeue(out _, TimeSpan.FromSeconds(10))); + Thread.Sleep(100); + Assert.That(queue.TryEnqueue(7), Is.True); + Assert.That(waiter.Wait(TimeSpan.FromSeconds(10)), Is.True); + Assert.That(waiter.Result, Is.True); + Assert.That(queue.GetStatistics().FailedDequeues, Is.EqualTo(1)); + } + + [Test, Timeout(30000)] + public void ConcurrentMessageQueue_AWaitingCall_CountsAsOneFailureNotOnePerPoll() + { + using var queue = ConcurrentMessageQueue.CreateOrOpen(N("MqFailCount"), 2, 16); + var buffer = new byte[16]; + + Assert.That(queue.TryDequeue(buffer, out _, TimeSpan.FromMilliseconds(300)), Is.False); + Assert.That(queue.GetStatistics().FailedReads, Is.EqualTo(1)); + + Assert.That(queue.TryEnqueue(new byte[] { 1 }), Is.True); + Assert.That(queue.TryEnqueue(new byte[] { 2 }), Is.True); + Assert.That(queue.TryEnqueue(new byte[] { 3 }, TimeSpan.FromMilliseconds(300)), Is.False); + Assert.That(queue.GetStatistics().FailedWrites, Is.EqualTo(1)); + } + public struct BlobSchema : IMemorySchema { public const string Data = "Data"; diff --git a/InterprocessMemory/ConcurrentMessageQueue.cs b/InterprocessMemory/ConcurrentMessageQueue.cs index 4efb155..2ef9614 100644 --- a/InterprocessMemory/ConcurrentMessageQueue.cs +++ b/InterprocessMemory/ConcurrentMessageQueue.cs @@ -364,8 +364,12 @@ private void ValidateAndLoadBuffer(int? requestedCapacity, int? requestedMaxMess /// Lock-free operation safe for concurrent writers. /// /// True if write succeeded, false if buffer is full + public bool TryEnqueue(ReadOnlySpan data) => TryEnqueueCore(data, countFailure: true); + + // countFailure is false while a timeout overload polls: the call counts as one failed write when it + // gives up, not once per poll. [MethodImpl(MethodImplOptions.AggressiveOptimization)] - public bool TryEnqueue(ReadOnlySpan data) + private bool TryEnqueueCore(ReadOnlySpan data, bool countFailure) { ThrowIfDisposed(); @@ -407,7 +411,7 @@ public bool TryEnqueue(ReadOnlySpan data) else if (diff < 0) { // Buffer is full - if (_statsEnabled) + if (_statsEnabled && countFailure) Interlocked.Increment(ref _header->FailedWrites); return false; } @@ -439,7 +443,7 @@ public bool TryEnqueue(ReadOnlySpan data) /// left intact so the caller can retry with an adequately sized buffer. /// [MethodImpl(MethodImplOptions.AggressiveOptimization)] - private int TryDequeueCore(Span destination) + private int TryDequeueCore(Span destination, bool countFailure = true) { ThrowIfDisposed(); @@ -493,7 +497,7 @@ private int TryDequeueCore(Span destination) else if (diff < 0) { // Buffer is empty - if (_statsEnabled) + if (_statsEnabled && countFailure) Interlocked.Increment(ref _header->FailedReads); return 0; } @@ -536,13 +540,17 @@ public bool TryEnqueue(ReadOnlySpan data, TimeSpan timeout, ThrowIfDisposed(); TimeoutHelper.Validate(timeout, nameof(timeout)); - var sw = Stopwatch.StartNew(); + long start = Stopwatch.GetTimestamp(); var spinner = new SpinWait(); - while (!TryEnqueue(data)) + while (!TryEnqueueCore(data, countFailure: false)) { - if (cancellationToken.IsCancellationRequested || TimeoutHelper.HasExpired(sw, timeout)) + if (cancellationToken.IsCancellationRequested || TimeoutHelper.HasExpired(start, timeout)) + { + if (_statsEnabled) + Interlocked.Increment(ref _header->FailedWrites); return false; + } spinner.SpinOnce(); } @@ -567,14 +575,17 @@ public bool TryDequeue( ThrowIfDisposed(); TimeoutHelper.Validate(timeout, nameof(timeout)); - var sw = Stopwatch.StartNew(); + long start = Stopwatch.GetTimestamp(); var spinner = new SpinWait(); - bytesWritten = 0; - while (!TryDequeue(destination, out bytesWritten)) + while ((bytesWritten = TryDequeueCore(destination, countFailure: false)) == 0) { - if (cancellationToken.IsCancellationRequested || TimeoutHelper.HasExpired(sw, timeout)) + if (cancellationToken.IsCancellationRequested || TimeoutHelper.HasExpired(start, timeout)) + { + if (_statsEnabled) + Interlocked.Increment(ref _header->FailedReads); return false; + } spinner.SpinOnce(); } diff --git a/InterprocessMemory/ConcurrentQueue.cs b/InterprocessMemory/ConcurrentQueue.cs index de1b7c4..c829a7b 100644 --- a/InterprocessMemory/ConcurrentQueue.cs +++ b/InterprocessMemory/ConcurrentQueue.cs @@ -208,7 +208,11 @@ private void ValidateAndLoad(int? requestedCapacity) [MethodImpl(MethodImplOptions.AggressiveInlining)] private static byte* GetSlotData(ConcurrentQueueSlot* slot) => (byte*)slot + SlotHeaderSize; - public bool TryEnqueue(in T item) + public bool TryEnqueue(in T item) => TryEnqueueCore(in item, countFailure: true); + + // countFailure is false while a timeout overload polls: the call counts as one failed enqueue when + // it gives up, not once per poll (4 s of waiting used to add about 1,800 failed enqueues). + private bool TryEnqueueCore(in T item, bool countFailure) { ThrowIfDisposed(); for (int spin = 0; spin < _maxSpins; spin++) @@ -230,7 +234,7 @@ public bool TryEnqueue(in T item) } else if (difference < 0) { - if (_statisticsEnabled) + if (_statisticsEnabled && countFailure) Interlocked.Increment(ref _header->FailedEnqueues); return false; } @@ -240,7 +244,9 @@ public bool TryEnqueue(in T item) return false; } - public bool TryDequeue(out T item) + public bool TryDequeue(out T item) => TryDequeueCore(out item, countFailure: true); + + private bool TryDequeueCore(out T item, bool countFailure) { ThrowIfDisposed(); for (int spin = 0; spin < _maxSpins; spin++) @@ -262,7 +268,7 @@ public bool TryDequeue(out T item) } else if (difference < 0) { - if (_statisticsEnabled) + if (_statisticsEnabled && countFailure) Interlocked.Increment(ref _header->FailedDequeues); item = default; return false; @@ -281,12 +287,16 @@ public bool TryEnqueue( CancellationToken cancellationToken = default) { TimeoutHelper.Validate(timeout, nameof(timeout)); - var sw = Stopwatch.StartNew(); + long start = Stopwatch.GetTimestamp(); var spinner = new SpinWait(); - while (!TryEnqueue(in item)) + while (!TryEnqueueCore(in item, countFailure: false)) { - if (cancellationToken.IsCancellationRequested || TimeoutHelper.HasExpired(sw, timeout)) + if (cancellationToken.IsCancellationRequested || TimeoutHelper.HasExpired(start, timeout)) + { + if (_statisticsEnabled) + Interlocked.Increment(ref _header->FailedEnqueues); return false; + } spinner.SpinOnce(); } return true; @@ -298,12 +308,14 @@ public bool TryDequeue( CancellationToken cancellationToken = default) { TimeoutHelper.Validate(timeout, nameof(timeout)); - var sw = Stopwatch.StartNew(); + long start = Stopwatch.GetTimestamp(); var spinner = new SpinWait(); - while (!TryDequeue(out item)) + while (!TryDequeueCore(out item, countFailure: false)) { - if (cancellationToken.IsCancellationRequested || TimeoutHelper.HasExpired(sw, timeout)) + if (cancellationToken.IsCancellationRequested || TimeoutHelper.HasExpired(start, timeout)) { + if (_statisticsEnabled) + Interlocked.Increment(ref _header->FailedDequeues); item = default; return false; } diff --git a/InterprocessMemory/SingleProducerByteStream.cs b/InterprocessMemory/SingleProducerByteStream.cs index 8b86055..1ac2093 100644 --- a/InterprocessMemory/SingleProducerByteStream.cs +++ b/InterprocessMemory/SingleProducerByteStream.cs @@ -76,12 +76,26 @@ private struct Header /// /// Gets the available space in bytes for writing /// - public long Available => CalculateAvailable(); + public long Available + { + get + { + ThrowIfDisposed(); + return CalculateAvailable(); + } + } /// /// Gets the used space in bytes (data ready for reading) /// - public long Used => CalculateUsed(); + public long Used + { + get + { + ThrowIfDisposed(); + return CalculateUsed(); + } + } /// /// Gets performance statistics for the buffer diff --git a/InterprocessMemory/StructuredMemory.cs b/InterprocessMemory/StructuredMemory.cs index e39db98..bcd6cf9 100644 --- a/InterprocessMemory/StructuredMemory.cs +++ b/InterprocessMemory/StructuredMemory.cs @@ -752,6 +752,7 @@ public WriteLock AcquireWriteLock() /// Lock acquisition timeout /// A disposable lock guard that releases the lock on dispose /// Thrown when the lock cannot be acquired within the timeout + /// The calling thread holds only a read lock (upgrading would deadlock) /// /// The lock is owned by the calling thread. Dispose the guard on that same thread; do not /// await inside the guarded scope, or throws @@ -769,6 +770,13 @@ public WriteLock AcquireWriteLock(TimeSpan timeout) return new WriteLock(null, _decrementWriteLockDepth); } + // The write lock waits for every reader to leave, and this thread is one of them. Waiting would + // block all other processes (new readers and writers queue behind the pending writer) until the + // timeout, or for good with Timeout.InfiniteTimeSpan. + if (_readLockDepth.Value > 0) + throw new InvalidOperationException( + "Cannot take the write lock while holding only a read lock. Release the read lock first."); + if (!_buffer.TryAcquireWriteLock(timeout)) throw new TimeoutException($"Failed to acquire write lock within {timeout}"); From c5360c0007c87eb7ebe77676e545552785f282dd Mon Sep 17 00:00:00 2001 From: zacala Date: Tue, 6 Oct 2026 16:14:24 +0000 Subject: [PATCH 21/29] Make the cross-process test harness fail loudly and stop leaking children - WaitForExit(timeout) returns before the asynchronous output readers have delivered the last lines, so assertions on the child's output could see an empty string (2 of 550 runs in a loop). A second WaitForExit() flushes them, and a timeout now reports what the child had printed. - A missing worker DLL was Assert.Ignore, which would turn every cross-process test, the only ones that prove the lock recovery, into a silent skip on CI. It fails when CI is set. - ReadLine on a lock holder had no timeout, and NUnit's [Timeout] cannot abort a blocked thread on .NET Core. The wait is bounded and the failure includes the child's stderr. - hold_* children slept forever, so a parent that died (crash, hang timeout, Ctrl-C) left them holding the lock. They end when their stdin closes, with a two minute bound. - The test that races Dispose against busy-polling threads moved into a child process (dispose_busy_poll). When the grace period loses its race the AccessViolationException ended the whole test host and took every other result with it (8 of 12 runs on 2 CPUs under load). It is tagged TimingSensitive because it is probabilistic. --- InterprocessMemory.TestWorker/Program.cs | 77 ++++++++++++++++++- InterprocessMemory.Tests/CrossProcessTests.cs | 58 +++++++++++--- .../LibraryHardeningTests.cs | 42 ---------- 3 files changed, 122 insertions(+), 55 deletions(-) diff --git a/InterprocessMemory.TestWorker/Program.cs b/InterprocessMemory.TestWorker/Program.cs index 91d9614..31ad337 100644 --- a/InterprocessMemory.TestWorker/Program.cs +++ b/InterprocessMemory.TestWorker/Program.cs @@ -18,8 +18,11 @@ /// concurrent_producer <name> <producerId> — enqueue 1000 unique integers /// try_write_lock <name> — try the cross-process write lock for 250 ms /// orphan_write_lock <name> — acquire a write lock and exit without releasing it -/// hold_write_lock <name> — acquire a write lock, print "holding", and keep it until killed -/// hold_read_lock <name> — acquire a read lock, print "holding", and keep it until killed +/// dispose_busy_poll <prefix> — 200 rounds of: pollers spin on a ConcurrentQueue while it is disposed +/// hold_write_lock <name> — acquire a write lock, print "holding", and keep it until killed or until +/// the parent closes our standard input (the parent died) +/// hold_read_lock <name> — acquire a read lock, print "holding", and keep it until killed or until +/// the parent closes our standard input (the parent died) /// if (args.Length < 2) { @@ -45,6 +48,7 @@ args.Length >= 3 ? int.Parse(args[2]) : 0), "try_write_lock" => TryWriteLock(bufferName), "orphan_write_lock" => OrphanWriteLock(bufferName), + "dispose_busy_poll" => DisposeBusyPoll(bufferName), "hold_write_lock" => HoldWriteLock(bufferName), "hold_read_lock" => HoldReadLock(bufferName), _ => Error($"Unknown role: {role}") @@ -201,6 +205,56 @@ static int OrphanWriteLock(string name) return 0; // Deliberately skip Dispose/Release; process teardown closes only the mapping handle. } +// The common shutdown pattern: consumers spin on TryDequeue while another thread disposes the queue. Without +// MemoryRegion.DisposeGracePeriod this killed the process with an AccessViolationException within a few hundred +// rounds. It is a mitigation, not a proof: a thread that is descheduled for longer than the grace period at the +// wrong moment can still fail, and that kills this process, which is why it runs here and not in the test host. +static int DisposeBusyPoll(string namePrefix) +{ + var random = new Random(42); + int pollerCount = Math.Max(3, Environment.ProcessorCount - 1); + + for (int round = 0; round < 200; round++) + { + string name = $"{namePrefix}_{round}"; + var queue = InterprocessMemory.ConcurrentQueue.CreateOrOpen(name, 64); + var pollers = new Thread[pollerCount]; + for (int i = 0; i < pollers.Length; i++) + { + pollers[i] = new Thread(() => + { + try + { + while (true) + { + queue.TryDequeue(out _); + queue.TryEnqueue(1); + } + } + catch (ObjectDisposedException) + { + // Expected: the queue was disposed under the poller. + } + }) { IsBackground = true }; + pollers[i].Start(); + } + + Thread.Sleep(random.Next(0, 3)); + queue.Dispose(); + + foreach (Thread poller in pollers) + { + if (!poller.Join(TimeSpan.FromSeconds(10))) + return Error($"round {round}: a poller did not stop"); + } + + MemoryRegion.Remove(name); + } + + Console.WriteLine("ok"); + return 0; +} + static int HoldWriteLock(string name) { using var region = MemoryRegion.OpenExisting(name); @@ -211,7 +265,7 @@ static int HoldWriteLock(string name) // to simulate a crash while the lock is held. Console.WriteLine("holding"); Console.Out.Flush(); - Thread.Sleep(Timeout.Infinite); + WaitUntilTheParentIsGone(); return 0; } @@ -223,10 +277,25 @@ static int HoldReadLock(string name) Console.WriteLine("holding"); Console.Out.Flush(); - Thread.Sleep(Timeout.Infinite); + WaitUntilTheParentIsGone(); return 0; } +// The test kills us in the normal case. If it dies first (crash, hang timeout, Ctrl-C) the pipe that is our +// standard input closes, which ends this wait instead of holding the lock, and the process, for ever. +// The upper bound covers a parent that hangs without dying. +static void WaitUntilTheParentIsGone() +{ + try + { + System.Threading.Tasks.Task.Run(() => Console.In.ReadToEnd()).Wait(TimeSpan.FromMinutes(2)); + } + catch (Exception) + { + // Nothing to wait on (no stdin): exit. + } +} + static int Error(string msg) { Console.Error.WriteLine(msg); diff --git a/InterprocessMemory.Tests/CrossProcessTests.cs b/InterprocessMemory.Tests/CrossProcessTests.cs index 011e600..ad1cfe3 100644 --- a/InterprocessMemory.Tests/CrossProcessTests.cs +++ b/InterprocessMemory.Tests/CrossProcessTests.cs @@ -29,9 +29,19 @@ public class CrossProcessTests [OneTimeSetUp] public void EnsureHelperExists() { - if (!File.Exists(HelperDll)) - Assert.Ignore($"Test worker binary not found at '{HelperDll}'. " + - "Build the InterprocessMemory.TestWorker project first."); + if (File.Exists(HelperDll)) + return; + + string message = $"Test worker binary not found at '{HelperDll}'. " + + "Build the InterprocessMemory.TestWorker project first."; + + // Ignoring would turn every cross-process test, the only ones that prove the lock recovery, into a + // silent skip on CI when the worker is not copied next to the tests. + if (Environment.GetEnvironmentVariable("CI") == "true" || + Environment.GetEnvironmentVariable("GITHUB_ACTIONS") == "true") + Assert.Fail(message); + + Assert.Ignore(message); } private static string GetUniqueName(string prefix) => @@ -41,13 +51,16 @@ private static string GetUniqueName(string prefix) => private static ProcessStartInfo CreateHelperStartInfo( string role, string bufferName, - string? extraArgument = null) + string? extraArgument = null, + bool redirectInput = false) { + string? host = Environment.GetEnvironmentVariable("DOTNET_HOST_PATH"); var psi = new ProcessStartInfo { - FileName = "dotnet", + FileName = string.IsNullOrEmpty(host) ? "dotnet" : host, RedirectStandardOutput = true, RedirectStandardError = true, + RedirectStandardInput = redirectInput, UseShellExecute = false, CreateNoWindow = true }; @@ -79,9 +92,15 @@ private static ProcessStartInfo CreateHelperStartInfo( if (!proc.WaitForExit(timeoutMs)) { proc.Kill(); - Assert.Fail($"Child process [{role}] timed out after {timeoutMs} ms"); + proc.WaitForExit(); + Assert.Fail($"Child process [{role}] timed out after {timeoutMs} ms. " + + $"stdout: '{stdout.ToString().Trim()}' stderr: '{stderr.ToString().Trim()}'"); } + // WaitForExit(int) returns as soon as the process has exited, without waiting for the asynchronous + // readers to deliver the last lines. This overload does, so the assertions see all the output. + proc.WaitForExit(); + return (proc.ExitCode, stdout.ToString().Trim(), stderr.ToString().Trim()); } @@ -300,15 +319,21 @@ public void CrossProcess_OrphanWriteLock_IsRecovered() /// private static Process StartLockHolder(string bufferName, string role = "hold_write_lock") { - Process process = Process.Start(CreateHelperStartInfo(role, bufferName))!; - string? line = process.StandardOutput.ReadLine(); + // Standard input is redirected so that the child ends when this process disappears. + Process process = Process.Start(CreateHelperStartInfo(role, bufferName, redirectInput: true))!; + + // ReadLine has no timeout and NUnit's [Timeout] cannot abort a blocked thread on .NET Core. + Task reading = Task.Run(() => process.StandardOutput.ReadLine()); + string? line = reading.Wait(TimeSpan.FromSeconds(20)) ? reading.Result : ""; if (line != "holding") { try { process.Kill(); } catch (InvalidOperationException) { /* already exited */ } + process.WaitForExit(); + string error = process.StandardError.ReadToEnd().Trim(); process.Dispose(); - Assert.Fail($"lock holder did not report 'holding' (got '{line}')"); + Assert.Fail($"lock holder did not report 'holding' (got '{line}', stderr: '{error}')"); } return process; @@ -413,6 +438,21 @@ public void CrossProcess_FiniteWait_RecoversPromptlyWhenOwnerDies() } } + [Test, Timeout(150000)] + [Category("TimingSensitive")] + public void Dispose_WhileThreadsBusyPollTheQueue_DoesNotCrashTheProcess() + { + // Runs in a child process: when the mitigation (MemoryRegion.DisposeGracePeriod) loses its race, the + // uncatchable AccessViolationException ends the process that races. In the test host it also took + // every other test result with it. On a loaded 2 core machine that happened in about 2 of 3 runs, so + // this belongs with the timing sensitive tests, not in the step that has to be green. + string prefix = GetUniqueName("BusyPoll"); + var (exit, stdout, stderr) = SpawnHelper("dispose_busy_poll", prefix, timeoutMs: 120000); + + Assert.That(exit, Is.EqualTo(0), $"the child died or failed: {stderr}"); + Assert.That(stdout, Does.Contain("ok")); + } + [Test, Timeout(40000)] public void CrossProcess_ReadWait_RecoversWhenWriterProcessDies() { diff --git a/InterprocessMemory.Tests/LibraryHardeningTests.cs b/InterprocessMemory.Tests/LibraryHardeningTests.cs index d3b4987..7cddeaf 100644 --- a/InterprocessMemory.Tests/LibraryHardeningTests.cs +++ b/InterprocessMemory.Tests/LibraryHardeningTests.cs @@ -241,48 +241,6 @@ public void Dispose_KeepsTheMappingForTheGracePeriod() } } - [Test, Timeout(120000)] - public void Dispose_WhileThreadsBusyPollTheQueue_DoesNotCrash() - { - // The common shutdown pattern: consumers spin on TryDequeue while another thread disposes. - // Without DisposeGracePeriod this killed the process with an AccessViolationException within a - // few hundred rounds on a 4-core machine. This is a mitigation test, not a proof: a thread that - // is descheduled for longer than the grace period at the wrong moment can still fail. - var random = new Random(42); - int pollerCount = Math.Max(3, Environment.ProcessorCount - 1); - - for (int round = 0; round < 200; round++) - { - string name = N($"BusyPoll{round}"); - var queue = InterprocessMemory.ConcurrentQueue.CreateOrOpen(name, 64); - var pollers = new Task[pollerCount]; - for (int i = 0; i < pollers.Length; i++) - { - pollers[i] = Task.Run(() => - { - try - { - while (true) - { - queue.TryDequeue(out _); - queue.TryEnqueue(1); - } - } - catch (ObjectDisposedException) - { - // Expected: the queue was disposed under the poller. - } - }); - } - - Thread.Sleep(random.Next(0, 3)); - queue.Dispose(); - - Assert.That(Task.WaitAll(pollers, TimeSpan.FromSeconds(10)), Is.True, $"round {round}"); - MemoryRegion.Remove(name); - } - } - private static int CountOpenFilesMatching(string name) { return Directory.GetFiles("/proc/self/fd").Count(path => From e1dfbdddaafbd29aed892a12bbc6a446610a08ec Mon Sep 17 00:00:00 2001 From: zacala Date: Tue, 6 Oct 2026 16:14:24 +0000 Subject: [PATCH 22/29] Run spinning test loops on dedicated threads Task.Run puts every spinning loop on a thread-pool worker. The pool starts with one worker per core and adds one about every half second, so on a small machine loops queued behind the first few start late or after the test's own cancellation timer, and the result depended on test order. Alone on 2 CPUs: Fairness_WriterUnderModerateReaders had a writer that never ran, the 16-thread Barrier test took 10 to 13 s, and the MPMC fairness test saw its consumer never run (producer counts [256, 0, 0, ...]). DedicatedThread.Run uses TaskCreationOptions.LongRunning. All of them pass alone on 2 CPUs now. The MPMC fairness ratio itself is a scheduling effect on a machine with fewer cores than threads (thousands to one on 2 CPUs), so it stays TimingSensitive with the reason written next to it. --- .../ChangeVerificationTests.cs | 13 +++++++--- .../ConcurrencyStabilityTests.cs | 24 +++++++++++-------- InterprocessMemory.Tests/DedicatedThread.cs | 17 +++++++++++++ README.md | 5 ++-- 4 files changed, 44 insertions(+), 15 deletions(-) create mode 100644 InterprocessMemory.Tests/DedicatedThread.cs diff --git a/InterprocessMemory.Tests/ChangeVerificationTests.cs b/InterprocessMemory.Tests/ChangeVerificationTests.cs index 222eddc..345a20f 100644 --- a/InterprocessMemory.Tests/ChangeVerificationTests.cs +++ b/InterprocessMemory.Tests/ChangeVerificationTests.cs @@ -542,7 +542,7 @@ public void Mpmc_MaxSpins_IntMaxValue_DoesNotThrow() // ── #8 InitializeOrOpen race-safe two-phase magic ─────────────────────── - [Test] + [Test, Timeout(60000)] public void InitializeOrOpen_ConcurrentSameProcessOpen_NoTornCapacityRead() { // Spawn N threads that all try to open the same buffer simultaneously. Exactly one @@ -557,7 +557,9 @@ public void InitializeOrOpen_ConcurrentSameProcessOpen_NoTornCapacityRead() var errors = new System.Collections.Concurrent.ConcurrentBag(); var buffers = new MemoryRegion?[Threads]; - Parallel.For(0, Threads, i => + // Dedicated threads: Parallel.For would run the sixteen barrier waits on thread-pool workers, and the + // pool adds them one at a time, so on a two core machine this took 10 to 13 seconds. + var threads = Enumerable.Range(0, Threads).Select(i => new Thread(() => { try { @@ -568,7 +570,12 @@ public void InitializeOrOpen_ConcurrentSameProcessOpen_NoTornCapacityRead() { errors.Add(ex); } - }); + })).ToArray(); + + foreach (Thread thread in threads) + thread.Start(); + foreach (Thread thread in threads) + Assert.That(thread.Join(TimeSpan.FromSeconds(30)), Is.True, "a thread did not finish opening the buffer"); try { diff --git a/InterprocessMemory.Tests/ConcurrencyStabilityTests.cs b/InterprocessMemory.Tests/ConcurrencyStabilityTests.cs index fe4e11b..0caba5c 100644 --- a/InterprocessMemory.Tests/ConcurrencyStabilityTests.cs +++ b/InterprocessMemory.Tests/ConcurrencyStabilityTests.cs @@ -46,7 +46,7 @@ public async Task Opt7_StatsDisabled_HighContentionReaders_NoCorruption() var errors = new ConcurrentBag(); var cts = new CancellationTokenSource(durationMs); - var writers = Enumerable.Range(0, writerCount).Select(w => Task.Run(() => + var writers = Enumerable.Range(0, writerCount).Select(w => DedicatedThread.Run(() => { Span stamp = stackalloc byte[4]; BitConverter.TryWriteBytes(stamp, writerMagics[w]); @@ -61,7 +61,7 @@ public async Task Opt7_StatsDisabled_HighContentionReaders_NoCorruption() } })).ToArray(); - var readers = Enumerable.Range(0, readerCount).Select(r => Task.Run(() => + var readers = Enumerable.Range(0, readerCount).Select(r => DedicatedThread.Run(() => { Span rd = stackalloc byte[4]; while (!cts.Token.IsCancellationRequested) @@ -247,7 +247,7 @@ public async Task StabilityA_OrphanCheckUnderLegitimateLockUse_NoFalsePositive() var checkCount = 0L; var holdCount = 0L; - var holder = Task.Run(() => + var holder = DedicatedThread.Run(() => { while (!cts.Token.IsCancellationRequested) { @@ -264,7 +264,7 @@ public async Task StabilityA_OrphanCheckUnderLegitimateLockUse_NoFalsePositive() } }); - var checkers = Enumerable.Range(0, 8).Select(_ => Task.Run(() => + var checkers = Enumerable.Range(0, 8).Select(_ => DedicatedThread.Run(() => { while (!cts.Token.IsCancellationRequested) { @@ -447,7 +447,7 @@ public async Task Fairness_WriterUnderModerateReaders_MakesProgress() var maxWriterWaitMs = 0L; var writerAcquires = 0; - var readers = Enumerable.Range(0, 8).Select(_ => Task.Run(() => + var readers = Enumerable.Range(0, 8).Select(_ => DedicatedThread.Run(() => { while (!cts.Token.IsCancellationRequested) { @@ -460,7 +460,7 @@ public async Task Fairness_WriterUnderModerateReaders_MakesProgress() } })).ToArray(); - var writer = Task.Run(() => + var writer = DedicatedThread.Run(() => { while (!cts.Token.IsCancellationRequested) { @@ -500,8 +500,12 @@ public async Task Fairness_WriterUnderModerateReaders_MakesProgress() // ── MPMC producer fairness ─────────────────────────────────────────────── - // The ratio depends on the core count and the scheduler (eight spinning producers share the - // thread pool), so CI runs this separately and does not fail the build on it. + // The ratio depends on the core count and the scheduler. Nine spinning threads (eight producers and the + // consumer) on a machine with fewer cores let the producer that is running win the next slot again and + // again while the others wait for a time slice: on 2 cores a ratio in the thousands was measured. The + // threads are dedicated, so a failure is that scheduling effect and not thread-pool starvation (which + // made the consumer never run, and every producer but one write nothing). CI runs this separately and + // reports a failure as a warning instead of failing the build. [Test] [Category("TimingSensitive")] [Timeout(30000)] @@ -518,7 +522,7 @@ public async Task Fairness_Mpmc_ProducersGetReasonableShare() var counts = new long[producerCount]; var cts = new CancellationTokenSource(durationMs); - var producers = Enumerable.Range(0, producerCount).Select(p => Task.Run(() => + var producers = Enumerable.Range(0, producerCount).Select(p => DedicatedThread.Run(() => { Span data = stackalloc byte[16]; BitConverter.TryWriteBytes(data, p); @@ -530,7 +534,7 @@ public async Task Fairness_Mpmc_ProducersGetReasonableShare() })).ToArray(); // Single consumer drains continuously so producers don't all jam on full-buffer. - var consumer = Task.Run(() => + var consumer = DedicatedThread.Run(() => { Span rd = stackalloc byte[128]; while (!cts.Token.IsCancellationRequested) diff --git a/InterprocessMemory.Tests/DedicatedThread.cs b/InterprocessMemory.Tests/DedicatedThread.cs new file mode 100644 index 0000000..6c21abc --- /dev/null +++ b/InterprocessMemory.Tests/DedicatedThread.cs @@ -0,0 +1,17 @@ +namespace InterprocessMemory.Tests; + +/// +/// Runs a loop that spins until it is cancelled on a thread of its own. +/// +/// With Task.Run such loops occupy thread-pool workers. The pool starts with one worker per +/// core and adds one every half second or so, so on a small machine (a two core CI runner) the loops queued +/// behind the first few start late, or never before the test's own cancellation timer, which also needs a +/// pool thread, has fired. The test then fails with a writer that never ran, a consumer that saw nothing, or +/// its timeout, depending on the order in which the work items were queued, and the result differs between +/// running the test alone and running the whole suite. +/// +internal static class DedicatedThread +{ + public static Task Run(Action action) => + Task.Factory.StartNew(action, CancellationToken.None, TaskCreationOptions.LongRunning, TaskScheduler.Default); +} diff --git a/README.md b/README.md index 1dcf540..c770ba0 100644 --- a/README.md +++ b/README.md @@ -336,8 +336,9 @@ cross-process lock exclusion, and orphan-lock recovery. request and every push to `main`. Two groups of tests are kept out of the blocking test step (use `--filter` to select or exclude them locally): -- `Category=TimingSensitive`: assertions that depend on thread scheduling, such as a fairness ratio. - CI runs them in a separate, non-blocking step because a busy shared runner can fail them. +- `Category=TimingSensitive`: assertions that depend on the core count and thread scheduling, such as a + fairness ratio, and the test that races `Dispose` against busy-polling threads in a child process. + CI runs them in a separate, non-blocking step and shows a failure as a warning on the run. - `Category=LongRunning`: the `[Explicit]` soak tests, which NUnit never runs unless asked to. CI does not run them. From 0805e033f74fe51f499c7ea412f2300a2ce4963b Mon Sep 17 00:00:00 2001 From: zacala Date: Tue, 6 Oct 2026 16:14:25 +0000 Subject: [PATCH 23/29] Show failures of the informational CI step and keep main runs continue-on-error reports a failed step as a success, so the timing sensitive step had been failing on every run while the PR showed green. A following step now adds a warning annotation and a section to the run summary when it failed. The step is skipped when the build failed. cancel-in-progress applies to pull requests only: consecutive pushes to main used to cancel the runs of the commits in between. --- .github/workflows/ci.yml | 27 +++++++++++++++++++++------ 1 file changed, 21 insertions(+), 6 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 30434d5..831388b 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -9,10 +9,11 @@ on: permissions: contents: read -# A newer push to the same branch or pull request supersedes the run in progress. +# A newer push to the same pull request supersedes the run in progress. Runs for main are never cancelled, +# so that every commit that lands there keeps its own result. concurrency: group: ci-${{ github.workflow }}-${{ github.ref }} - cancel-in-progress: true + cancel-in-progress: ${{ github.event_name == 'pull_request' }} env: DOTNET_NOLOGO: true @@ -58,11 +59,12 @@ jobs: # Builds every project, including the benchmarks and the child-process test worker that the # cross-process tests start. Packing on every build is not needed here. - name: Build + id: build run: dotnet build InterprocessMemory.sln --configuration Release --no-restore -p:GeneratePackageOnBuild=false - # --blame names the running test when the test host dies, which is what an - # AccessViolationException from a memory-mapped region looks like; the hang timeout stops a - # stuck test from holding the runner until the job timeout. + # The blame collector (enabled by --blame-hang-timeout) names the running test when the test host + # dies, which is what an AccessViolationException from a memory-mapped region looks like; the hang + # timeout stops a stuck test from holding the runner until the job timeout. - name: Test run: > dotnet test InterprocessMemory.Tests/InterprocessMemory.Tests.csproj @@ -77,7 +79,8 @@ jobs: # Tests whose outcome depends on the core count and the scheduler. They still run and show up # in the log and the artifact, but a failure here does not fail the job. - name: Test (timing sensitive, informational) - if: ${{ !cancelled() }} + id: timing + if: ${{ !cancelled() && steps.build.conclusion == 'success' }} continue-on-error: true run: > dotnet test InterprocessMemory.Tests/InterprocessMemory.Tests.csproj @@ -87,6 +90,18 @@ jobs: --logger "trx;LogFileName=timing-sensitive-results.trx" --results-directory TestResults + # continue-on-error shows a failed step as a success, so say it out loud: a warning annotation on the + # run and a section in its summary. Otherwise a test that fails on every run looks green for ever. + - name: Report timing sensitive failures + if: ${{ !cancelled() && steps.timing.outcome == 'failure' }} + run: | + echo "::warning title=Timing sensitive tests failed on ${{ matrix.os }}::This step does not fail the job. See the step log and timing-sensitive-results.trx in the test-results artifact." + { + echo "### Timing sensitive tests failed on ${{ matrix.os }}" + echo "" + echo "These tests depend on scheduling and do not fail the job, but a failure is still a result to read: see the log of the step above and the test-results artifact." + } >> "$GITHUB_STEP_SUMMARY" + - name: Upload test results if: always() uses: actions/upload-artifact@v4 From ad9583f0266f363e5eac4bad2502a81de5bece4b Mon Sep 17 00:00:00 2001 From: zacala Date: Fri, 9 Oct 2026 04:55:49 +0000 Subject: [PATCH 24/29] Fix region creation races and two lifetime defects - OpenExisting that arrived between the creator making the backing file and writing its header reported an empty file or a zero magic number as corrupt (13 of 300 races). It waits up to two seconds for a region that is still being created. - Linux: both processes of a creation race saw an empty file and took the creator's role; the one that failed later (141 of 300 races with two capacities) unlinked the file the other used, splitting one name into two regions. FileMode.CreateNew decides who creates and sizes the file, and only that process unlinks it on failure. - FilePath mode let MemoryMappedFile.CreateFromFile grow an existing file to the requested capacity before anything checked it, so a wrong capacity left the file permanently resized. The file is opened here with sharing, a size mismatch is rejected without touching it, and an older format or foreign file is reported as that instead of as a size problem. - GetMemory returned a Memory whose manager did not reference the region, so dropping the region let the finalizer unmap it under the memory. The manager now holds its owner. - Linux: a killed lock owner that its parent had not reaped is a zombie; GetProcessById and HasExited call it alive, so its lock was never recovered. /proc//stat state Z or X now counts as gone. Each new test fails on the previous code and passes now. --- CHANGELOG.md | 15 ++ InterprocessMemory.Tests/CrossProcessTests.cs | 53 +++++ .../LibraryHardeningTests.cs | 209 +++++++++++++++++ InterprocessMemory/MemoryRegion.cs | 216 ++++++++++++++---- InterprocessMemory/UnmanagedMemoryManager.cs | 11 +- 5 files changed, 462 insertions(+), 42 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index a4e4cbc..50c293e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -41,6 +41,21 @@ Changes since the 3.0.0 release. Migrating from 2.x: see [MIGRATION.md](MIGRATIO - The timeout overloads of `ConcurrentQueue` and `ConcurrentMessageQueue` counted a failed enqueue or dequeue on every poll (about 1,800 for 4 s of waiting). A call now counts once, when it gives up, and not at all when it succeeds after waiting. They also no longer allocate a `Stopwatch` per call. +- `OpenExisting` racing the process that creates the region reported "invalid header" or "empty" about once in + eleven races instead of waiting the few microseconds until the creator had written the header. It waits up + to two seconds for a region that is still being created. +- Linux: two processes calling `CreateOrOpen` for a new name at the same moment both became the creator, and + the one that then failed (another capacity or region kind) deleted the file the other was using, so later + openers got a second, separate region. `FileMode.CreateNew` decides who creates and sizes the file, and + only that process removes it when construction fails. +- `MemoryRegionOptions.FilePath`: an existing file of another size was grown to the requested capacity before + anything checked it, which left it permanently resized and could never be undone. It is now rejected + without being modified (an older format or a foreign file is reported as such), and the file is opened + with sharing so that a second process can map it. +- `MemoryRegion.GetMemory` returned a `Memory` that did not keep the region reachable; a caller that + dropped the region and kept the memory let the finalizer unmap it, which ended the process. +- Linux: a lock owner that had been killed but not yet reaped by its parent (a zombie) counted as alive. It is + recognised as gone now. - A waiting reader now recovers a write lock whose owner process died, like a waiting writer does. It used to wait for its whole timeout. - Disposing a region while another thread waits for one of its locks no longer crashes the process; diff --git a/InterprocessMemory.Tests/CrossProcessTests.cs b/InterprocessMemory.Tests/CrossProcessTests.cs index ad1cfe3..05b44af 100644 --- a/InterprocessMemory.Tests/CrossProcessTests.cs +++ b/InterprocessMemory.Tests/CrossProcessTests.cs @@ -626,6 +626,59 @@ public void CrossProcess_LiveOwner_RecordsItsPidNamespaceAndClearsItOnRelease() Assert.That(view.ReadInt64(88), Is.EqualTo(0L), "a stale namespace would mislead the next owner's waiters"); } + [Test, Timeout(40000)] + public void CrossProcess_OwnerThatIsAZombie_IsRecovered() + { + // A killed process whose parent has not reaped it still exists for GetProcessById and HasExited, so + // it looked alive. This happens in a container whose init does not reap, or under a busy supervisor. + if (!OperatingSystem.IsLinux() || !File.Exists("/bin/sh")) + Assert.Ignore("Needs Linux /proc and /bin/sh to make a zombie."); + + // The shell starts a background sleep, prints its pid and replaces itself with another sleep, which + // never waits for children: once the background sleep is killed it stays a zombie. + var psi = new ProcessStartInfo("/bin/sh") + { + RedirectStandardOutput = true, + UseShellExecute = false, + CreateNoWindow = true + }; + psi.ArgumentList.Add("-c"); + psi.ArgumentList.Add("sleep 600 & echo $!; exec sleep 600"); + + using Process parent = Process.Start(psi)!; + try + { + int zombiePid = int.Parse(parent.StandardOutput.ReadLine()!); + using (Process child = Process.GetProcessById(zombiePid)) + child.Kill(); + + bool isZombie = false; + for (int i = 0; i < 100 && !isZombie; i++) + { + try + { isZombie = File.ReadAllText($"/proc/{zombiePid}/stat").Contains(") Z"); } + catch (IOException) { break; } + if (!isZombie) + Thread.Sleep(50); + } + + if (!isZombie) + Assert.Ignore("The killed process was reaped at once here, so there is no zombie to test with."); + + string name = GetUniqueName("Zombie"); + using var region = MemoryRegion.CreateOrOpen(name, 256); + HoldLockAs(name, zombiePid, MemoryRegion.CurrentPidNamespace); + + Assert.That(region.IsWriteLockOrphaned(), Is.True, "a zombie holds nothing"); + Assert.That(region.TryAcquireWriteLock(TimeSpan.FromSeconds(5)), Is.True); + region.ReleaseWriteLock(); + } + finally + { + KillAndWait(parent); + } + } + [Test, Timeout(30000)] public void CrossProcess_DeadReaderProcess_NeedsForceResetLocks() { diff --git a/InterprocessMemory.Tests/LibraryHardeningTests.cs b/InterprocessMemory.Tests/LibraryHardeningTests.cs index 7cddeaf..47a19e0 100644 --- a/InterprocessMemory.Tests/LibraryHardeningTests.cs +++ b/InterprocessMemory.Tests/LibraryHardeningTests.cs @@ -758,6 +758,215 @@ public void ConcurrentMessageQueue_AWaitingCall_CountsAsOneFailureNotOnePerPoll( Assert.That(queue.GetStatistics().FailedWrites, Is.EqualTo(1)); } + [Test, Timeout(120000)] + public void OpenExisting_RacingTheCreator_WaitsInsteadOfReportingACorruptHeader() + { + // The creator makes the backing file, sizes it and writes the header one step after another. + // An opener that arrived in between found an empty file or a zero magic number and reported + // "invalid header" (about 9% of 300 races) instead of waiting the few microseconds. + var failures = new System.Collections.Concurrent.ConcurrentBag(); + + for (int round = 0; round < 300; round++) + { + string name = N($"OpenRace{round}"); + using var go = new ManualResetEventSlim(false); + MemoryRegion? created = null; + + var creator = new Thread(() => + { + go.Wait(); + created = MemoryRegion.CreateOrOpen(name, 256); + }); + var opener = new Thread(() => + { + go.Wait(); + var sw = System.Diagnostics.Stopwatch.StartNew(); + while (sw.Elapsed < TimeSpan.FromSeconds(5)) + { + try + { + using var region = MemoryRegion.OpenExisting(name); + return; + } + catch (FileNotFoundException) + { + // Not created yet: the only error a caller is expected to retry. + } + catch (Exception ex) + { + failures.Add($"round {round}: {ex.GetType().Name}: {ex.Message}"); + return; + } + } + }); + + creator.Start(); + opener.Start(); + go.Set(); + Assert.That(creator.Join(TimeSpan.FromSeconds(30)), Is.True); + Assert.That(opener.Join(TimeSpan.FromSeconds(30)), Is.True); + created?.Dispose(); + MemoryRegion.Remove(name); + } + + Assert.That(failures, Is.Empty, $"{failures.Count} of 300 races failed, e.g. {failures.FirstOrDefault()}"); + } + + [Test, Timeout(120000)] + public void CreateOrOpen_LosingTheCreationRace_DoesNotDeleteTheWinnersFile() + { + if (!OperatingSystem.IsLinux()) + Assert.Ignore("The race is about the file in /dev/shm."); + + // Both processes saw an empty file, both took the creator's role, and the one whose capacity did + // not match then deleted the file that the other was using: later openers got a second, separate + // region under the same name (981 of 3000 races). + int deleted = 0; + int unexpected = 0; + + for (int round = 0; round < 300; round++) + { + string name = N($"CreateRace{round}"); + using var go = new Barrier(2); + MemoryRegion?[] regions = new MemoryRegion?[2]; + Exception?[] errors = new Exception?[2]; + + Thread Start(int index, long capacity) => new Thread(() => + { + try + { + go.SignalAndWait(); + regions[index] = MemoryRegion.CreateOrOpen(name, capacity); + } + catch (Exception ex) + { + errors[index] = ex; + } + }); + + Thread a = Start(0, 1024); + Thread b = Start(1, 2048); + a.Start(); + b.Start(); + Assert.That(a.Join(TimeSpan.FromSeconds(30)), Is.True); + Assert.That(b.Join(TimeSpan.FromSeconds(30)), Is.True); + + int winners = regions.Count(r => r != null); + if (winners != 1 || errors.Count(e => e is InvalidOperationException) != 1) + unexpected++; + else if (!File.Exists("/dev/shm/" + name)) + deleted++; + + foreach (var region in regions) + region?.Dispose(); + MemoryRegion.Remove(name); + } + + Assert.That(unexpected, Is.EqualTo(0), "exactly one creator must win and the other must report the size mismatch"); + Assert.That(deleted, Is.EqualTo(0), "the loser must not unlink the winner's file"); + } + + [Test] + public void FilePath_ExistingFileOfAnotherSize_IsRejectedWithoutChangingIt() + { + string path = Path.Combine(Path.GetTempPath(), $"ipm_resize_{Guid.NewGuid():N}.bin"); + string name = N("FileResize"); + try + { + using (MemoryRegion.CreateOrOpen(name, 1000, new MemoryRegionOptions { FilePath = path })) + { + } + + long before = new FileInfo(path).Length; + + Assert.Throws(() => + MemoryRegion.CreateOrOpen(name, 2000, new MemoryRegionOptions { FilePath = path })); + Assert.That(new FileInfo(path).Length, Is.EqualTo(before), + "the file used to be grown to the requested size before anything was checked"); + + using var again = MemoryRegion.CreateOrOpen(name, 1000, new MemoryRegionOptions { FilePath = path }); + Assert.That(again.Capacity, Is.EqualTo(1000)); + } + finally + { + try { File.Delete(path); } catch (IOException) { } + } + } + + [Test] + public void FilePath_ExistingVersion2File_IsReportedAsOldFormatAndNotResized() + { + string path = Path.Combine(Path.GetTempPath(), $"ipm_v2size_{Guid.NewGuid():N}.bin"); + try + { + byte[] bytes = new byte[128 + 4096]; + BitConverter.TryWriteBytes(bytes.AsSpan(0), 0x48504D53u); + BitConverter.TryWriteBytes(bytes.AsSpan(4), 2u); + File.WriteAllBytes(path, bytes); + + var error = Assert.Throws(() => + MemoryRegion.CreateOrOpen(N("V2Size"), 8192, new MemoryRegionOptions { FilePath = path })); + + Assert.That(error!.Message, Does.Contain("2.x")); + Assert.That(File.ReadAllBytes(path), Is.EqualTo(bytes), "the old file must not be modified"); + } + finally + { + try { File.Delete(path); } catch (IOException) { } + } + } + + [Test] + public void FilePath_TwoInstancesOfTheSameFile_ShareTheMemory() + { + if (OperatingSystem.IsWindows()) + Assert.Ignore("Windows names the mapping after the region; a second named mapping of one region is a separate case."); + + string path = Path.Combine(Path.GetTempPath(), $"ipm_shared_{Guid.NewGuid():N}.bin"); + string name = N("FileShared"); + var options = new MemoryRegionOptions { FilePath = path }; + try + { + using var first = MemoryRegion.CreateOrOpen(name, 256, options); + using var second = MemoryRegion.OpenExisting(name, options); + + first.Write(new byte[] { 7, 8, 9 }, 0); + var read = new byte[3]; + second.Read(read, 0); + Assert.That(read, Is.EqualTo(new byte[] { 7, 8, 9 })); + } + finally + { + try { File.Delete(path); } catch (IOException) { } + } + } + + [System.Runtime.CompilerServices.MethodImpl(System.Runtime.CompilerServices.MethodImplOptions.NoInlining)] + private static (Memory Memory, WeakReference Region) GetMemoryAndForgetTheRegion(string name) + { + var region = MemoryRegion.CreateOrOpen(name, 256); + return (region.GetMemory(0, 64), new WeakReference(region)); + } + + [Test, Timeout(30000)] + public void GetMemory_KeepsTheRegionAliveForAsLongAsTheMemoryIsUsed() + { + // The Memory wraps a raw pointer into the mapping. Nothing tied it to the MemoryRegion, so a + // caller that dropped the region and kept the Memory let the finalizer unmap the view under it. + var (memory, region) = GetMemoryAndForgetTheRegion(N("Rooted")); + + for (int i = 0; i < 3; i++) + { + GC.Collect(); + GC.WaitForPendingFinalizers(); + } + + Assert.That(region.IsAlive, Is.True, "the region was finalized while its Memory was still in use"); + memory.Span[0] = 42; + Assert.That(memory.Span[0], Is.EqualTo((byte)42)); + GC.KeepAlive(memory); + } + public struct BlobSchema : IMemorySchema { public const string Data = "Data"; diff --git a/InterprocessMemory/MemoryRegion.cs b/InterprocessMemory/MemoryRegion.cs index d382e17..42b04c1 100644 --- a/InterprocessMemory/MemoryRegion.cs +++ b/InterprocessMemory/MemoryRegion.cs @@ -78,6 +78,12 @@ public sealed unsafe class MemoryRegion : IMemoryRegion // How often a lock waiter re-probes whether the owning process is still alive. private const long OrphanCheckIntervalMs = 250; + // OpenExisting (and a creator that lost the race to create the file) can arrive between the creator + // making the backing file and its first header write. The file is created empty, sized right after, and + // its header is written right after that: a few microseconds apart, so a short wait is enough, and a + // file that is still empty after it has no live creator. + private const int OpenerWaitMs = 2000; + // A held write lock with no owner recorded is normal for a few nanoseconds: the owner sets // WriterLockState first and writes its identity right after, and clears the identity before it // clears the state. A process killed in one of those two windows leaves the lock held with nobody @@ -204,6 +210,25 @@ private static long TryReadLinuxStartTicks(int pid) } } + /// + /// True when /proc/<pid>/stat reports the process as a zombie (Z) or dead (X). + /// + private static bool IsLinuxZombie(int pid) + { + try + { + string stat = File.ReadAllText("/proc/" + pid + "/stat"); + + // The state is the first field after the parenthesised command name (see TryReadLinuxStartTicks). + int commEnd = stat.LastIndexOf(')'); + return commEnd >= 0 && commEnd + 2 < stat.Length && stat[commEnd + 2] is 'Z' or 'X'; + } + catch + { + return false; + } + } + /// /// True when the process currently using is provably not the one /// that recorded (PID reuse). @@ -482,22 +507,70 @@ private void Initialize() /// private void CreateMmfFromExplicitFilePath(long totalSize) { - if (_createOrOpen) + string fullPath = Path.GetFullPath(_options.FilePath!); + string? mapName = OperatingSystem.IsWindows() ? _name : null; + + // The file is opened here, with sharing, instead of by MemoryMappedFile.CreateFromFile(path, ...): + // that overload grows an existing file to the requested capacity before anything has checked that + // it is the region the caller meant, which left a file of the wrong size permanently resized, and + // it does not share the file with a second process on Windows. + var file = new FileStream(fullPath, _createOrOpen ? FileMode.OpenOrCreate : FileMode.Open, + FileAccess.ReadWrite, FileShare.ReadWrite); + try { - string fullPath = Path.GetFullPath(_options.FilePath!); - long existing = File.Exists(fullPath) ? new FileInfo(fullPath).Length : 0; - if (existing < totalSize) - EnsureFreeSpace(Path.GetDirectoryName(fullPath) ?? fullPath, totalSize - existing, _name); + if (_createOrOpen) + { + long length = file.Length; + if (length == 0) + { + EnsureFreeSpace(Path.GetDirectoryName(fullPath) ?? fullPath, totalSize, _name); + file.SetLength(totalSize); + } + else if (length != totalSize) + { + ThrowForExistingFileOfAnotherSize(file, fullPath, length, totalSize); + } + } + + _mmf = MemoryMappedFile.CreateFromFile( + file, + mapName, + _createOrOpen ? totalSize : 0, + MemoryMappedFileAccess.ReadWrite, + HandleInheritability.None, + leaveOpen: false); // the MMF takes ownership of the FileStream } + catch + { + file.Dispose(); + throw; + } + } - string? mapName = OperatingSystem.IsWindows() ? _name : null; - FileMode mode = _createOrOpen ? FileMode.OpenOrCreate : FileMode.Open; - _mmf = MemoryMappedFile.CreateFromFile( - _options.FilePath!, - mode, - mapName, - _createOrOpen ? totalSize : 0, - MemoryMappedFileAccess.ReadWrite); + /// + /// An existing file of another size is not this region. A file that clearly is not a version 3 region + /// (an older format, or something else entirely) is reported as that, as opening it without a + /// requested capacity does; otherwise it is a region with another capacity. The file is not modified. + /// + private void ThrowForExistingFileOfAnotherSize(FileStream file, string path, long actualLength, long totalSize) + { + if (actualLength >= sizeof(uint)) + { + Span magic = stackalloc byte[sizeof(uint)]; + file.Position = 0; + if (file.Read(magic) == magic.Length) + { + uint observed = BitConverter.ToUInt32(magic); + if (observed != 0 && + observed != SharedHeader.MagicNumber && + observed != SharedHeader.MagicInitializing) + ThrowForUnexpectedMagic(observed); + } + } + + throw new InvalidOperationException( + $"Existing shared memory '{path}' has size {actualLength} but {totalSize} was requested. " + + "Either match the existing size or remove the file."); } /// @@ -535,40 +608,71 @@ private void CreateMmfFromWindowsNamedRegion(long totalSize) private void CreateMmfFromLinuxDevShm(long totalSize) { _backingFilePath = "/dev/shm/" + _name; - FileMode mode = _createOrOpen ? FileMode.OpenOrCreate : FileMode.Open; - _backingFile = new FileStream(_backingFilePath, - mode, FileAccess.ReadWrite, FileShare.ReadWrite); - if (_backingFile.Length == 0) + // FileMode.CreateNew lets the kernel decide who creates the file. Two processes that both saw an + // empty file used to take the creator's role, and the one that then failed (another capacity, + // another region kind) deleted the file the other was already using, which split one name into + // two independent regions. Only the process that created the file may size it, and only it + // unlinks the file when construction fails. + for (int attempt = 0; _backingFile == null; attempt++) { - if (!_createOrOpen) + if (_createOrOpen) { - _backingFile.Dispose(); - _backingFile = null; - throw new InvalidDataException( - $"Existing shared memory '{_backingFilePath}' is empty and has not been initialized."); + try + { + _backingFile = new FileStream(_backingFilePath, + FileMode.CreateNew, FileAccess.ReadWrite, FileShare.ReadWrite); + _createdBackingFile = true; + break; + } + catch (IOException) when (File.Exists(_backingFilePath)) + { + // Somebody else created it first: open it below. + } } - // Fresh region — set the size up front. Subsequent openers will see this length - // and skip the SetLength call below. Mark that WE were the process to size this - // file: if construction fails between here and AcquirePointer, Cleanup will - // unlink the file so the next caller doesn't trip over a half-initialized blob. - _createdBackingFile = true; + try + { + _backingFile = new FileStream(_backingFilePath, + FileMode.Open, FileAccess.ReadWrite, FileShare.ReadWrite); + } + catch (FileNotFoundException) when (_createOrOpen && attempt < 3) + { + // Removed between the two attempts: try to create it again. + } + } + if (_createdBackingFile) + { // SetLength only reserves address space in tmpfs. If the filesystem cannot hold the region, // the first write past what fits kills the process with SIGBUS, which cannot be caught. EnsureFreeSpace("/dev/shm", totalSize, _name); _backingFile.SetLength(totalSize); } - else if (_createOrOpen && _backingFile.Length != totalSize) + else { - // Mirror the Windows behavior where capacity mismatch on open throws. + // The creator makes the file empty and sizes it a moment later; wait for that. + WaitForLength(_backingFile, OpenerWaitMs); + long actualLen = _backingFile.Length; - _backingFile.Dispose(); - _backingFile = null; - throw new InvalidOperationException( - $"Existing shared memory '{_backingFilePath}' has size {actualLen} but {totalSize} was requested. " + - $"Either match the existing size or remove the file."); + if (actualLen == 0) + { + _backingFile.Dispose(); + _backingFile = null; + throw new InvalidDataException( + $"Existing shared memory '{_backingFilePath}' is empty and has not been initialized. " + + "If its creator crashed, stop every user of the region and call MemoryRegion.Remove(name)."); + } + + if (_createOrOpen && actualLen != totalSize) + { + // Mirror the Windows behavior where capacity mismatch on open throws. + _backingFile.Dispose(); + _backingFile = null; + throw new InvalidOperationException( + $"Existing shared memory '{_backingFilePath}' has size {actualLen} but {totalSize} was requested. " + + $"Either match the existing size or remove the file."); + } } _mmf = MemoryMappedFile.CreateFromFile( @@ -581,6 +685,30 @@ private void CreateMmfFromLinuxDevShm(long totalSize) _backingFile = null; // ownership transferred — don't double-dispose } + private static void WaitForLength(FileStream file, int timeoutMs) + { + var sw = Stopwatch.StartNew(); + while (file.Length == 0 && sw.ElapsedMilliseconds < timeoutMs) + Thread.Sleep(1); + } + + /// + /// Throws the exception for a header whose magic number is neither the version 3 one nor the marker of a + /// header that is still being written. + /// + private void ThrowForUnexpectedMagic(uint observed) + { + if (observed == 0x48504D53) + { + throw new InvalidDataException( + $"Memory region '{_name}' uses the 2.x format. Stop all 2.x processes, " + + "remove the old region, and recreate it with InterprocessMemory 3.0."); + } + + throw new InvalidDataException( + $"Memory region '{_name}' has an invalid version 3 header (magic=0x{observed:X8})."); + } + /// /// Throws when the filesystem that will hold a new region has fewer than /// bytes available. Best effort: a filesystem that cannot report @@ -694,15 +822,15 @@ private bool InitializeOrOpen() break; if (observed != SharedHeader.MagicInitializing) { - if (observed == 0x48504D53) + // OpenExisting can arrive after the creator made the file but before its first header + // write. That is not a corrupt header: wait for the creator. + if (observed == 0 && !_createOrOpen && sw.ElapsedMilliseconds < OpenerWaitMs) { - throw new InvalidDataException( - $"Memory region '{_name}' uses the 2.x format. Stop all 2.x processes, " + - "remove the old region, and recreate it with InterprocessMemory 3.0."); + Thread.Sleep(1); + continue; } - throw new InvalidDataException( - $"Memory region '{_name}' has an invalid version 3 header (magic=0x{observed:X8})."); + ThrowForUnexpectedMagic(observed); } if (sw.Elapsed > TimeSpan.FromSeconds(5)) throw new TimeoutException( @@ -948,7 +1076,7 @@ public Memory GetMemory(long offset, int length) ValidateOffset(offset, length); byte* ptr = GetDataPtr() + offset; - return new UnmanagedMemoryManager(ptr, length).Memory; + return new UnmanagedMemoryManager(ptr, length, owner: this).Memory; } /// @@ -1231,6 +1359,12 @@ private bool OwnerProcessIsGone(SharedHeader* header, int ownerPid) if (process.HasExited) return true; + // A process that was killed but not yet reaped by its parent (a container whose init does not + // reap, a supervisor that is busy) still exists for GetProcessById and HasExited, but it holds + // nothing any more. + if (OperatingSystem.IsLinux() && IsLinuxZombie(ownerPid)) + return true; + // PID-reuse defense: even when a process with this PID exists, it might be an // unrelated process that the OS recycled the PID for after the real owner died. // Compare the captured start time; mismatch ⇒ impostor ⇒ orphan. diff --git a/InterprocessMemory/UnmanagedMemoryManager.cs b/InterprocessMemory/UnmanagedMemoryManager.cs index 369159a..a817ecd 100644 --- a/InterprocessMemory/UnmanagedMemoryManager.cs +++ b/InterprocessMemory/UnmanagedMemoryManager.cs @@ -22,10 +22,16 @@ internal sealed unsafe class UnmanagedMemoryManager : MemoryManager where private readonly T* _pointer; private readonly int _length; - public UnmanagedMemoryManager(T* pointer, int length) + // The object that owns the mapping the pointer points into. Holding it makes every Memory keep + // it reachable: without this, a caller that drops the MemoryRegion and keeps only the Memory lets + // the garbage collector finalize the region, which unmaps the memory under the Memory. + private readonly object? _owner; + + public UnmanagedMemoryManager(T* pointer, int length, object? owner = null) { _pointer = pointer; _length = length; + _owner = owner; } public override Span GetSpan() => new(_pointer, _length); @@ -43,6 +49,9 @@ public override MemoryHandle Pin(int elementIndex = 0) public override void Unpin() { } + /// The object that owns the mapping behind the pointer, when it was given. + internal object? Owner => _owner; + protected override void Dispose(bool disposing) { } } } From 5da9d6478bd4f89717d4e2dcbd6ea60bd6d678bb Mon Sep 17 00:00:00 2001 From: zacala Date: Fri, 9 Oct 2026 05:00:14 +0000 Subject: [PATCH 25/29] Make the lock guards refuse copies and out-of-order release A guard knew only whether it had taken the region lock, so disposing a copy of it decremented the thread's depth a second time. A thread inside an outer lock then believed it held none, and the next automatic lock tried to take the lock it already held (timeout, or for good with an infinite timeout). Releasing a write guard while a read guard taken inside it was open removed the region lock under that read guard. The guards of StructuredMemory and SharedArray now carry the depth at which they were taken. A release at another depth, or of a write lock under an open read guard, throws SynchronizationLockException before it changes anything, and the guard stays valid so it can be released properly. The cached depth delegates of StructuredMemory are gone, the guard holds its owner instead. Dispose of either class with a guard still open on the calling thread left the cross-process write lock held until the process ended; its owner is alive, so no waiter recovered it. Dispose releases the calling thread's lock, and a guard that outlives the instance disposes quietly. --- CHANGELOG.md | 8 + .../LibraryHardeningTests.cs | 66 ++++++ InterprocessMemory.Tests/SharedArrayTests.cs | 72 +++++++ InterprocessMemory/SharedArray.cs | 115 ++++++++--- InterprocessMemory/StructuredMemory.cs | 188 ++++++++++-------- README.md | 5 +- 6 files changed, 336 insertions(+), 118 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 50c293e..5c0b3ba 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -33,6 +33,14 @@ Changes since the 3.0.0 release. Migrating from 2.x: see [MIGRATION.md](MIGRATIO waiter looked the owner's pid up in its own namespace and did not find it. The owner now records its PID namespace (the inode of `/proc/self/ns/pid`, in the reserved part of the header) and a waiter in another namespace no longer declares the owner dead from its pid. +- The lock guards of `StructuredMemory` and `SharedArray` now remember the depth at which they were taken. + Disposing a copy of a guard a second time used to decrement the thread's depth again, so a thread inside an + outer lock believed it held none and tried to take the lock it already held; releasing a write guard before + a read guard taken inside it removed the protection under that read guard. Both throw + `SynchronizationLockException` now, before anything changes, and the guard stays valid. +- Disposing a `StructuredMemory` or `SharedArray` while one of its guards was open on the calling thread + left the cross-process lock held until the process ended (its owner was alive, so nobody recovered it). It + releases that lock now; a guard that outlives its instance can still be disposed. - `StructuredMemory.AcquireWriteLock()` called while the thread holds a read guard set the writer flag and waited for that thread's own read lock, blocking every other process until the timeout. It now throws `InvalidOperationException` at once, like the automatic write lock and `SharedArray` do. diff --git a/InterprocessMemory.Tests/LibraryHardeningTests.cs b/InterprocessMemory.Tests/LibraryHardeningTests.cs index 47a19e0..2143d98 100644 --- a/InterprocessMemory.Tests/LibraryHardeningTests.cs +++ b/InterprocessMemory.Tests/LibraryHardeningTests.cs @@ -967,6 +967,72 @@ public void GetMemory_KeepsTheRegionAliveForAsLongAsTheMemoryIsUsed() GC.KeepAlive(memory); } + [Test, Timeout(30000)] + public void StructuredMemory_LockGuardCopyDisposedTwice_IsRefusedWithoutDisturbingTheLockState() + { + string name = N("GuardCopy"); + using var memory = StructuredMemory.CreateOrOpen(name, new GuidSchema()); + using var peer = StructuredMemory.OpenExisting(name, new GuidSchema()); + + using (memory.AcquireWriteLock()) + { + var inner = memory.AcquireWriteLock(); + var copy = inner; + inner.Dispose(); + + Assert.Throws(() => copy.Dispose()); + + // Still inside the outer guard: an automatic lock must not try to take the region lock again. + memory.Write("Value", Guid.NewGuid()); + } + + Assert.That(Task.Run(() => { using (peer.AcquireWriteLock(TimeSpan.FromSeconds(2))) { } }).Wait(TimeSpan.FromSeconds(5)), Is.True); + } + + [Test, Timeout(30000)] + public void StructuredMemory_WriteReleasedBeforeTheReadGuardInsideIt_IsRefusedAndTheLockStaysHeld() + { + string name = N("GuardOrder"); + using var memory = StructuredMemory.CreateOrOpen(name, new GuidSchema()); + using var peer = StructuredMemory.OpenExisting(name, new GuidSchema()); + + var write = memory.AcquireWriteLock(); + var read = memory.AcquireReadLock(); + + Assert.Throws(() => write.Dispose()); + Assert.That(Task.Run(() => peer.Read("Value")).Wait(TimeSpan.FromMilliseconds(300)), Is.False, + "the write lock must still be held"); + + read.Dispose(); + write.Dispose(); + Assert.That(Task.Run(() => peer.Read("Value")).Wait(TimeSpan.FromSeconds(5)), Is.True); + } + + [Test, Timeout(30000)] + public void StructuredMemory_DisposeWithAnOpenGuardOnThisThread_DoesNotLeaveTheCrossProcessLockHeld() + { + string name = N("DisposeOpenGuard"); + var memory = StructuredMemory.CreateOrOpen(name, new GuidSchema()); + using var peer = StructuredMemory.OpenExisting(name, new GuidSchema()); + + var guard = memory.AcquireWriteLock(); + memory.Dispose(); + + Assert.That(Task.Run(() => { using (peer.AcquireWriteLock(TimeSpan.FromSeconds(2))) { } }).Wait(TimeSpan.FromSeconds(5)), Is.True, + "the lock was still held after the instance was disposed"); + + Assert.DoesNotThrow(() => guard.Dispose(), "a guard that outlives its instance is harmless"); + } + + // A 16 byte field: written and read under the shared lock, which is what the guard tests need. + public struct GuidSchema : IMemorySchema + { + public IEnumerable GetFields() + { + yield return FieldDefinition.Scalar("Value"); + } + } + public struct BlobSchema : IMemorySchema { public const string Data = "Data"; diff --git a/InterprocessMemory.Tests/SharedArrayTests.cs b/InterprocessMemory.Tests/SharedArrayTests.cs index b0eb4cf..ecdf6b0 100644 --- a/InterprocessMemory.Tests/SharedArrayTests.cs +++ b/InterprocessMemory.Tests/SharedArrayTests.cs @@ -321,6 +321,78 @@ public void WideElements_AreNeverObservedTorn_AcrossInstances() Assert.That(torn, Is.EqualTo(0), $"{torn} of {reads} reads were torn"); } + [Test, Timeout(30000)] + public void LockGuard_CopyDisposedTwice_IsRefusedWithoutDisturbingTheLockState() + { + // A ref struct can still be copied. The second Dispose used to decrement the thread's depth again, + // so the thread believed it held no lock while it was still inside an outer one. + string name = LockName("GuardCopy"); + using var array = SharedArray.CreateOrOpen(name, 4); + using var peer = SharedArray.OpenExisting(name); + + using (array.AcquireWriteLock()) + { + var inner = array.AcquireWriteLock(); + var copy = inner; + inner.Dispose(); + + bool refused = false; + try + { copy.Dispose(); } + catch (SynchronizationLockException) { refused = true; } + Assert.That(refused, Is.True); + + // Still inside the outer guard: the wide indexer must not try to lock again. + array[0] = Wide(1); + Assert.That(array[0].A, Is.EqualTo(1)); + } + + // And the region lock was released exactly once: another instance gets it at once. + Assert.That(Task.Run(() => { using (peer.AcquireWriteLock(TimeSpan.FromSeconds(2))) { } }).Wait(TimeSpan.FromSeconds(5)), Is.True); + } + + [Test, Timeout(30000)] + public void LockGuard_WriteReleasedBeforeTheReadGuardInsideIt_IsRefusedAndTheLockStaysHeld() + { + string name = LockName("GuardOrder"); + using var array = SharedArray.CreateOrOpen(name, 4); + using var peer = SharedArray.OpenExisting(name); + + var write = array.AcquireWriteLock(); + var read = array.AcquireReadLock(); + + bool refused = false; + try + { write.Dispose(); } + catch (SynchronizationLockException) { refused = true; } + Assert.That(refused, Is.True); + + // The write lock is still held, so another instance still cannot read a wide element. + Assert.That(Task.Run(() => peer[0]).Wait(TimeSpan.FromMilliseconds(300)), Is.False); + + read.Dispose(); + write.Dispose(); + Assert.That(Task.Run(() => peer[0]).Wait(TimeSpan.FromSeconds(5)), Is.True); + } + + [Test, Timeout(30000)] + public void Dispose_WithAnOpenGuardOnThisThread_DoesNotLeaveTheCrossProcessLockHeld() + { + // The owner of the leaked lock is this very process, which is alive, so no waiter would ever + // have recovered it: every other process timed out for as long as this one ran. + string name = LockName("DisposeOpenGuard"); + var array = SharedArray.CreateOrOpen(name, 4); + using var peer = SharedArray.OpenExisting(name); + + var guard = array.AcquireWriteLock(); + array.Dispose(); + + Assert.That(Task.Run(() => { using (peer.AcquireWriteLock(TimeSpan.FromSeconds(2))) { } }).Wait(TimeSpan.FromSeconds(5)), Is.True, + "the lock was still held after the array was disposed"); + + guard.Dispose(); // a guard that outlives its array is harmless + } + // Two bytes: the width that used to be copied as a one byte store plus a two byte store. public struct Pair2 { public byte X, Y; } diff --git a/InterprocessMemory/SharedArray.cs b/InterprocessMemory/SharedArray.cs index 24cc14c..d77bcdc 100644 --- a/InterprocessMemory/SharedArray.cs +++ b/InterprocessMemory/SharedArray.cs @@ -248,27 +248,27 @@ private void WriteElement(int index, T value) private T ReadElementLocked(int index) { - bool tookRegionLock = EnterRead(MemoryRegionOptions.DefaultLockTimeout); + LockTicket ticket = EnterRead(MemoryRegionOptions.DefaultLockTimeout); try { return ReadElement(index); } finally { - ExitRead(tookRegionLock); + ExitRead(ticket); } } private void WriteElementLocked(int index, T value) { - bool tookRegionLock = EnterWrite(MemoryRegionOptions.DefaultLockTimeout); + LockTicket ticket = EnterWrite(MemoryRegionOptions.DefaultLockTimeout); try { WriteElement(index, value); } finally { - ExitWrite(tookRegionLock); + ExitWrite(ticket); } } @@ -297,14 +297,14 @@ public void CopyTo(int startIndex, Span destination) return; } - bool tookRegionLock = EnterRead(MemoryRegionOptions.DefaultLockTimeout); + LockTicket ticket = EnterRead(MemoryRegionOptions.DefaultLockTimeout); try { CopyToCore(startIndex, destination); } finally { - ExitRead(tookRegionLock); + ExitRead(ticket); } } @@ -343,14 +343,14 @@ public void CopyFrom(int startIndex, ReadOnlySpan source) return; } - bool tookRegionLock = EnterWrite(MemoryRegionOptions.DefaultLockTimeout); + LockTicket ticket = EnterWrite(MemoryRegionOptions.DefaultLockTimeout); try { CopyFromCore(startIndex, source); } finally { - ExitWrite(tookRegionLock); + ExitWrite(ticket); } } @@ -391,14 +391,14 @@ public void Fill(T value, int startIndex = 0, int count = -1) return; } - bool tookRegionLock = EnterWrite(MemoryRegionOptions.DefaultLockTimeout); + LockTicket ticket = EnterWrite(MemoryRegionOptions.DefaultLockTimeout); try { FillCore(value, startIndex, count); } finally { - ExitWrite(tookRegionLock); + ExitWrite(ticket); } } @@ -507,7 +507,25 @@ public void ForceResetLocks() private bool IsHoldingWriteLock() => _writeLockDepth.Value > 0; /// Returns true when the region lock was taken, false for a reentrant acquisition. - private bool EnterWrite(TimeSpan timeout) + /// + /// Proof of one acquisition: the depth the thread was at right afterwards, and whether this + /// acquisition took the region lock (false for a reentrant one). Releasing checks the depth, so a copy + /// of a guard, a guard released out of order or one released on another thread is refused instead of + /// corrupting the bookkeeping of the thread that really holds the lock. + /// + internal readonly struct LockTicket + { + public readonly bool TookRegionLock; + public readonly int Depth; + + public LockTicket(bool tookRegionLock, int depth) + { + TookRegionLock = tookRegionLock; + Depth = depth; + } + } + + private LockTicket EnterWrite(TimeSpan timeout) { ThrowIfDisposed(); TimeoutHelper.Validate(timeout, nameof(timeout)); @@ -525,25 +543,37 @@ private bool EnterWrite(TimeSpan timeout) tookRegionLock = true; } - _writeLockDepth.Value++; - return tookRegionLock; + return new LockTicket(tookRegionLock, ++_writeLockDepth.Value); } - private void ExitWrite(bool tookRegionLock) + private void ExitWrite(LockTicket ticket) { + // Dispose() already released what this thread held. + if (_disposed != 0) + return; + + if (_writeLockDepth.Value != ticket.Depth) + throw new SynchronizationLockException( + "A write lock guard was released twice, out of order or on another thread. Nothing was released."); + + // A read guard taken inside the write lock holds no region lock of its own; releasing the write + // lock under it would leave that guard reading without any protection. + if (ticket.TookRegionLock && _readLockDepth.Value > 0) + throw new SynchronizationLockException( + "The write lock cannot be released while a read lock guard taken inside it is still open. Nothing was released."); + try { - if (tookRegionLock) + if (ticket.TookRegionLock) _buffer.ReleaseWriteLock(); } finally { - _writeLockDepth.Value--; + _writeLockDepth.Value = ticket.Depth - 1; } } - /// Returns true when the region lock was taken, false for a reentrant acquisition. - private bool EnterRead(TimeSpan timeout) + private LockTicket EnterRead(TimeSpan timeout) { ThrowIfDisposed(); TimeoutHelper.Validate(timeout, nameof(timeout)); @@ -556,20 +586,26 @@ private bool EnterRead(TimeSpan timeout) tookRegionLock = true; } - _readLockDepth.Value++; - return tookRegionLock; + return new LockTicket(tookRegionLock, ++_readLockDepth.Value); } - private void ExitRead(bool tookRegionLock) + private void ExitRead(LockTicket ticket) { + if (_disposed != 0) + return; + + if (_readLockDepth.Value != ticket.Depth) + throw new SynchronizationLockException( + "A read lock guard was released twice, out of order or on another thread. Nothing was released."); + try { - if (tookRegionLock) + if (ticket.TookRegionLock) _buffer.ReleaseReadLock(); } finally { - _readLockDepth.Value--; + _readLockDepth.Value = ticket.Depth - 1; } } @@ -581,12 +617,12 @@ private void ExitRead(bool tookRegionLock) public ref struct WriteLock { private SharedArray? _owner; - private readonly bool _tookRegionLock; + private readonly LockTicket _ticket; - internal WriteLock(SharedArray owner, bool tookRegionLock) + internal WriteLock(SharedArray owner, LockTicket ticket) { _owner = owner; - _tookRegionLock = tookRegionLock; + _ticket = ticket; } /// Releases the write lock; disposing more than once has no further effect. @@ -596,8 +632,10 @@ public void Dispose() if (owner is null) return; + // A refused release (a copy, or an order that would leave a read guard unprotected) throws + // before anything changes, and the guard stays valid so that it can be released properly. + owner.ExitWrite(_ticket); _owner = null; - owner.ExitWrite(_tookRegionLock); } } @@ -608,12 +646,12 @@ public void Dispose() public ref struct ReadLock { private SharedArray? _owner; - private readonly bool _tookRegionLock; + private readonly LockTicket _ticket; - internal ReadLock(SharedArray owner, bool tookRegionLock) + internal ReadLock(SharedArray owner, LockTicket ticket) { _owner = owner; - _tookRegionLock = tookRegionLock; + _ticket = ticket; } /// Releases the read lock; disposing more than once has no further effect. @@ -623,8 +661,8 @@ public void Dispose() if (owner is null) return; + owner.ExitRead(_ticket); _owner = null; - owner.ExitRead(_tookRegionLock); } } @@ -645,6 +683,21 @@ public void Dispose() if (Interlocked.Exchange(ref _disposed, 1) != 0) return; + // A guard that is still open on this thread would otherwise leave the cross-process lock held + // until the process ends (its owner is alive, so no waiter would recover it). Guards held by + // other threads cannot be reached from here: stop and join those threads before disposing. + try + { + if (_writeLockDepth.Value > 0) + _buffer.ReleaseWriteLock(); + else if (_readLockDepth.Value > 0) + _buffer.ReleaseReadLock(); + } + catch (Exception ex) when (ex is SynchronizationLockException or ObjectDisposedException) + { + // The lock was taken over (orphan recovery or ForceResetLocks) or the region is gone. + } + // No finalizer: if Dispose is never called, the MemoryRegion's own finalizer unmaps the memory. _buffer?.Dispose(); _writeLockDepth.Dispose(); diff --git a/InterprocessMemory/StructuredMemory.cs b/InterprocessMemory/StructuredMemory.cs index bcd6cf9..eaeb2d0 100644 --- a/InterprocessMemory/StructuredMemory.cs +++ b/InterprocessMemory/StructuredMemory.cs @@ -53,8 +53,6 @@ public sealed unsafe class StructuredMemory : IDisposable where TSchema // A method group is converted to a new delegate on every use, which allocated on each // automatically locked read or write. Convert once per instance instead. - private readonly Action _decrementWriteLockDepth; - private readonly Action _decrementReadLockDepth; /// /// Gets the schema instance defining the memory layout @@ -110,8 +108,6 @@ internal StructuredMemory( _schema = schema; _compatibility = compatibility; - _decrementWriteLockDepth = DecrementWriteLockDepth; - _decrementReadLockDepth = DecrementReadLockDepth; _fields = BuildFieldMetadata(schema); if (_fields.Count == 0) @@ -765,10 +761,7 @@ public WriteLock AcquireWriteLock(TimeSpan timeout) // Reentrant: if already holding write lock, just increment depth if (_writeLockDepth.Value > 0) - { - IncrementWriteLockDepth(); - return new WriteLock(null, _decrementWriteLockDepth); - } + return new WriteLock(this, null, ++_writeLockDepth.Value); // The write lock waits for every reader to leave, and this thread is one of them. Waiting would // block all other processes (new readers and writers queue behind the pending writer) until the @@ -780,8 +773,7 @@ public WriteLock AcquireWriteLock(TimeSpan timeout) if (!_buffer.TryAcquireWriteLock(timeout)) throw new TimeoutException($"Failed to acquire write lock within {timeout}"); - IncrementWriteLockDepth(); - return new WriteLock(_buffer, _decrementWriteLockDepth); + return new WriteLock(this, _buffer, ++_writeLockDepth.Value); } /// @@ -815,16 +807,12 @@ public ReadLock AcquireReadLock(TimeSpan timeout) // Reentrant: if already holding any lock, just increment depth if (_readLockDepth.Value > 0 || _writeLockDepth.Value > 0) - { - IncrementReadLockDepth(); - return new ReadLock(null, _decrementReadLockDepth); - } + return new ReadLock(this, null, ++_readLockDepth.Value); if (!_buffer.TryAcquireReadLock(timeout)) throw new TimeoutException($"Failed to acquire read lock within {timeout}"); - IncrementReadLockDepth(); - return new ReadLock(_buffer, _decrementReadLockDepth); + return new ReadLock(this, _buffer, ++_readLockDepth.Value); } /// @@ -1136,21 +1124,6 @@ private void ThrowIfDisposed() [MethodImpl(MethodImplOptions.AggressiveInlining)] private int GetWriteLockDepth() => _writeLockDepth.Value; - [MethodImpl(MethodImplOptions.AggressiveInlining)] - private void IncrementWriteLockDepth() => _writeLockDepth.Value++; - - [MethodImpl(MethodImplOptions.AggressiveInlining)] - private void DecrementWriteLockDepth() => _writeLockDepth.Value--; - - [MethodImpl(MethodImplOptions.AggressiveInlining)] - private int GetReadLockDepth() => _readLockDepth.Value; - - [MethodImpl(MethodImplOptions.AggressiveInlining)] - private void IncrementReadLockDepth() => _readLockDepth.Value++; - - [MethodImpl(MethodImplOptions.AggressiveInlining)] - private void DecrementReadLockDepth() => _readLockDepth.Value--; - [MethodImpl(MethodImplOptions.AggressiveInlining)] private bool IsHoldingAnyLock() => _writeLockDepth.Value > 0 || _readLockDepth.Value > 0; @@ -1184,6 +1157,21 @@ public void Dispose() if (Interlocked.Exchange(ref _disposed, 1) != 0) return; + // A guard that is still open on this thread would otherwise leave the cross-process lock held + // until the process ends (its owner is alive, so no waiter would recover it). Guards held by + // other threads cannot be reached from here: stop and join those threads before disposing. + try + { + if (_writeLockDepth.Value > 0) + _buffer.ReleaseWriteLock(); + else if (_readLockDepth.Value > 0) + _buffer.ReleaseReadLock(); + } + catch (Exception ex) when (ex is SynchronizationLockException or ObjectDisposedException) + { + // The lock was taken over (orphan recovery or ForceResetLocks) or the region is gone. + } + _buffer?.Dispose(); _writeLockDepth.Dispose(); _readLockDepth.Dispose(); @@ -1198,27 +1186,74 @@ private static void EnsureScalarField(FieldMetadata metadata) } } + internal void ExitWriteLock(IMemoryRegion? regionToRelease, int depth) + { + // Dispose() already released what this thread held. + if (_disposed != 0) + return; + + if (_writeLockDepth.Value != depth) + throw new SynchronizationLockException( + "A write lock guard was disposed twice, out of order or on another thread. Nothing was released."); + + // A read guard taken inside the write lock holds no region lock of its own; releasing the write + // lock under it would leave that guard reading without any protection. + if (regionToRelease != null && _readLockDepth.Value > 0) + throw new SynchronizationLockException( + "The write lock cannot be released while a read lock guard taken inside it is still open. Nothing was released."); + + try + { + regionToRelease?.ReleaseWriteLock(); + } + finally + { + _writeLockDepth.Value = depth - 1; + } + } + + internal void ExitReadLock(IMemoryRegion? regionToRelease, int depth) + { + if (_disposed != 0) + return; + + if (_readLockDepth.Value != depth) + throw new SynchronizationLockException( + "A read lock guard was disposed twice, out of order or on another thread. Nothing was released."); + + try + { + regionToRelease?.ReleaseReadLock(); + } + finally + { + _readLockDepth.Value = depth - 1; + } + } + /// - /// RAII wrapper for write lock with double-dispose and reentrant safety. - /// When acquired via reentrant path, buffer is null and Dispose only decrements the depth counter. + /// RAII wrapper for the write lock. A reentrant acquisition holds no region of its own and only + /// decrements the depth when it is disposed. /// - /// Do not copy this struct. The double-dispose guard (an - /// Interlocked.Exchange on _onDispose) operates on the struct's own field, - /// not a shared one. Copying the struct duplicates the field, and each copy's - /// Dispose() will independently decrement the lock depth — corrupting the - /// reentrant depth counter and potentially releasing the underlying buffer lock twice. - /// Always consume via using var ... = AcquireWriteLock(...), never var copy = lock;. + /// Disposing the same variable twice has no further effect. A copy of the guard is a + /// different matter: it remembers the depth at which the lock was taken, and disposing it after the + /// original (or in any other order than last-acquired-first-released) throws + /// without touching the lock state, instead of silently + /// corrupting the depth counter of the thread. Consume it with using var ... = + /// AcquireWriteLock(...). /// public struct WriteLock : IDisposable { - private IMemoryRegion? _buffer; - private Action? _onDispose; + private StructuredMemory? _owner; + private readonly IMemoryRegion? _regionToRelease; + private readonly int _depth; private readonly int _ownerThreadId; - internal WriteLock(IMemoryRegion? buffer, Action? onDispose = null) + internal WriteLock(StructuredMemory owner, IMemoryRegion? regionToRelease, int depth) { - _buffer = buffer; - _onDispose = onDispose; + _owner = owner; + _regionToRelease = regionToRelease; + _depth = depth; _ownerThreadId = Environment.CurrentManagedThreadId; } @@ -1227,11 +1262,13 @@ internal WriteLock(IMemoryRegion? buffer, Action? onDispose = null) /// /// /// Called on a different thread than the one that acquired the lock, which is what happens - /// when the guarded scope contains an await. Nothing is released in that case. + /// when the guarded scope contains an await, or called on a copy of a guard that was + /// already released, or out of order. Nothing is released in these cases. /// public void Dispose() { - if (_onDispose == null) + StructuredMemory? owner = _owner; + if (owner is null) return; // The lock and its reentrancy depth belong to the acquiring thread. Releasing from @@ -1241,40 +1278,28 @@ public void Dispose() "A write lock guard must be disposed on the thread that acquired it. " + "Do not await inside a lock scope."); - var onDispose = Interlocked.Exchange(ref _onDispose, null); - if (onDispose != null) - { - try - { - _buffer?.ReleaseWriteLock(); - } - finally - { - _buffer = null; - onDispose.Invoke(); - } - } + // A refused release (a copy, or an order that would leave a read guard unprotected) throws + // before anything changes, and the guard stays valid so that it can be released properly. + owner.ExitWriteLock(_regionToRelease, _depth); + _owner = null; } } /// - /// RAII wrapper for read lock with double-dispose and reentrant safety. - /// When acquired via reentrant path, buffer is null and Dispose only decrements the depth counter. - /// - /// Do not copy this struct. Same reasoning as — - /// copies will each run Dispose(), double-decrementing the reentrant depth and - /// potentially releasing the underlying buffer lock twice. + /// RAII wrapper for the read lock; see for the rules about copies and order. /// public struct ReadLock : IDisposable { - private IMemoryRegion? _buffer; - private Action? _onDispose; + private StructuredMemory? _owner; + private readonly IMemoryRegion? _regionToRelease; + private readonly int _depth; private readonly int _ownerThreadId; - internal ReadLock(IMemoryRegion? buffer, Action? onDispose = null) + internal ReadLock(StructuredMemory owner, IMemoryRegion? regionToRelease, int depth) { - _buffer = buffer; - _onDispose = onDispose; + _owner = owner; + _regionToRelease = regionToRelease; + _depth = depth; _ownerThreadId = Environment.CurrentManagedThreadId; } @@ -1282,12 +1307,14 @@ internal ReadLock(IMemoryRegion? buffer, Action? onDispose = null) /// Releases the read lock if not already released. /// /// - /// Called on a different thread than the one that acquired the lock. Nothing is released - /// in that case. See . + /// Called on a different thread than the one that acquired the lock, or on a copy of a guard that + /// was already released, or out of order. Nothing is released in these cases. + /// See . /// public void Dispose() { - if (_onDispose == null) + StructuredMemory? owner = _owner; + if (owner is null) return; if (Environment.CurrentManagedThreadId != _ownerThreadId) @@ -1295,19 +1322,8 @@ public void Dispose() "A read lock guard must be disposed on the thread that acquired it. " + "Do not await inside a lock scope."); - var onDispose = Interlocked.Exchange(ref _onDispose, null); - if (onDispose != null) - { - try - { - _buffer?.ReleaseReadLock(); - } - finally - { - _buffer = null; - onDispose.Invoke(); - } - } + owner.ExitReadLock(_regionToRelease, _depth); + _owner = null; } } diff --git a/README.md b/README.md index c770ba0..59631db 100644 --- a/README.md +++ b/README.md @@ -228,7 +228,10 @@ fields form one transaction. The lock guards returned by `AcquireWriteLock()` and `AcquireReadLock()` must be disposed on the thread that acquired them. Keep the guarded scope synchronous: an `await` inside it lets the guard -be disposed on another thread, which throws `SynchronizationLockException`. +be disposed on another thread, which throws `SynchronizationLockException`. So does disposing a *copy* of +a guard after the original, or releasing the guards in another order than last-taken-first-released: the +lock state is left untouched and the guard stays valid, so it can still be released properly. Disposing +the instance while one of its guards is open on the calling thread releases that lock. ## Shared arrays From 1017f046ac751fe9c3be1eca902eeb70700eb317 Mon Sep 17 00:00:00 2001 From: zacala Date: Fri, 9 Oct 2026 05:02:22 +0000 Subject: [PATCH 26/29] Check the schema version before the size of a structured-memory region OpenExisting compared the region size with the schema size before looking at the stored version, so no SchemaCompatibility mode could ever open a region of another size: Forward and Full only worked when an appended field happened to fit in the 64 byte rounding, and Strict reported a size mismatch instead of the version mismatch it exists to report. The existing Strict test passed for that wrong reason. The version is checked first. For different versions the region must only be at least as large as the schema, so an older schema can read the larger region of a newer one (it uses its first fields); a region smaller than the schema is rejected at open time. The same version still needs the exact size. SchemaCompatibility documents what is and is not checked, and the README has a short section on it. --- CHANGELOG.md | 6 ++ .../LibraryHardeningTests.cs | 98 +++++++++++++++++++ .../StructuredMemoryTests.cs | 7 +- InterprocessMemory/SchemaTypes.cs | 20 +++- InterprocessMemory/StructuredMemory.cs | 30 ++++-- README.md | 9 ++ 6 files changed, 158 insertions(+), 12 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 5c0b3ba..ad73f64 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -33,6 +33,12 @@ Changes since the 3.0.0 release. Migrating from 2.x: see [MIGRATION.md](MIGRATIO waiter looked the owner's pid up in its own namespace and did not find it. The owner now records its PID namespace (the inode of `/proc/self/ns/pid`, in the reserved part of the header) and a waiter in another namespace no longer declares the owner dead from its pid. +- `StructuredMemory.OpenExisting` checked the size of the region before the schema version, so no + `SchemaCompatibility` mode could open a region of another size (`Forward` and `Full` only worked when an + appended field fitted in the padding) and `Strict` reported a size mismatch instead of the version. The + version is checked first now; for different versions the region must only be at least as large as the + schema, so an older schema can read a larger region written by a newer one. The same version still needs + the exact size. See "Schema versions" in the README. - The lock guards of `StructuredMemory` and `SharedArray` now remember the depth at which they were taken. Disposing a copy of a guard a second time used to decrement the thread's depth again, so a thread inside an outer lock believed it held none and tried to take the lock it already held; releasing a write guard before diff --git a/InterprocessMemory.Tests/LibraryHardeningTests.cs b/InterprocessMemory.Tests/LibraryHardeningTests.cs index 2143d98..9ad40dc 100644 --- a/InterprocessMemory.Tests/LibraryHardeningTests.cs +++ b/InterprocessMemory.Tests/LibraryHardeningTests.cs @@ -1033,6 +1033,104 @@ public IEnumerable GetFields() } } + public struct EvolvingV1 : IVersionedSchema + { + public int Version => 1; + public bool IsCompatibleWith(int otherVersion) => true; + + public IEnumerable GetFields() + { + yield return FieldDefinition.Scalar("A"); + yield return FieldDefinition.Scalar("B"); + } + } + + // The same fields as V1 plus an appended one that makes the region larger. + public struct EvolvingV2 : IVersionedSchema + { + public int Version => 2; + public bool IsCompatibleWith(int otherVersion) => true; + + public IEnumerable GetFields() + { + yield return FieldDefinition.Scalar("A"); + yield return FieldDefinition.Scalar("B"); + yield return FieldDefinition.String("Label", 100); + } + } + + [Test] + public void StructuredMemory_OlderSchema_CanOpenALargerRegionOfANewerVersion_WithForwardCompatibility() + { + // The size check ran before the version check, so no compatibility mode could ever open a region of + // a different size: Forward and Full only worked when the appended field fitted in the padding. + string name = N("EvolveForward"); + using var writer = StructuredMemory.CreateOrOpen(name, new EvolvingV2()); + writer.Write("A", 42); + writer.Write("B", 2.5); + + using var reader = StructuredMemory.OpenExisting(name, new EvolvingV1(), SchemaCompatibility.Forward); + Assert.That(reader.Read("A"), Is.EqualTo(42)); + Assert.That(reader.Read("B"), Is.EqualTo(2.5)); + + using var full = StructuredMemory.OpenExisting(name, new EvolvingV1(), SchemaCompatibility.Full); + Assert.That(full.Read("A"), Is.EqualTo(42)); + } + + [Test] + public void StructuredMemory_StrictOrWrongDirection_ReportsTheVersionMismatchNotTheSize() + { + string name = N("EvolveStrict"); + using var writer = StructuredMemory.CreateOrOpen(name, new EvolvingV2()); + + var strict = Assert.Throws(() => + StructuredMemory.OpenExisting(name, new EvolvingV1())); + Assert.That(strict!.Message, Does.Contain("version mismatch")); + + // Backward means "the region is older than the schema"; this region is newer. + Assert.Throws(() => + StructuredMemory.OpenExisting(name, new EvolvingV1(), SchemaCompatibility.Backward)); + } + + [Test] + public void StructuredMemory_NewerSchema_CannotOpenASmallerRegion() + { + // A region cannot be smaller than the schema that opens it, whatever the mode says: failing here is + // better than failing at the first read of an appended field. + string name = N("EvolveBackward"); + using var writer = StructuredMemory.CreateOrOpen(name, new EvolvingV1()); + + var error = Assert.Throws(() => + StructuredMemory.OpenExisting(name, new EvolvingV2(), SchemaCompatibility.Backward)); + Assert.That(error!.Message, Does.Contain("contains only")); + + Assert.Throws(() => + StructuredMemory.OpenExisting(name, new EvolvingV2(), SchemaCompatibility.Full)); + } + + [Test] + public void StructuredMemory_SameVersionOfAnotherSize_IsStillRejectedAsASizeMismatch() + { + string name = N("EvolveSameVersion"); + using var writer = StructuredMemory.CreateOrOpen(name, new EvolvingV1()); + + Assert.Throws(() => + StructuredMemory.OpenExisting(name, new SameVersionDifferentSize(), SchemaCompatibility.Full)); + } + + public struct SameVersionDifferentSize : IVersionedSchema + { + public int Version => 1; + public bool IsCompatibleWith(int otherVersion) => true; + + public IEnumerable GetFields() + { + yield return FieldDefinition.Scalar("A"); + yield return FieldDefinition.Scalar("B"); + yield return FieldDefinition.String("Extra", 100); + } + } + public struct BlobSchema : IMemorySchema { public const string Data = "Data"; diff --git a/InterprocessMemory.Tests/StructuredMemoryTests.cs b/InterprocessMemory.Tests/StructuredMemoryTests.cs index 039de79..89e07c3 100644 --- a/InterprocessMemory.Tests/StructuredMemoryTests.cs +++ b/InterprocessMemory.Tests/StructuredMemoryTests.cs @@ -318,11 +318,14 @@ public void VersionedSchema_StrictMode_DifferentVersion_ShouldThrow() using var memory1 = new StructuredMemory(uniqueName, schemaV1, create: true); memory1.Write(VersionedSchemaV1.IntField, 42); - // Try to open with V2 in strict mode - should throw + // Try to open with V2 in strict mode - should throw, and because the VERSION differs. This used to + // throw InvalidDataException for the size of the region, which was checked before the version, so + // the test passed without the compatibility mode ever being consulted. var schemaV2 = new VersionedSchemaV2(); - Assert.Throws(() => + var error = Assert.Throws(() => new StructuredMemory( uniqueName, schemaV2, create: false, compatibility: SchemaCompatibility.Strict)); + Assert.That(error!.Message, Does.Contain("version mismatch")); } [Test] diff --git a/InterprocessMemory/SchemaTypes.cs b/InterprocessMemory/SchemaTypes.cs index 71a72fc..fb3961f 100644 --- a/InterprocessMemory/SchemaTypes.cs +++ b/InterprocessMemory/SchemaTypes.cs @@ -52,15 +52,29 @@ public enum FieldTypeCode } /// - /// Schema compatibility mode for version handling + /// Schema compatibility mode for version handling. + /// + /// The library checks the schema version, the size of the region and, when the schema implements + /// , . It does not compare + /// the field layout of two different versions: a newer schema must only append fields (or change nothing + /// that is stored), and a schema that changes anything else has to say so in + /// . A region is never smaller than the schema that opens it, + /// whatever the mode. /// public enum SchemaCompatibility { /// Exact version match required Strict, - /// Allow reading from newer compatible versions + /// + /// Allow opening a region written by a newer version of the schema. The region may be larger than this + /// schema needs; this schema uses its first fields. Use OpenExisting: CreateOrOpen asks for + /// a region of exactly this schema's size. + /// Forward, - /// Allow reading from older compatible versions + /// + /// Allow opening a region written by an older version of the schema. Since the region cannot be smaller + /// than this schema, that means a version change that did not grow the region. + /// Backward, /// Allow both forward and backward compatibility Full diff --git a/InterprocessMemory/StructuredMemory.cs b/InterprocessMemory/StructuredMemory.cs index eaeb2d0..86b9e50 100644 --- a/InterprocessMemory/StructuredMemory.cs +++ b/InterprocessMemory/StructuredMemory.cs @@ -38,6 +38,7 @@ public sealed unsafe class StructuredMemory : IDisposable where TSchema private static readonly TimeSpan DefaultLockTimeout = TimeSpan.FromSeconds(5); private readonly IMemoryRegion _buffer; + private readonly long _totalSize; // bytes the schema needs: header plus all fields // Address of the first byte after the region header, valid until Dispose; used for the // lock-free scalar path (AtomicAccess). private readonly byte* _dataBase; @@ -120,6 +121,7 @@ internal StructuredMemory( _schemaHash = ComputeSchemaHash(); long totalSize = SchemaHeaderSize + CalculateTotalSize(_fields); + _totalSize = totalSize; // No statistics are exposed here, so skip the region's per-call counters. var regionOptions = new MemoryRegionOptions { EnableStatistics = false }; @@ -131,15 +133,10 @@ internal StructuredMemory( } else { + // The size is checked in ValidateSchemaCompatibility, once the stored schema version is + // known: a region written by a newer version of the schema is legitimately larger. _buffer = MemoryRegion.OpenExisting( name, regionOptions, RegionKind.StructuredMemory); - if (_buffer.Capacity != totalSize) - { - _buffer.Dispose(); - throw new InvalidDataException( - $"Structured-memory size mismatch: schema requires {totalSize} bytes, " + - $"but the region contains {_buffer.Capacity} bytes."); - } } try @@ -888,6 +885,15 @@ private void ValidateSchemaCompatibility() int storedFieldCount = BitConverter.ToInt32(header.Slice(8)); int storedHash = BitConverter.ToInt32(header.Slice(12)); + long regionSize = _buffer.Capacity; + + if (StoredSchemaVersion == SchemaVersion && regionSize != _totalSize) + { + throw new InvalidDataException( + $"Structured-memory size mismatch: schema requires {_totalSize} bytes, " + + $"but the region contains {regionSize} bytes."); + } + if (StoredSchemaVersion == SchemaVersion && storedFieldCount != _fields.Count) { throw new InvalidOperationException( @@ -913,6 +919,16 @@ private void ValidateSchemaCompatibility() $"Compatibility mode: {_compatibility}"); } + // Whatever the versions are, the schema can only use what the region holds. A region written + // by a newer schema that appended fields is larger and an older reader uses its prefix; + // a region written by an older, smaller schema cannot hold the fields of this one. + if (regionSize < _totalSize) + { + throw new InvalidDataException( + $"Structured-memory size mismatch: schema version {SchemaVersion} requires {_totalSize} bytes, " + + $"but the region (schema version {StoredSchemaVersion}) contains only {regionSize} bytes."); + } + // Schema-side veto: if the schema itself can declare incompatibility for this // specific pair (e.g., v3 schema knows it cannot safely read v1 even under Full mode), // honor that. Previously IVersionedSchema.IsCompatibleWith was never invoked, leaving diff --git a/README.md b/README.md index 59631db..4e5f6bd 100644 --- a/README.md +++ b/README.md @@ -233,6 +233,15 @@ a guard after the original, or releasing the guards in another order than last-t lock state is left untouched and the guard stays valid, so it can still be released properly. Disposing the instance while one of its guards is open on the calling thread releases that lock. +### Schema versions + +`StructuredMemory.OpenExisting(name, schema, compatibility)` accepts a region of another schema +version according to `SchemaCompatibility`: `Strict` needs the same version, `Forward` also accepts a region +written by a newer version (it may be larger: the older schema uses its first fields, so a newer version must +only append fields), `Backward` a region written by an older version, `Full` both. A region is never smaller +than the schema that opens it. The library does not compare the field layout of two different versions; a schema +that changes more than appended fields has to say so in `IVersionedSchema.IsCompatibleWith`. + ## Shared arrays ```csharp From 407d7af07ffa2ca22be3795c3b2aa60ab875221e Mon Sep 17 00:00:00 2001 From: zacala Date: Fri, 9 Oct 2026 05:03:30 +0000 Subject: [PATCH 27/29] Let SharedArray.Fill and Clear handle elements of 64 KiB or more FillCore staged the value in a T[] (stackalloc or ArrayPool). A managed array cannot hold elements of 64 KiB or more, and a method that mentions T[] or ArrayPool for such a T fails to load at all, so Fill and Clear threw TypeLoadException before writing anything. Independently, the batch was up to 4096 elements whatever their size: 128 MiB of temporary buffer per call for 32 KiB elements. Elements of 32 KiB or more are written one by one from a method that has no array in it; smaller ones are staged in batches of at most 64 KiB. --- CHANGELOG.md | 3 ++ InterprocessMemory.Tests/SharedArrayTests.cs | 43 ++++++++++++++++++++ InterprocessMemory/SharedArray.cs | 28 ++++++++++++- 3 files changed, 72 insertions(+), 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index ad73f64..7920121 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -39,6 +39,9 @@ Changes since the 3.0.0 release. Migrating from 2.x: see [MIGRATION.md](MIGRATIO version is checked first now; for different versions the region must only be at least as large as the schema, so an older schema can read a larger region written by a newer one. The same version still needs the exact size. See "Schema versions" in the README. +- `SharedArray.Fill` and `Clear` threw `TypeLoadException` for elements of 64 KiB or more (a managed array + cannot hold them, and the staging buffer was a `T[]`) and staged up to 4096 elements per batch whatever their + size, 128 MiB for 32 KiB elements. Large elements are written one by one, and a batch is limited to 64 KiB. - The lock guards of `StructuredMemory` and `SharedArray` now remember the depth at which they were taken. Disposing a copy of a guard a second time used to decrement the thread's depth again, so a thread inside an outer lock believed it held none and tried to take the lock it already held; releasing a write guard before diff --git a/InterprocessMemory.Tests/SharedArrayTests.cs b/InterprocessMemory.Tests/SharedArrayTests.cs index ecdf6b0..6c3ac09 100644 --- a/InterprocessMemory.Tests/SharedArrayTests.cs +++ b/InterprocessMemory.Tests/SharedArrayTests.cs @@ -393,6 +393,49 @@ public void Dispose_WithAnOpenGuardOnThisThread_DoesNotLeaveTheCrossProcessLockH guard.Dispose(); // a guard that outlives its array is harmless } + // Larger than a managed array may hold per element (64 KiB), and a mid-sized one that is batched. + [System.Runtime.InteropServices.StructLayout(System.Runtime.InteropServices.LayoutKind.Sequential, Size = 100_000)] + public struct Big100K { public int Tag; } + + [System.Runtime.InteropServices.StructLayout(System.Runtime.InteropServices.LayoutKind.Sequential, Size = 20_000)] + public struct Mid20K { public int Tag; } + + [Test, Timeout(60000)] + public void FillAndClear_ElementsOf64KiBOrMore_Work() + { + // Fill's staging buffer was a T[] and an ArrayPool rental. A managed array cannot hold elements of + // 64 KiB or more, so the method failed to load with a TypeLoadException before writing anything. + using var array = SharedArray.CreateOrOpen(LockName("Fill100K"), 3); + + array.Fill(new Big100K { Tag = 7 }); + Assert.That(array[0].Tag, Is.EqualTo(7)); + Assert.That(array[1].Tag, Is.EqualTo(7)); + Assert.That(array[2].Tag, Is.EqualTo(7)); + + array.Fill(new Big100K { Tag = 9 }, 1, 1); + Assert.That(array[0].Tag, Is.EqualTo(7)); + Assert.That(array[1].Tag, Is.EqualTo(9)); + Assert.That(array[2].Tag, Is.EqualTo(7)); + + array.Clear(); + Assert.That(array[0].Tag, Is.EqualTo(0)); + Assert.That(array[1].Tag, Is.EqualTo(0)); + Assert.That(array[2].Tag, Is.EqualTo(0)); + } + + [Test, Timeout(60000)] + public void FillAndClear_MidSizedElements_AreBatchedWithinTheByteBudget() + { + using var array = SharedArray.CreateOrOpen(LockName("Fill20K"), 10); + + array.Fill(new Mid20K { Tag = 5 }); + for (int i = 0; i < 10; i++) + Assert.That(array[i].Tag, Is.EqualTo(5), $"element {i}"); + + array.Clear(); + Assert.That(array[9].Tag, Is.EqualTo(0)); + } + // Two bytes: the width that used to be copied as a one byte store plus a two byte store. public struct Pair2 { public byte X, Y; } diff --git a/InterprocessMemory/SharedArray.cs b/InterprocessMemory/SharedArray.cs index d77bcdc..e033725 100644 --- a/InterprocessMemory/SharedArray.cs +++ b/InterprocessMemory/SharedArray.cs @@ -402,10 +402,34 @@ public void Fill(T value, int startIndex = 0, int count = -1) } } + // A batch of this many bytes is the most that is staged at a time. + private const int FillBatchBytes = 64 * 1024; + private void FillCore(T value, int startIndex, int count) { - // Batch fill: create a filled buffer and write in chunks - int batchCount = Math.Min(count, 4096); + // A managed array cannot hold elements of 64 KiB or more, and a method that so much as mentions + // T[] or ArrayPool for such a T does not even load (TypeLoadException when it is compiled, + // before a single element is written). A batch of such elements would be one or two elements + // anyway, so they are copied one by one from a method that has no array in it. + if (_elementSize >= FillBatchBytes / 2) + FillOneByOne(value, startIndex, count); + else + FillInBatches(value, startIndex, count); + } + + private void FillOneByOne(T value, int startIndex, int count) + { + ReadOnlySpan one = MemoryMarshal.CreateReadOnlySpan(ref value, 1); + for (int i = 0; i < count; i++) + CopyFromCore(startIndex + i, one); + } + + [MethodImpl(MethodImplOptions.NoInlining)] + private void FillInBatches(T value, int startIndex, int count) + { + // Batch fill: create a filled buffer and write in chunks. The batch is bounded in bytes as well as + // in elements: 4096 elements of 32 KiB would be a 128 MiB temporary buffer on every call. + int batchCount = Math.Min(count, Math.Min(4096, Math.Max(1, FillBatchBytes / _elementSize))); int batchBytes = batchCount * _elementSize; // Use stackalloc for small batches, ArrayPool for large. From 5280430acf4e4760dd65c010b9b42b402d9a165e Mon Sep 17 00:00:00 2001 From: zacala Date: Fri, 9 Oct 2026 05:04:04 +0000 Subject: [PATCH 28/29] Document the MPMC stall after a crash and warn against mixing 3.0.0 A process that dies between claiming a slot of ConcurrentQueue or ConcurrentMessageQueue and publishing it leaves the slot claimed for good. Nothing can recover that automatically, so the README (Recovering after a crash) and the XML docs of both classes say so and name Remove as the way out. SingleProducerQueue and SingleProducerByteStream are not affected. 3.0.0 takes the write lock from every live owner on Linux, including one that runs a later version. The README and the CHANGELOG tell users to update all processes that share a region together. --- CHANGELOG.md | 4 ++++ InterprocessMemory/ConcurrentMessageQueue.cs | 10 ++++++++++ InterprocessMemory/ConcurrentQueue.cs | 6 ++++++ README.md | 13 +++++++++++++ 4 files changed, 33 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 7920121..f445235 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,10 @@ Changes since the 3.0.0 release. Migrating from 2.x: see [MIGRATION.md](MIGRATION.md). +**Do not mix 3.0.0 with a later version in processes that share a region on Linux.** 3.0.0 compares +`Process.StartTime`, which differs between observers, so it takes the write lock away from every live owner, +including one that runs a later version. Update all processes that use a region together. + ## Unreleased ### Behavior changes diff --git a/InterprocessMemory/ConcurrentMessageQueue.cs b/InterprocessMemory/ConcurrentMessageQueue.cs index 2ef9614..132de15 100644 --- a/InterprocessMemory/ConcurrentMessageQueue.cs +++ b/InterprocessMemory/ConcurrentMessageQueue.cs @@ -13,6 +13,12 @@ namespace InterprocessMemory /// Thread-safe for concurrent access from multiple writers and readers. /// Uses sequence numbers for coordination instead of simple head/tail pointers. /// Cross-platform: backed by which supports Windows and Linux. + /// + /// A process that dies inside TryEnqueue or TryDequeue, after it claimed a slot and before it + /// published or released it, leaves that slot claimed for good: consumers see an empty queue, or producers a + /// full one, although other slots hold messages. There is no automatic recovery; stop all users and call + /// , which discards the queued messages. + /// /// public sealed unsafe class ConcurrentMessageQueue : IDisposable { @@ -437,6 +443,10 @@ private bool TryEnqueueCore(ReadOnlySpan data, bool countFailure) /// is thrown WITHOUT consuming the message — caller can /// retry with a larger buffer. Use to size the destination safely. /// + /// + /// Whether an empty queue is counted in the statistics. The timeout overload polls with false and counts + /// once when it gives up. + /// /// Number of bytes read, or 0 if buffer is empty /// /// Thrown when the next message does not fit in . The slot is diff --git a/InterprocessMemory/ConcurrentQueue.cs b/InterprocessMemory/ConcurrentQueue.cs index c829a7b..743b2cd 100644 --- a/InterprocessMemory/ConcurrentQueue.cs +++ b/InterprocessMemory/ConcurrentQueue.cs @@ -34,6 +34,12 @@ internal struct ConcurrentQueueSlot /// /// Fixed-size lock-free queue for multiple producers and multiple consumers across processes. + /// + /// A process that dies inside or , after it + /// claimed a slot and before it published or released it, leaves that slot claimed for good: consumers see an + /// empty queue, or producers a full one, although other slots hold data. There is no automatic recovery; + /// stop all users and call , which discards the queued items. + /// /// public sealed unsafe class ConcurrentQueue : IDisposable where T : unmanaged { diff --git a/README.md b/README.md index 4e5f6bd..e4806f7 100644 --- a/README.md +++ b/README.md @@ -299,6 +299,11 @@ modifying the existing bytes. Version 3 does not migrate live 2.x regions. Stop every 2.x process, remove the named/file-backed region, and recreate it with version 3. See [MIGRATION.md](MIGRATION.md). +Do not run 3.0.0 and a later version in processes that share a region on Linux. 3.0.0 compares the +owner's `Process.StartTime`, which differs between observers, so it takes the write lock away from every +live owner (including one of the later version, whose recorded start tick it can never match). Update all +processes that use a region together. See [CHANGELOG.md](CHANGELOG.md). + ## Disposing while other threads are running Stop and join every thread that uses an instance before you dispose it. The lock-free members @@ -327,6 +332,14 @@ Windows named sections disappear when their last handle closes. On Linux a regio every writer times out. Read locks have no owner, so this cannot be detected automatically. `GetLockOwnerInfo().ReaderCount` shows the stale count; once no process is inside a critical section, call `MemoryRegion.ForceResetLocks()` (or the same method on `StructuredMemory` or `SharedArray`). +- A process that dies **inside** `TryEnqueue` or `TryDequeue` of `ConcurrentQueue` or + `ConcurrentMessageQueue`, after it has claimed a slot and before it has published or released it, leaves + that slot claimed for good. Consumers then see an empty queue (or, after a full turn of the ring, producers + see a full one) although the other slots hold data, and nothing can tell the slot is stale. The window is a + few nanoseconds wide, but there is no automatic recovery: stop all users and call + `MemoryRegion.Remove(name)`, which discards the queued items. `SingleProducerQueue` and + `SingleProducerByteStream` publish with one store and are not affected: a restarted producer or consumer + simply carries on. - A creator that dies during initialization makes every opener time out. A region with a different capacity or element type than the one you now want is rejected as well. In both cases stop all users and call `MemoryRegion.Remove(name)` (pass the same `MemoryRegionOptions` for file-backed From c36dedecba5d21393675a66b35fa9651966c5060 Mon Sep 17 00:00:00 2001 From: zacala Date: Fri, 9 Oct 2026 05:09:21 +0000 Subject: [PATCH 29/29] Keep the Fill tests and Fill itself off the stack on a 1 MiB thread The Fill/Clear test for 100 KB elements overflowed the stack on Windows and took the whole test host with it (380 tests had run, the rest were lost). Every temporary copy of a 100 KB struct is 100 KB of stack and the test frame held twelve of them, one per array[i].Tag expression and Fill argument, more than the 1 MiB of a Windows thread. Linux threads have 8 MiB, which is why nothing showed there. Reproduced locally with ulimit -s 1024. The test uses the smallest size that a managed array cannot hold (64 KiB) and reads and fills through NoInlining helpers, so no frame holds more than one or two copies. SharedArray passes the fill value by reference from Fill down to the copy instead of copying it into every frame. --- InterprocessMemory.Tests/SharedArrayTests.cs | 47 +++++++++++++------- InterprocessMemory/SharedArray.cs | 18 ++++---- 2 files changed, 40 insertions(+), 25 deletions(-) diff --git a/InterprocessMemory.Tests/SharedArrayTests.cs b/InterprocessMemory.Tests/SharedArrayTests.cs index 6c3ac09..a7e347b 100644 --- a/InterprocessMemory.Tests/SharedArrayTests.cs +++ b/InterprocessMemory.Tests/SharedArrayTests.cs @@ -393,34 +393,47 @@ public void Dispose_WithAnOpenGuardOnThisThread_DoesNotLeaveTheCrossProcessLockH guard.Dispose(); // a guard that outlives its array is harmless } - // Larger than a managed array may hold per element (64 KiB), and a mid-sized one that is batched. - [System.Runtime.InteropServices.StructLayout(System.Runtime.InteropServices.LayoutKind.Sequential, Size = 100_000)] - public struct Big100K { public int Tag; } + // The smallest size that a managed array cannot hold per element (64 KiB), and a mid-sized one that is + // batched. Every temporary copy of the first one is 64 KiB of stack, and a thread of 1 MiB (Windows) does not + // have many: the tests below read elements through NoInlining helpers instead of `array[i].Tag`, which would + // leave a temporary per expression in the test's own frame (twelve of them overflowed the stack on Windows). + [System.Runtime.InteropServices.StructLayout(System.Runtime.InteropServices.LayoutKind.Sequential, Size = 65_536)] + public struct Big64K { public int Tag; } [System.Runtime.InteropServices.StructLayout(System.Runtime.InteropServices.LayoutKind.Sequential, Size = 20_000)] public struct Mid20K { public int Tag; } + [System.Runtime.CompilerServices.MethodImpl(System.Runtime.CompilerServices.MethodImplOptions.NoInlining)] + private static int TagOf(SharedArray array, int index) => array[index].Tag; + + [System.Runtime.CompilerServices.MethodImpl(System.Runtime.CompilerServices.MethodImplOptions.NoInlining)] + private static int TagOf(SharedArray array, int index) => array[index].Tag; + + [System.Runtime.CompilerServices.MethodImpl(System.Runtime.CompilerServices.MethodImplOptions.NoInlining)] + private static void FillWith(SharedArray array, int tag, int startIndex = 0, int count = -1) => + array.Fill(new Big64K { Tag = tag }, startIndex, count); + [Test, Timeout(60000)] public void FillAndClear_ElementsOf64KiBOrMore_Work() { // Fill's staging buffer was a T[] and an ArrayPool rental. A managed array cannot hold elements of // 64 KiB or more, so the method failed to load with a TypeLoadException before writing anything. - using var array = SharedArray.CreateOrOpen(LockName("Fill100K"), 3); + using var array = SharedArray.CreateOrOpen(LockName("Fill64K"), 3); - array.Fill(new Big100K { Tag = 7 }); - Assert.That(array[0].Tag, Is.EqualTo(7)); - Assert.That(array[1].Tag, Is.EqualTo(7)); - Assert.That(array[2].Tag, Is.EqualTo(7)); + FillWith(array, 7); + Assert.That(TagOf(array, 0), Is.EqualTo(7)); + Assert.That(TagOf(array, 1), Is.EqualTo(7)); + Assert.That(TagOf(array, 2), Is.EqualTo(7)); - array.Fill(new Big100K { Tag = 9 }, 1, 1); - Assert.That(array[0].Tag, Is.EqualTo(7)); - Assert.That(array[1].Tag, Is.EqualTo(9)); - Assert.That(array[2].Tag, Is.EqualTo(7)); + FillWith(array, 9, 1, 1); + Assert.That(TagOf(array, 0), Is.EqualTo(7)); + Assert.That(TagOf(array, 1), Is.EqualTo(9)); + Assert.That(TagOf(array, 2), Is.EqualTo(7)); array.Clear(); - Assert.That(array[0].Tag, Is.EqualTo(0)); - Assert.That(array[1].Tag, Is.EqualTo(0)); - Assert.That(array[2].Tag, Is.EqualTo(0)); + Assert.That(TagOf(array, 0), Is.EqualTo(0)); + Assert.That(TagOf(array, 1), Is.EqualTo(0)); + Assert.That(TagOf(array, 2), Is.EqualTo(0)); } [Test, Timeout(60000)] @@ -430,10 +443,10 @@ public void FillAndClear_MidSizedElements_AreBatchedWithinTheByteBudget() array.Fill(new Mid20K { Tag = 5 }); for (int i = 0; i < 10; i++) - Assert.That(array[i].Tag, Is.EqualTo(5), $"element {i}"); + Assert.That(TagOf(array, i), Is.EqualTo(5), $"element {i}"); array.Clear(); - Assert.That(array[9].Tag, Is.EqualTo(0)); + Assert.That(TagOf(array, 9), Is.EqualTo(0)); } // Two bytes: the width that used to be copied as a one byte store plus a two byte store. diff --git a/InterprocessMemory/SharedArray.cs b/InterprocessMemory/SharedArray.cs index e033725..678f2d4 100644 --- a/InterprocessMemory/SharedArray.cs +++ b/InterprocessMemory/SharedArray.cs @@ -387,14 +387,14 @@ public void Fill(T value, int startIndex = 0, int count = -1) if (!s_needsLock || IsHoldingWriteLock()) { - FillCore(value, startIndex, count); + FillCore(in value, startIndex, count); return; } LockTicket ticket = EnterWrite(MemoryRegionOptions.DefaultLockTimeout); try { - FillCore(value, startIndex, count); + FillCore(in value, startIndex, count); } finally { @@ -405,27 +405,29 @@ public void Fill(T value, int startIndex = 0, int count = -1) // A batch of this many bytes is the most that is staged at a time. private const int FillBatchBytes = 64 * 1024; - private void FillCore(T value, int startIndex, int count) + // The value is passed by reference down to the copy: a by-value parameter of an element of tens of KiB + // puts a copy of it on the stack in every frame, and a thread of 1 MiB (Windows) does not have many. + private void FillCore(in T value, int startIndex, int count) { // A managed array cannot hold elements of 64 KiB or more, and a method that so much as mentions // T[] or ArrayPool for such a T does not even load (TypeLoadException when it is compiled, // before a single element is written). A batch of such elements would be one or two elements // anyway, so they are copied one by one from a method that has no array in it. if (_elementSize >= FillBatchBytes / 2) - FillOneByOne(value, startIndex, count); + FillOneByOne(in value, startIndex, count); else - FillInBatches(value, startIndex, count); + FillInBatches(in value, startIndex, count); } - private void FillOneByOne(T value, int startIndex, int count) + private void FillOneByOne(in T value, int startIndex, int count) { - ReadOnlySpan one = MemoryMarshal.CreateReadOnlySpan(ref value, 1); + ReadOnlySpan one = MemoryMarshal.CreateReadOnlySpan(ref Unsafe.AsRef(in value), 1); for (int i = 0; i < count; i++) CopyFromCore(startIndex + i, one); } [MethodImpl(MethodImplOptions.NoInlining)] - private void FillInBatches(T value, int startIndex, int count) + private void FillInBatches(in T value, int startIndex, int count) { // Batch fill: create a filled buffer and write in chunks. The batch is bounded in bytes as well as // in elements: 4096 elements of 32 KiB would be a 128 MiB temporary buffer on every call.