Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
29 commits
Select commit Hold shift + click to select a range
48c9cd5
Fix lock ownership on Linux, lock-wait robustness, and generic struct…
vic-py Oct 3, 2026
0b0bb0d
Make time-based lock takeover opt-in and add crash-recovery tools
vic-py Oct 3, 2026
cedf8b4
Narrow the Dispose race with a grace period and document the contract
vic-py Oct 4, 2026
8e88ffc
Fix SharedArray leak on failed open and remove avoidable per-call cost
vic-py Oct 4, 2026
adec015
Remove finalizers that release nothing
vic-py Oct 4, 2026
576273e
Share one power-of-two rounding helper between the ring buffers
vic-py Oct 4, 2026
be510cb
Publish SharedArray and StructuredMemory headers with proper ordering
vic-py Oct 4, 2026
da75f6a
Fix two MPMC stress tests that hang or fail when run on their own
vic-py Oct 4, 2026
19baa34
Add cross-process locking to SharedArray
vic-py Oct 5, 2026
b0b52dd
Add Linux/Windows CI matrix
vic-py Oct 5, 2026
cdafe2d
Bring XML docs and code comments up to date
vic-py Oct 5, 2026
93dd60d
Update README and MIGRATION, add CHANGELOG
vic-py Oct 5, 2026
edb3742
Run the 64-reader/4-writer stress test on dedicated threads
vic-py Oct 5, 2026
b896dc5
Always run the timing-sensitive CI step
vic-py Oct 5, 2026
3805dba
Read and write 1, 2, 4 and 8 byte elements with one typed access
vic-py Oct 6, 2026
1056b8a
Recover write locks that have no owner, and let readers recover dead …
vic-py Oct 6, 2026
67334f8
Give the MPMC queues at least two slots
vic-py Oct 6, 2026
0604f84
Check the free space before creating a region
vic-py Oct 6, 2026
8096097
Do not decide from a pid that the lock owner is gone across PID names…
vic-py Oct 6, 2026
4e4123a
Fix three small defects found in review
vic-py Oct 6, 2026
c5360c0
Make the cross-process test harness fail loudly and stop leaking chil…
vic-py Oct 6, 2026
e1dfbdd
Run spinning test loops on dedicated threads
vic-py Oct 6, 2026
0805e03
Show failures of the informational CI step and keep main runs
vic-py Oct 6, 2026
ad9583f
Fix region creation races and two lifetime defects
vic-py Oct 9, 2026
5da9d64
Make the lock guards refuse copies and out-of-order release
vic-py Oct 9, 2026
1017f04
Check the schema version before the size of a structured-memory region
vic-py Oct 9, 2026
407d7af
Let SharedArray.Fill and Clear handle elements of 64 KiB or more
vic-py Oct 9, 2026
5280430
Document the MPMC stall after a crash and warn against mixing 3.0.0
vic-py Oct 9, 2026
c36dede
Keep the Fill tests and Fill itself off the stack on a 1 MiB thread
vic-py Oct 9, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
111 changes: 111 additions & 0 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,111 @@
name: CI

on:
push:
branches: [main]
pull_request:
workflow_dispatch:

permissions:
contents: read

# A newer push to the same pull request supersedes the run in progress. Runs for main are never cancelled,
# so that every commit that lands there keeps its own result.
concurrency:
group: ci-${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: ${{ github.event_name == 'pull_request' }}

env:
DOTNET_NOLOGO: true
DOTNET_CLI_TELEMETRY_OPTOUT: true

jobs:
test:
name: Build and test (${{ matrix.os }})
runs-on: ${{ matrix.os }}
timeout-minutes: 30

strategy:
# Report both platforms even when one fails: the library keeps its state differently on each
# (Windows named sections vanish with their last handle, Linux files in /dev/shm persist).
fail-fast: false
matrix:
os: [ubuntu-latest, windows-latest]

defaults:
run:
# One shell on both platforms keeps quoting of the filter expressions identical.
shell: bash

steps:
- name: Check out
uses: actions/checkout@v4

- name: Set up .NET
uses: actions/setup-dotnet@v4
with:
dotnet-version: 8.0.x

- name: Cache NuGet packages
uses: actions/cache@v4
with:
path: ~/.nuget/packages
key: nuget-${{ runner.os }}-${{ hashFiles('**/*.csproj') }}
restore-keys: nuget-${{ runner.os }}-

- name: Restore
run: dotnet restore InterprocessMemory.sln

# Builds every project, including the benchmarks and the child-process test worker that the
# cross-process tests start. Packing on every build is not needed here.
- name: Build
id: build
run: dotnet build InterprocessMemory.sln --configuration Release --no-restore -p:GeneratePackageOnBuild=false

# The blame collector (enabled by --blame-hang-timeout) names the running test when the test host
# dies, which is what an AccessViolationException from a memory-mapped region looks like; the hang
# timeout stops a stuck test from holding the runner until the job timeout.
- name: Test
run: >
dotnet test InterprocessMemory.Tests/InterprocessMemory.Tests.csproj
--configuration Release --no-build
--filter "Category!=TimingSensitive&Category!=LongRunning"
--blame-hang-timeout 10m --blame-hang-dump-type none
--logger "trx;LogFileName=test-results.trx"
--results-directory TestResults

# Tests marked LongRunning (minutes each) are explicit-only and are left to manual runs.
#
# Tests whose outcome depends on the core count and the scheduler. They still run and show up
# in the log and the artifact, but a failure here does not fail the job.
- name: Test (timing sensitive, informational)
id: timing
if: ${{ !cancelled() && steps.build.conclusion == 'success' }}
continue-on-error: true
run: >
dotnet test InterprocessMemory.Tests/InterprocessMemory.Tests.csproj
--configuration Release --no-build
--filter "Category=TimingSensitive"
--blame-hang-timeout 5m --blame-hang-dump-type none
--logger "trx;LogFileName=timing-sensitive-results.trx"
--results-directory TestResults

# continue-on-error shows a failed step as a success, so say it out loud: a warning annotation on the
# run and a section in its summary. Otherwise a test that fails on every run looks green for ever.
- name: Report timing sensitive failures
if: ${{ !cancelled() && steps.timing.outcome == 'failure' }}
run: |
echo "::warning title=Timing sensitive tests failed on ${{ matrix.os }}::This step does not fail the job. See the step log and timing-sensitive-results.trx in the test-results artifact."
{
echo "### Timing sensitive tests failed on ${{ matrix.os }}"
echo ""
echo "These tests depend on scheduling and do not fail the job, but a failure is still a result to read: see the log of the step above and the test-results artifact."
} >> "$GITHUB_STEP_SUMMARY"

- name: Upload test results
if: always()
uses: actions/upload-artifact@v4
with:
name: test-results-${{ matrix.os }}
path: TestResults
if-no-files-found: ignore
119 changes: 119 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,119 @@
# Changelog

Changes since the 3.0.0 release. Migrating from 2.x: see [MIGRATION.md](MIGRATION.md).

**Do not mix 3.0.0 with a later version in processes that share a region on Linux.** 3.0.0 compares
`Process.StartTime`, which differs between observers, so it takes the write lock away from every live owner,
including one that runs a later version. Update all processes that use a region together.

## Unreleased

### Behavior changes

- `MemoryRegion.ReleaseWriteLock` throws `SynchronizationLockException` when the calling thread does
not own the lock (it used to be ignored). The `StructuredMemory<T>` lock guards do the same when they
are disposed on another thread, which is what an `await` inside a lock scope causes.
- `MemoryRegionOptions.OrphanLockTimeout` defaults to `TimeSpan.Zero` (disabled) instead of 30 seconds.
A lock whose owner process has exited is still recovered by default; a lock held by a live process is
taken over only if you set a timeout. `DefaultOrphanLockTimeout` keeps its value as a suggestion.
- `Dispose()` of a `MemoryRegion`, and of every container that owns one, takes at least
`MemoryRegion.DisposeGracePeriod` (10 ms) longer.
- `SharedArray<T>` takes the shared region lock per element access for element sizes other than
1, 2, 4 and 8 bytes, so values are no longer torn. These accesses are slower; take the lock once with
`AcquireReadLock()` / `AcquireWriteLock()` for bulk work.
- An out-of-range queue `capacity` is reported with the parameter name `capacity`.

### Fixed

- Linux: a live write-lock owner was reported as an orphan and its lock was taken over, because
`Process.StartTime` differs between observers. The owner is now identified by its `/proc/<pid>/stat`
start tick count.
- Waiting for a lock with `Timeout.InfiniteTimeSpan` now recovers from an owner that dies later
(the owner is probed every 250 ms).
- A write lock whose owner was killed between taking the lock and recording its pid (or between clearing the
pid and releasing the lock) stayed held forever, because the orphan check needs a pid. A waiter now
clears a lock that has had no owner for two seconds.
- Linux: processes in different PID namespaces that share `/dev/shm` (containers) took each other's locks, because a
waiter looked the owner's pid up in its own namespace and did not find it. The owner now records its PID
namespace (the inode of `/proc/self/ns/pid`, in the reserved part of the header) and a waiter in another
namespace no longer declares the owner dead from its pid.
- `StructuredMemory<T>.OpenExisting` checked the size of the region before the schema version, so no
`SchemaCompatibility` mode could open a region of another size (`Forward` and `Full` only worked when an
appended field fitted in the padding) and `Strict` reported a size mismatch instead of the version. The
version is checked first now; for different versions the region must only be at least as large as the
schema, so an older schema can read a larger region written by a newer one. The same version still needs
the exact size. See "Schema versions" in the README.
- `SharedArray<T>.Fill` and `Clear` threw `TypeLoadException` for elements of 64 KiB or more (a managed array
cannot hold them, and the staging buffer was a `T[]`) and staged up to 4096 elements per batch whatever their
size, 128 MiB for 32 KiB elements. Large elements are written one by one, and a batch is limited to 64 KiB.
- The lock guards of `StructuredMemory<T>` and `SharedArray<T>` now remember the depth at which they were taken.
Disposing a copy of a guard a second time used to decrement the thread's depth again, so a thread inside an
outer lock believed it held none and tried to take the lock it already held; releasing a write guard before
a read guard taken inside it removed the protection under that read guard. Both throw
`SynchronizationLockException` now, before anything changes, and the guard stays valid.
- Disposing a `StructuredMemory<T>` or `SharedArray<T>` while one of its guards was open on the calling thread
left the cross-process lock held until the process ended (its owner was alive, so nobody recovered it). It
releases that lock now; a guard that outlives its instance can still be disposed.
- `StructuredMemory<T>.AcquireWriteLock()` called while the thread holds a read guard set the writer flag and
waited for that thread's own read lock, blocking every other process until the timeout. It now throws
`InvalidOperationException` at once, like the automatic write lock and `SharedArray<T>` do.
- `SingleProducerByteStream.Available` and `.Used` read the unmapped header after `Dispose()`; they throw
`ObjectDisposedException` like the other members.
- The timeout overloads of `ConcurrentQueue<T>` and `ConcurrentMessageQueue` counted a failed enqueue or
dequeue on every poll (about 1,800 for 4 s of waiting). A call now counts once, when it gives up, and
not at all when it succeeds after waiting. They also no longer allocate a `Stopwatch` per call.
- `OpenExisting` racing the process that creates the region reported "invalid header" or "empty" about once in
eleven races instead of waiting the few microseconds until the creator had written the header. It waits up
to two seconds for a region that is still being created.
- Linux: two processes calling `CreateOrOpen` for a new name at the same moment both became the creator, and
the one that then failed (another capacity or region kind) deleted the file the other was using, so later
openers got a second, separate region. `FileMode.CreateNew` decides who creates and sizes the file, and
only that process removes it when construction fails.
- `MemoryRegionOptions.FilePath`: an existing file of another size was grown to the requested capacity before
anything checked it, which left it permanently resized and could never be undone. It is now rejected
without being modified (an older format or a foreign file is reported as such), and the file is opened
with sharing so that a second process can map it.
- `MemoryRegion.GetMemory` returned a `Memory<byte>` that did not keep the region reachable; a caller that
dropped the region and kept the memory let the finalizer unmap it, which ended the process.
- Linux: a lock owner that had been killed but not yet reaped by its parent (a zombie) counted as alive. It is
recognised as gone now.
- A waiting reader now recovers a write lock whose owner process died, like a waiting writer does. It used
to wait for its whole timeout.
- Disposing a region while another thread waits for one of its locks no longer crashes the process;
the waiter fails with `ObjectDisposedException`.
- `ConcurrentQueue<T>` and `ConcurrentMessageQueue` with a capacity of 1 overwrote the stored item when a second
one was enqueued and then never delivered anything again. They need two slots, so a requested capacity
of 1 is now raised to 2, and an existing region that stored a capacity of 1 is rejected as invalid
(remove it with `MemoryRegion.Remove`).
- Linux: creating a region larger than the free space of `/dev/shm` (Docker's default is 64 MB) succeeded and
the process was killed with an uncatchable `SIGBUS` at the first write that did not fit. `CreateOrOpen`
now throws `IOException` up front; a file-backed region (`MemoryRegionOptions.FilePath`) is checked too.
- Generic unmanaged structs (`ValueTuple`, `KeyValuePair<,>`) work in all typed containers.
Fingerprints of types that already worked are unchanged, so existing regions stay compatible.
- Two byte `SharedArray<T>` elements could be read half written by another process (a two byte copy is a
one byte store plus a two byte store). Elements of 1, 2, 4 and 8 bytes are now read and written with one
typed load/store. `StructuredMemory<T>` had the same defect for 2 byte scalars and, because it only locked
values wider than 8 bytes, also for 3, 5, 6 and 7 byte values and for small arrays: only 1, 2, 4 and 8
byte scalars are lock-free now, everything else (including every array) takes the shared lock.
- `SharedArray<T>` disposes its region when opening fails because of a different element type or length.
- `SharedArray<T>` and `StructuredMemory<T>` publish their header magic after the other fields, which
prevents a spurious format error on weakly ordered CPUs.
- `Dispose()` racing a call that is still running on another thread is mitigated by
`DisposeGracePeriod`. This is best effort; stop and join threads before disposing.

### Added

- `MemoryRegion.Remove(name, options)` deletes the backing storage of a region left unusable by a crash.
- `MemoryRegion.ForceResetLocks()` and `StructuredMemory<T>.ForceResetLocks()` /
`SharedArray<T>.ForceResetLocks()` clear lock state left behind by a crashed process.
- `LockOwnerInfo.ReaderCount` for diagnosing a stale reader count.
- `MemoryRegion.DisposeGracePeriod` (static, process-wide).
- `SharedArray<T>.AcquireReadLock()` / `AcquireWriteLock()` with timeout overloads, returning
reentrant `ref struct` guards.
- GitHub Actions workflow that builds and tests on Linux and Windows.

### Removed

- Finalizers on `ConcurrentMessageQueue`, `SingleProducerByteStream`, `SharedArray<T>` and
`StructuredMemory<T>`. The `MemoryRegion` finalizer still unmaps the memory when `Dispose` is never
called.
99 changes: 99 additions & 0 deletions InterprocessMemory.TestWorker/Program.cs
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,11 @@
/// concurrent_producer &lt;name&gt; &lt;producerId&gt; — enqueue 1000 unique integers
/// try_write_lock &lt;name&gt; — try the cross-process write lock for 250 ms
/// orphan_write_lock &lt;name&gt; — acquire a write lock and exit without releasing it
/// dispose_busy_poll &lt;prefix&gt; — 200 rounds of: pollers spin on a ConcurrentQueue while it is disposed
/// hold_write_lock &lt;name&gt; — acquire a write lock, print "holding", and keep it until killed or until
/// the parent closes our standard input (the parent died)
/// hold_read_lock &lt;name&gt; — acquire a read lock, print "holding", and keep it until killed or until
/// the parent closes our standard input (the parent died)
/// </summary>
if (args.Length < 2)
{
Expand All @@ -43,6 +48,9 @@
args.Length >= 3 ? int.Parse(args[2]) : 0),
"try_write_lock" => TryWriteLock(bufferName),
"orphan_write_lock" => OrphanWriteLock(bufferName),
"dispose_busy_poll" => DisposeBusyPoll(bufferName),
"hold_write_lock" => HoldWriteLock(bufferName),
"hold_read_lock" => HoldReadLock(bufferName),
_ => Error($"Unknown role: {role}")
};

Expand Down Expand Up @@ -197,6 +205,97 @@ static int OrphanWriteLock(string name)
return 0; // Deliberately skip Dispose/Release; process teardown closes only the mapping handle.
}

// The common shutdown pattern: consumers spin on TryDequeue while another thread disposes the queue. Without
// MemoryRegion.DisposeGracePeriod this killed the process with an AccessViolationException within a few hundred
// rounds. It is a mitigation, not a proof: a thread that is descheduled for longer than the grace period at the
// wrong moment can still fail, and that kills this process, which is why it runs here and not in the test host.
static int DisposeBusyPoll(string namePrefix)
{
var random = new Random(42);
int pollerCount = Math.Max(3, Environment.ProcessorCount - 1);

for (int round = 0; round < 200; round++)
{
string name = $"{namePrefix}_{round}";
var queue = InterprocessMemory.ConcurrentQueue<long>.CreateOrOpen(name, 64);
var pollers = new Thread[pollerCount];
for (int i = 0; i < pollers.Length; i++)
{
pollers[i] = new Thread(() =>
{
try
{
while (true)
{
queue.TryDequeue(out _);
queue.TryEnqueue(1);
}
}
catch (ObjectDisposedException)
{
// Expected: the queue was disposed under the poller.
}
}) { IsBackground = true };
pollers[i].Start();
}

Thread.Sleep(random.Next(0, 3));
queue.Dispose();

foreach (Thread poller in pollers)
{
if (!poller.Join(TimeSpan.FromSeconds(10)))
return Error($"round {round}: a poller did not stop");
}

MemoryRegion.Remove(name);
}

Console.WriteLine("ok");
return 0;
}

static int HoldWriteLock(string name)
{
using var region = MemoryRegion.OpenExisting(name);
if (!region.TryAcquireWriteLock(TimeSpan.FromSeconds(5)))
return Error("failed to acquire held test lock");

// The parent reads this line, probes the lock while we are alive, and then kills us
// to simulate a crash while the lock is held.
Console.WriteLine("holding");
Console.Out.Flush();
WaitUntilTheParentIsGone();
return 0;
}

static int HoldReadLock(string name)
{
using var region = MemoryRegion.OpenExisting(name);
if (!region.TryAcquireReadLock(TimeSpan.FromSeconds(5)))
return Error("failed to acquire held test read lock");

Console.WriteLine("holding");
Console.Out.Flush();
WaitUntilTheParentIsGone();
return 0;
}

// The test kills us in the normal case. If it dies first (crash, hang timeout, Ctrl-C) the pipe that is our
// standard input closes, which ends this wait instead of holding the lock, and the process, for ever.
// The upper bound covers a parent that hangs without dying.
static void WaitUntilTheParentIsGone()
{
try
{
System.Threading.Tasks.Task.Run(() => Console.In.ReadToEnd()).Wait(TimeSpan.FromMinutes(2));
}
catch (Exception)
{
// Nothing to wait on (no stdin): exit.
}
}

static int Error(string msg)
{
Console.Error.WriteLine(msg);
Expand Down
Loading
Loading