From cf2456029c8010b57fa535544b36d7be1445aafa Mon Sep 17 00:00:00 2001 From: Zacgoose <107489668+Zacgoose@users.noreply.github.com> Date: Mon, 5 Oct 2026 16:12:09 +0800 Subject: [PATCH 01/24] fix(orchestrator): bound re-drive verifications to one per run and eight at once --- Services/Orchestration/OrchestratorService.cs | 39 ++++- .../OrchestratorRedriveGuardTests.cs | 163 ++++++++++++++++++ .../OrchestratorSequentialTests.cs | 2 + .../OrchestratorStaleRunningTests.cs | 2 + tests/Craft.Tests/RunRemainingCounterTests.cs | 4 + 5 files changed, 208 insertions(+), 2 deletions(-) create mode 100644 tests/Craft.Tests/OrchestratorRedriveGuardTests.cs diff --git a/Services/Orchestration/OrchestratorService.cs b/Services/Orchestration/OrchestratorService.cs index bc7843f..f5a6edb 100644 --- a/Services/Orchestration/OrchestratorService.cs +++ b/Services/Orchestration/OrchestratorService.cs @@ -25,7 +25,7 @@ namespace Craft.Orchestration; /// 6. After 3 interruptions (host crash/reboot), a task is marked Failed /// 7. PostExecStatus tracks PostExecution lifecycle for crash resilience /// -public class OrchestratorService : IJobDescriptorStateWriter +public class OrchestratorService : IJobDescriptorStateWriter, IDisposable { internal readonly ILogger _logger; private readonly PowerShellRunnerService _psRunner; @@ -80,6 +80,20 @@ public class OrchestratorService : IJobDescriptorStateWriter /// private readonly ConcurrentDictionary _redriveBackoff = new(); + /// Runs with a re-drive verification in flight; a tick skips them instead of starting another. + private readonly ConcurrentDictionary _redriveInFlight = new(); + + /// Caps concurrent re-drive verifications across all runs; a tick that finds none free skips the run. + private readonly SemaphoreSlim _redriveSlots = new(RedriveMaxConcurrent, RedriveMaxConcurrent); + + private const int RedriveMaxConcurrent = 8; + + public void Dispose() + { + _redriveSlots.Dispose(); + GC.SuppressFinalize(this); + } + /// Per-run status/re-drive tick cadence, from Orchestrator:StatusTimerIntervalSeconds. private readonly TimeSpan _statusInterval; @@ -1807,9 +1821,16 @@ private async Task RedrivePendingTasksAsync(OrchestratorRun run) // verifies each candidate against the queue table (one point read apiece; the candidate set is // small), returning only tasks the pump can actually still claim. Anything else is a ghost to // re-enqueue. + // The sweep starts this for every live run each tick without awaiting it, so unguarded a slow + // verification overlapped the next tick's for the same run and the reads piled up in-process. + if (!_redriveInFlight.TryAdd(run.Name, 0)) return; + HashSet dispatchable; + var holdsSlot = false; try { + await _redriveSlots.WaitAsync(); + holdsSlot = true; Interlocked.Increment(ref _redriveStorageReads); dispatchable = await _queue.GetDispatchableTaskIdsAsync( run.Name, candidates.Select(t => t.Id).ToList()); @@ -1821,8 +1842,22 @@ private async Task RedrivePendingTasksAsync(OrchestratorRun run) _logger.LogWarning(ex, "[Scheduler] Could not read queued tasks for {Run} — skipping re-drive", run.Name); return; } + finally + { + if (holdsSlot) _redriveSlots.Release(); + _redriveInFlight.TryRemove(run.Name, out _); + } - var orphaned = candidates.Where(t => !dispatchable.Contains(t.Id)).ToList(); + // Re-check under the lock: a candidate can be claimed, run and finish while waiting for a slot or the + // read, and its row is then gone for the right reason. + List orphaned; + lock (_lock) + { + orphaned = candidates + .Where(t => !dispatchable.Contains(t.Id)) + .Where(t => t.Status == "Pending" && !_jobManager.IsQueuedOrRunning($"{run.Name}-{t.Id}")) + .ToList(); + } if (orphaned.Count == 0) { // Verified clean: grow the interval (double, capped) so this run's next storage read is further diff --git a/tests/Craft.Tests/OrchestratorRedriveGuardTests.cs b/tests/Craft.Tests/OrchestratorRedriveGuardTests.cs new file mode 100644 index 0000000..57c1221 --- /dev/null +++ b/tests/Craft.Tests/OrchestratorRedriveGuardTests.cs @@ -0,0 +1,163 @@ +using System.Collections.Concurrent; +using System.Reflection; +using Craft.Configuration; +using Craft.Orchestration; +using Craft.Storage; +using Microsoft.Extensions.Logging.Abstractions; + +namespace Craft.Tests; + +/// +/// Pins the re-drive watchdog's bounds: the status sweep starts it for every live run each tick without +/// awaiting it, so it must never run two verifications for one run, never more than a few at once across +/// runs, and never re-drive a candidate that finished while its verification was waiting. +/// +public class OrchestratorRedriveGuardTests +{ + private sealed record Harness(OrchestratorService Svc, JobQueueStore Queue, + RunRemainingCounterTests.ConditionalStore Backing, string IndexTable); + + private static async Task NewHarnessAsync() + { + var settings = new CraftSettings { Orchestrator = { TablePrefix = "rdg" + Guid.NewGuid().ToString("N")[..8] } }; + var backing = new RunRemainingCounterTests.ConditionalStore(); + var queue = new JobQueueStore(NullLogger.Instance, settings, backing); + await queue.InitializeAsync(); + + var svc = (OrchestratorService)System.Runtime.CompilerServices.RuntimeHelpers + .GetUninitializedObject(typeof(OrchestratorService)); + Set(svc, "_logger", NullLogger.Instance); + Set(svc, "_queue", queue); + Set(svc, "_settings", settings); + Set(svc, "_lock", new object()); + Set(svc, "_activeSequentialDrivers", new ConcurrentDictionary()); + Set(svc, "_requeueFailures", new ConcurrentDictionary()); + Set(svc, "_deferrals", NewFieldValue("_deferrals")); + Set(svc, "_redriveBackoff", NewFieldValue("_redriveBackoff")); + Set(svc, "_redriveInFlight", NewFieldValue("_redriveInFlight")); + Set(svc, "_redriveSlots", new SemaphoreSlim(8, 8)); + Set(svc, "_redriveBackoffEnabled", false); + Set(svc, "_redriveBase", TimeSpan.FromSeconds(60)); + var jm = (JobManager)System.Runtime.CompilerServices.RuntimeHelpers.GetUninitializedObject(typeof(JobManager)); + var jobsField = typeof(JobManager).GetField("_jobs", BindingFlags.NonPublic | BindingFlags.Instance)!; + jobsField.SetValue(jm, Activator.CreateInstance(jobsField.FieldType)); + Set(svc, "_jobManager", jm); + return new Harness(svc, queue, backing, $"{settings.Orchestrator.TablePrefix}QueueIndex"); + } + + private static object NewFieldValue(string field) => + Activator.CreateInstance(typeof(OrchestratorService) + .GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)!.FieldType)!; + + private static void Set(object target, string field, object? value) => + typeof(OrchestratorService).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)! + .SetValue(target, value); + + private static Task Redrive(Harness h, OrchestratorRun run) => + (Task)typeof(OrchestratorService).GetMethod("RedrivePendingTasksAsync", BindingFlags.NonPublic | BindingFlags.Instance)! + .Invoke(h.Svc, [run])!; + + private static OrchestratorRun MakeRun(string name, int count) => new() + { + Name = name, + Status = "Running", + Priority = 4, + StartedUtc = DateTime.UtcNow, + Tasks = Enumerable.Range(0, count) + .Select(i => new OrchestratorTaskItem { Id = $"{name}_t{i}", Status = "Pending" }).ToList() + }; + + /// Holds every index read open until , counting how many are held at once. + private sealed class ReadGate + { + private readonly TaskCompletionSource _open = new(TaskCreationOptions.RunContinuationsAsynchronously); + private int _held; + public int Entered; + public int MaxHeld; + + public ReadGate(Harness h) => h.Backing.OnPartitionQuery = async table => + { + if (table != h.IndexTable) return; + Interlocked.Increment(ref Entered); + var held = Interlocked.Increment(ref _held); + int seen; + while (held > (seen = Volatile.Read(ref MaxHeld)) && Interlocked.CompareExchange(ref MaxHeld, held, seen) != seen) { } + await _open.Task; + Interlocked.Decrement(ref _held); + }; + + public void Release() => _open.TrySetResult(); + + public async Task WaitForEnteredAsync(int count) + { + for (var i = 0; i < 200 && Volatile.Read(ref Entered) < count; i++) await Task.Delay(10); + } + } + + [Fact] + public async Task SecondTick_WhileAVerificationIsInFlight_DoesNotStartAnother() + { + var h = await NewHarnessAsync(); + var run = MakeRun("inflight", 3); + await h.Queue.EnqueueBatchAsync(run.Name, run.Tasks.Select(t => (t.Id, 4)).ToList(), DateTime.UtcNow); + var gate = new ReadGate(h); + + var first = Redrive(h, run); + await gate.WaitForEnteredAsync(1); + var second = Redrive(h, run); + + Assert.True(second.IsCompleted); + gate.Release(); + await first; + Assert.Equal(1, gate.Entered); + } + + [Fact] + public async Task ManyRuns_VerifyAtMostEightAtOnce_AndAllEventuallyVerify() + { + var h = await NewHarnessAsync(); + var runs = Enumerable.Range(0, 20).Select(i => MakeRun($"cap{i:D2}", 2)).ToList(); + foreach (var run in runs) + await h.Queue.EnqueueBatchAsync(run.Name, run.Tasks.Select(t => (t.Id, 4)).ToList(), DateTime.UtcNow); + var gate = new ReadGate(h); + + var ticks = runs.Select(r => Redrive(h, r)).ToList(); + await gate.WaitForEnteredAsync(8); + await Task.Delay(100); + + Assert.Equal(8, gate.Entered); + gate.Release(); + await Task.WhenAll(ticks); + Assert.Equal(20, gate.Entered); + Assert.True(gate.MaxHeld <= 8); + } + + [Fact] + public async Task CandidateThatFinishesDuringTheRead_IsNotReDriven() + { + var h = await NewHarnessAsync(); + var run = MakeRun("finished", 1); + var gate = new ReadGate(h); + + var tick = Redrive(h, run); + await gate.WaitForEnteredAsync(1); + run.Tasks[0].Status = "Completed"; + gate.Release(); + await tick; + await Task.Delay(200); + + Assert.Empty(await h.Queue.GetQueuedTaskIdsAsync(run.Name)); + } + + [Fact] + public async Task OrphanStillPendingAfterTheRead_IsReDriven() + { + var h = await NewHarnessAsync(); + var run = MakeRun("orphan", 1); + + await Redrive(h, run); + for (var i = 0; i < 100 && (await h.Queue.GetQueuedTaskIdsAsync(run.Name)).Count == 0; i++) await Task.Delay(20); + + Assert.Equal(["orphan_t0"], await h.Queue.GetQueuedTaskIdsAsync(run.Name)); + } +} diff --git a/tests/Craft.Tests/OrchestratorSequentialTests.cs b/tests/Craft.Tests/OrchestratorSequentialTests.cs index f5824e2..22ac2e7 100644 --- a/tests/Craft.Tests/OrchestratorSequentialTests.cs +++ b/tests/Craft.Tests/OrchestratorSequentialTests.cs @@ -113,6 +113,8 @@ private static async Task NewHarnessAsync() Set(svc, "_requeueFailures", new ConcurrentDictionary()); Set(svc, "_deferrals", NewFieldDict(svc, "_deferrals")); Set(svc, "_redriveBackoff", NewFieldDict(svc, "_redriveBackoff")); + Set(svc, "_redriveInFlight", NewFieldDict(svc, "_redriveInFlight")); + Set(svc, "_redriveSlots", new SemaphoreSlim(8, 8)); Set(svc, "_shedParameters", false); Set(svc, "_redriveBackoffEnabled", false); // pin the sequential logic, not the backoff timing Set(svc, "_redriveBase", TimeSpan.FromSeconds(60)); diff --git a/tests/Craft.Tests/OrchestratorStaleRunningTests.cs b/tests/Craft.Tests/OrchestratorStaleRunningTests.cs index 2d6c5fb..bfddb21 100644 --- a/tests/Craft.Tests/OrchestratorStaleRunningTests.cs +++ b/tests/Craft.Tests/OrchestratorStaleRunningTests.cs @@ -85,6 +85,8 @@ private static async Task NewHarnessAsync() Set(svc, "_requeueFailures", new ConcurrentDictionary()); Set(svc, "_deferrals", NewFieldValue(svc, "_deferrals")); Set(svc, "_redriveBackoff", NewFieldValue(svc, "_redriveBackoff")); + Set(svc, "_redriveInFlight", NewFieldValue(svc, "_redriveInFlight")); + Set(svc, "_redriveSlots", new SemaphoreSlim(8, 8)); Set(svc, "_shedParameters", false); var jm = (JobManager)System.Runtime.CompilerServices.RuntimeHelpers.GetUninitializedObject(typeof(JobManager)); var jobsField = typeof(JobManager).GetField("_jobs", BindingFlags.NonPublic | BindingFlags.Instance)!; diff --git a/tests/Craft.Tests/RunRemainingCounterTests.cs b/tests/Craft.Tests/RunRemainingCounterTests.cs index 38a287e..2319abe 100644 --- a/tests/Craft.Tests/RunRemainingCounterTests.cs +++ b/tests/Craft.Tests/RunRemainingCounterTests.cs @@ -38,6 +38,9 @@ internal sealed class ConditionalStore : ICraftTableStore /// Set to run at the start of any table query — lets a test model unreachable storage. public Action? OnBeforeQuery { get; set; } + /// Awaited at the start of a partition query, with the table name — lets a test hold a read open. + public Func? OnPartitionQuery { get; set; } + private Dictionary<(string, string), StoreRow> Table(string t) => _tables.TryGetValue(t, out var x) ? x : _tables[t] = new(); @@ -116,6 +119,7 @@ public Task TryReplaceBatchAsync(string table, string partitionKey, IReadO public async IAsyncEnumerable QueryPartitionAsync(string table, string partitionKey, [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) { + if (OnPartitionQuery != null) await OnPartitionQuery(table); foreach (var r in Ordered(table).Where(r => r.PartitionKey == partitionKey)) { yield return r; From 16f687d86309667f8a8cd8882107911d56d99de4 Mon Sep 17 00:00:00 2001 From: Zacgoose <107489668+Zacgoose@users.noreply.github.com> Date: Mon, 5 Oct 2026 16:35:35 +0800 Subject: [PATCH 02/24] perf(storage): find split-entity part rows by RowKey range and only the owning entity's rows --- Services/Storage/AzureTableStore.cs | 15 ++++++-- .../AzureTableStoreLargeEntityTests.cs | 37 +++++++++++++++++++ 2 files changed, 48 insertions(+), 4 deletions(-) diff --git a/Services/Storage/AzureTableStore.cs b/Services/Storage/AzureTableStore.cs index 2bd6a25..476ce3b 100644 --- a/Services/Storage/AzureTableStore.cs +++ b/Services/Storage/AzureTableStore.cs @@ -80,6 +80,7 @@ private TableClient Client(string table) => private TableServiceClient Service => _service ??= new TableServiceClient(_connectionString.Value, _clientOptions); private static readonly string[] select = new[] { "PartitionKey", "RowKey" }; + private static readonly string[] PartLookupProperties = ["PartitionKey", "RowKey", EntitySplitter.OriginalEntityIdKey]; public async Task PingAsync(CancellationToken ct = default) { @@ -623,7 +624,10 @@ private async Task> RecoverMissingPartRowsAsync(string table, var filter = $"{partitionClause} and {BuildRowKeyPrefixClause(entityId)}"; await foreach (var row in Client(table).QueryAsync(filter: filter, cancellationToken: ct)) { - if (seen.Add((row.PartitionKey, row.RowKey))) + // The prefix range also holds unrelated keys ("task1-partner"); only this entity's rows belong. + var owned = row.RowKey == entityId || + (row.TryGetValue(EntitySplitter.OriginalEntityIdKey, out var owner) && owner?.ToString() == entityId); + if (owned && seen.Add((row.PartitionKey, row.RowKey))) rows.Add(row); } } @@ -675,11 +679,14 @@ private async Task> RecoverMissingPartRowsAsync(string table, private async Task RemoveStalePartRowsAsync(string table, string partitionKey, string originalRowKey, HashSet live, CancellationToken ct) { - var filter = $"PartitionKey eq '{Escape(partitionKey)}' and {EntitySplitter.OriginalEntityIdKey} eq '{Escape(originalRowKey)}'"; + // A RowKey range is an index seek; filtering on the marker alone scans the whole partition, which + // every plain delete paid. The range also catches other keys sharing the prefix, so confirm the owner. + var filter = $"PartitionKey eq '{Escape(partitionKey)}' and {BuildRowKeyPrefixClause($"{originalRowKey}-part")}"; var stale = new List(); - await foreach (var row in Client(table).QueryAsync(filter: filter, select: select, cancellationToken: ct)) + await foreach (var row in Client(table).QueryAsync(filter: filter, select: PartLookupProperties, cancellationToken: ct)) { - if (!live.Contains(row.RowKey)) + if (row.TryGetValue(EntitySplitter.OriginalEntityIdKey, out var owner) && owner?.ToString() == originalRowKey + && !live.Contains(row.RowKey)) stale.Add(row.RowKey); } diff --git a/tests/Craft.Tests/AzureTableStoreLargeEntityTests.cs b/tests/Craft.Tests/AzureTableStoreLargeEntityTests.cs index 20a57c4..1338d28 100644 --- a/tests/Craft.Tests/AzureTableStoreLargeEntityTests.cs +++ b/tests/Craft.Tests/AzureTableStoreLargeEntityTests.cs @@ -237,4 +237,41 @@ public async Task Delete_RemovesEveryPartRow_OfASplitEntity() await foreach (var _ in fx.Store.QueryPartitionAsync(fx.Table, "p")) any = true; Assert.False(any); } + + [Fact] + public async Task Delete_TakesOnlyItsOwnPartRows_NotNeighboursSharingThePrefix() + { + await using var fx = await Fixture.TryConnectAsync(); + if (fx == null) return; + + // "task1" splits; "task1-partner" is an unrelated plain row and "task1-part9" an unrelated split + // entity, so both sit inside task1's "-part" key range without belonging to it. + await fx.Store.UpsertAsync(fx.Table, Row("p", "task1", Text(1_200_000))); + await fx.Store.UpsertAsync(fx.Table, Row("p", "task1-partner", "small")); + await fx.Store.UpsertAsync(fx.Table, Row("p", "task1-part9", Text(1_200_000, 'y'))); + + await fx.Store.DeleteAsync(fx.Table, "p", "task1"); + + Assert.Null(await fx.Store.GetAsync(fx.Table, "p", "task1")); + Assert.Equal("small", (await fx.Store.GetAsync(fx.Table, "p", "task1-partner"))!.GetString("ParametersJson")); + Assert.Equal(Text(1_200_000, 'y'), (await fx.Store.GetAsync(fx.Table, "p", "task1-part9"))!.GetString("ParametersJson")); + Assert.True(await fx.PhysicalRowCountAsync("p") > 2); + } + + [Fact] + public async Task FilteredQuery_OnASplitEntity_DoesNotReturnNeighboursSharingThePrefix() + { + await using var fx = await Fixture.TryConnectAsync(); + if (fx == null) return; + + await fx.Store.UpsertAsync(fx.Table, Row("p", "task1", Text(1_200_000))); + await fx.Store.UpsertAsync(fx.Table, Row("p", "task1-partner", "small")); + await fx.Store.UpsertAsync(fx.Table, Row("p", "task1-part9", Text(1_200_000, 'y'))); + + var rows = new List(); + await foreach (var r in fx.Store.QueryTableAsync(fx.Table, "PartitionKey eq 'p' and RowKey eq 'task1'")) rows.Add(r); + + Assert.Equal(["task1"], rows.Select(r => r.RowKey)); + Assert.Equal(Text(1_200_000), rows[0].GetString("ParametersJson")); + } } From 1464c73c02a3739fb333ce977895ef97eaaf526e Mon Sep 17 00:00:00 2001 From: Zacgoose <107489668+Zacgoose@users.noreply.github.com> Date: Mon, 5 Oct 2026 17:30:25 +0800 Subject: [PATCH 03/24] feat(queue): claim oldest run first within a priority band - queue RowKey is {runEpoch}|{run}|{task}; the epoch is stored once per run so re-enqueues keep one row - schema v3 migration re-keys an existing backlog, writing partitions in parallel --- Services/Storage/JobQueueStore.cs | 117 ++++++++--- .../Craft.Tests/JobQueueDispatchableTests.cs | 8 +- tests/Craft.Tests/JobQueueFifoTests.cs | 195 ++++++++++++++++++ .../Craft.Tests/JobQueueIndexBackfillTests.cs | 2 +- tests/Craft.Tests/JobQueueStoreTests.cs | 10 +- tests/Craft.Tests/RunRemainingCounterTests.cs | 10 +- tests/Craft.Tests/TableKeyTests.cs | 2 +- 7 files changed, 302 insertions(+), 42 deletions(-) create mode 100644 tests/Craft.Tests/JobQueueFifoTests.cs diff --git a/Services/Storage/JobQueueStore.cs b/Services/Storage/JobQueueStore.cs index 0dd5f93..3d0cc0b 100644 --- a/Services/Storage/JobQueueStore.cs +++ b/Services/Storage/JobQueueStore.cs @@ -12,8 +12,8 @@ namespace Craft.Storage; /// /// Key design, and it is doing real work: /// -/// PartitionKey "P04" — the priority bucket, zero-padded so it sorts numerically. -/// RowKey "{queuedTicks:D19}-{run}-{task}" — time-ordered within a bucket, unique by construction. +/// PartitionKey "P04" — the priority bucket, zero-padded so it sorts numerically. +/// RowKey "{runEpochTicks:D19}|{run}|{task}" — oldest run first within a bucket, one row per task. /// /// Azure Table returns rows ordered by partition key then row key, so a single unfiltered read yields /// the highest-priority, oldest-first work FIRST, across every run. That is the cross-run priority @@ -98,16 +98,26 @@ public Task WaitForWorkAsync(TimeSpan pollInterval, CancellationToken ct = private const string SchemaPartition = "$schema"; private const string SchemaRowKey = "queue-index"; + /// + /// The row in a run's index partition holding the run's queue epoch: the time its first task was queued, + /// in whole seconds. Every queue key for the run is prefixed with it, so a bucket drains oldest run + /// first, and a re-enqueue of a task builds the same key however much later it happens. + /// + private const string EpochRowKey = "$epoch"; + /// /// Current on-disk schema version, applied once per storage account by : /// 1 — the run index exists (see the RUN INDEX note above). /// 2 — queue RowKeys are deterministic per (run, task) — {run}|{task} — instead of /// time-prefixed, so re-dispatching a task UPDATES its row instead of writing a second one /// (the duplicate-execution class). Enqueue time moves to the QueuedUtc property. + /// 3 — the key gains the run's epoch as a prefix — {epoch}|{run}|{task} — so a bucket drains + /// oldest run first instead of alphabetically by run name, which starved late-sorting runs. + /// Still deterministic per (run, task), because the epoch is stored once per run. /// A single forward migration takes any older account straight to this version; there is no /// backward-compatible dual-read — after the migration only the new key scheme is used. /// - private const int SchemaVersion = 2; + private const int SchemaVersion = 3; /// /// Rows buffered before the backfill flushes. Bounds peak memory on a very large queue — the @@ -117,6 +127,9 @@ public Task WaitForWorkAsync(TimeSpan pollInterval, CancellationToken ct = /// private const int BackfillFlushThreshold = 5_000; + /// Partition writes in flight at once during the migration. + private const int MigrationConcurrency = 16; + public JobQueueStore(ILogger logger, CraftSettings settings, ICraftTableStore store) { _logger = logger; @@ -138,18 +151,53 @@ internal static string Bucket(int priority) => "P" + Math.Clamp(priority, 0, MaxPriorityBucket).ToString("D2", CultureInfo.InvariantCulture); /// - /// The queue RowKey: deterministic per (run, task), so re-dispatching a task upserts its one row - /// instead of writing a second, time-prefixed one — the cause of a task running 2-6× (schema v2). + /// The queue RowKey: the run's epoch, then the run and task. Deterministic per (run, task) for a given + /// run epoch, so re-dispatching a task upserts its one row instead of writing a second (schema v2), and + /// ordered oldest run first within a bucket (schema v3). v2's bare {run}|{task} ordered a bucket + /// alphabetically, so a run whose name sorted late never ran while earlier-sorting runs kept arriving. /// /// Both components are escaped so the key is legal and the '|' separator is unambiguous; the row /// itself carries RunName/TaskId as properties, so nothing needs to parse the key back apart. - /// - /// The trade: within a priority bucket Azure now returns rows in run|task order rather than - /// oldest-first. Priority still orders across buckets, and a fan-out enqueues its tasks together, so - /// sub-priority FIFO fairness is the only thing given up — cheap next to never running a task twice. /// - internal static string BuildRowKey(string runName, string taskId) => - $"{EscapeKeyComponent(runName)}|{EscapeKeyComponent(taskId)}"; + internal static string BuildRowKey(DateTime runEpochUtc, string runName, string taskId) => + $"{runEpochUtc.Ticks.ToString("D19", CultureInfo.InvariantCulture)}|{EscapeKeyComponent(runName)}|{EscapeKeyComponent(taskId)}"; + + /// Whole seconds, so an epoch builds the same key before and after a storage round trip. + internal static DateTime EpochOf(DateTime utc) => + new(utc.Ticks - utc.Ticks % TimeSpan.TicksPerSecond, DateTimeKind.Utc); + + /// The run epoch a v3 key starts with, or null for an older key. + internal static DateTime? ParseEpoch(string rowKey) + { + if (rowKey.Length < 21 || rowKey[19] != '|') return null; + if (!long.TryParse(rowKey.AsSpan(0, 19), NumberStyles.None, CultureInfo.InvariantCulture, out var ticks)) + return null; + if (ticks <= 0 || ticks > DateTime.MaxValue.Ticks) return null; + return new DateTime(ticks, DateTimeKind.Utc); + } + + /// + /// The run's stored epoch, or recorded as it on the run's first enqueue. + /// ponytail: read-then-write rather than insert-if-absent, so two concurrent FIRST enqueues of one run + /// could record different epochs. A run is dispatched from one place and both dispatch and re-drive + /// check the index before enqueuing, so this is a narrow window, not a duplicate path. + /// + private async Task RunEpochAsync(string runName, DateTime queuedUtc, CancellationToken ct) + { + var stored = (await _store.GetAsync(_indexTable, IndexPartition(runName), EpochRowKey, ct)) + ?.GetDateTimeOffset("EpochUtc"); + if (stored != null) return EpochOf(stored.Value.UtcDateTime); + + var epoch = EpochOf(queuedUtc); + await _store.UpsertAsync(_indexTable, EpochRow(runName, epoch), ct); + return epoch; + } + + private static StoreRow EpochRow(string runName, DateTime epoch) => + new(IndexPartition(runName), EpochRowKey) + { + Properties = { ["RunName"] = runName, ["EpochUtc"] = new DateTimeOffset(epoch, TimeSpan.Zero) } + }; /// /// Escape a run or task id for use inside a queue RowKey: the Azure-illegal key characters plus '|' @@ -233,7 +281,7 @@ private static StoreRow IndexRow(string runName, string taskId, string bucket, s Properties = { ["TaskId"] = taskId, ["RunName"] = runName } }; - /// Add one task to the queue. Idempotent for a given (queuedUtc, run, task). + /// Add one task to the queue. Idempotent per (run, task): the key comes from the run's stored epoch. /// /// Queue row first, index row second. The queue row is what makes the task actually run; the index /// only accelerates lookups. If the process dies between the two the task still executes, and the @@ -246,7 +294,7 @@ public async Task EnqueueAsync(string runName, string taskId, int priority, Date CancellationToken ct = default) { var bucket = Bucket(priority); - var rowKey = BuildRowKey(runName, taskId); + var rowKey = BuildRowKey(await RunEpochAsync(runName, queuedUtc, ct), runName, taskId); await _store.UpsertAsync(_queueTable, new StoreRow(bucket, rowKey) { @@ -276,12 +324,15 @@ public async Task EnqueueAsync(string runName, string taskId, int priority, Date public async Task EnqueueBatchAsync(string runName, IReadOnlyList<(string TaskId, int Priority)> tasks, DateTime queuedUtc, CancellationToken ct = default) { + if (tasks.Count == 0) return; + var indexRows = new List(tasks.Count); + var epoch = await RunEpochAsync(runName, queuedUtc, ct); foreach (var byBucket in tasks.GroupBy(t => Bucket(t.Priority))) { var queuedOffset = new DateTimeOffset(queuedUtc, TimeSpan.Zero); - var rows = byBucket.Select(t => new StoreRow(byBucket.Key, BuildRowKey(runName, t.TaskId)) + var rows = byBucket.Select(t => new StoreRow(byBucket.Key, BuildRowKey(epoch, runName, t.TaskId)) { Properties = { @@ -297,13 +348,13 @@ public async Task EnqueueBatchAsync(string runName, IReadOnlyList<(string TaskId await _store.UpsertBatchAsync(_queueTable, byBucket.Key, rows, ct); indexRows.AddRange(byBucket.Select(t => - IndexRow(runName, t.TaskId, byBucket.Key, BuildRowKey(runName, t.TaskId)))); + IndexRow(runName, t.TaskId, byBucket.Key, BuildRowKey(epoch, runName, t.TaskId)))); } if (indexRows.Count > 0) await _store.UpsertBatchAsync(_indexTable, IndexPartition(runName), indexRows, ct); - if (tasks.Count > 0) WakePump(); + WakePump(); } /// A queued task this worker now owns, with the row key needed to release it. @@ -817,6 +868,7 @@ private async Task MigrateSchemaAsync(CancellationToken ct) var migrated = 0; var rekeyed = 0; var skipped = 0; + var epochs = new Dictionary(StringComparer.Ordinal); static void AddRow(Dictionary> map, string key, StoreRow row) { @@ -829,18 +881,22 @@ static void AddKey(Dictionary> map, string key, string rowK list.Add(rowKey); } + // Chunked to a transaction's worth, so the one big bucket partition goes in parallel too. + Task EachAsync(Dictionary> map, Func, Task> write) => + Parallel.ForEachAsync(map.SelectMany(kv => kv.Value.Chunk(100).Select(c => (kv.Key, c))), + new ParallelOptions { MaxDegreeOfParallelism = MigrationConcurrency, CancellationToken = ct }, + async (part, _) => await write(part.Key, part.c)); + async Task FlushAsync() { // New rows first, so a crash before the deletes leaves BOTH and the re-run converges — never - // the index advertising a queue row that no longer exists. - foreach (var (bucket, rows) in newQueue) - await _store.UpsertBatchAsync(_queueTable, bucket, rows, ct); - foreach (var (partition, rows) in newIndex) - await _store.UpsertBatchAsync(_indexTable, partition, rows, ct); - foreach (var (bucket, keys) in oldQueue) - await _store.DeleteBatchAsync(_queueTable, bucket, keys, ct); - foreach (var (partition, keys) in oldIndex) - await _store.DeleteBatchAsync(_indexTable, partition, keys, ct); + // the index advertising a queue row that no longer exists. Within a phase the partitions are + // independent, and the index has one per run, so they go in parallel: one at a time, a + // 16k-run backlog was ~32k sequential round trips while the pump waited to claim. + await EachAsync(newQueue, (bucket, rows) => _store.UpsertBatchAsync(_queueTable, bucket, rows, ct)); + await EachAsync(newIndex, (partition, rows) => _store.UpsertBatchAsync(_indexTable, partition, rows, ct)); + await EachAsync(oldQueue, (bucket, keys) => _store.DeleteBatchAsync(_queueTable, bucket, keys, ct)); + await EachAsync(oldIndex, (partition, keys) => _store.DeleteBatchAsync(_indexTable, partition, keys, ct)); migrated += buffered; newQueue.Clear(); newIndex.Clear(); oldQueue.Clear(); oldIndex.Clear(); @@ -854,9 +910,18 @@ async Task FlushAsync() if (string.IsNullOrEmpty(runName) || string.IsNullOrEmpty(taskId)) { skipped++; continue; } var bucket = row.PartitionKey; - var newKey = BuildRowKey(runName, taskId); var runPartition = IndexPartition(runName); + // One epoch per run. Rows already on a v3 key (a re-run after a crash) sort ahead of the run's + // unmigrated rows in the scan, so their epoch is seen first and the rest follow it. + if (!epochs.TryGetValue(runName, out var epoch)) + { + epoch = ParseEpoch(row.RowKey) ?? EpochOf(QueuedUtcOf(row)); + epochs[runName] = epoch; + AddRow(newIndex, runPartition, EpochRow(runName, epoch)); + } + var newKey = BuildRowKey(epoch, runName, taskId); + // The new-scheme row: same bucket + claim state + priority, key deterministic, enqueue time as // a property (from the row, or the legacy key, or now). AddRow(newQueue, bucket, new StoreRow(bucket, newKey) diff --git a/tests/Craft.Tests/JobQueueDispatchableTests.cs b/tests/Craft.Tests/JobQueueDispatchableTests.cs index 3821a71..6a0b9fe 100644 --- a/tests/Craft.Tests/JobQueueDispatchableTests.cs +++ b/tests/Craft.Tests/JobQueueDispatchableTests.cs @@ -50,11 +50,11 @@ public async Task AnIndexRowWithNoQueueRowIsNotDispatchable_ButTheIndexStillList { var (queue, backing) = NewQueue(); await queue.InitializeAsync(); - await queue.EnqueueBatchAsync("R", [("a", 4), ("b", 4)], DateTime.UtcNow); + await queue.EnqueueBatchAsync("R", [("a", 4), ("b", 4)], DateTime.UnixEpoch); // Delete ONLY b's queue row, leaving its index row — the exact divergence a crash between the // two deletes, or a run carried in from a pre-pump build, leaves behind. - await backing.DeleteAsync(QueueTable, JobQueueStore.Bucket(4), JobQueueStore.BuildRowKey("R", "b")); + await backing.DeleteAsync(QueueTable, JobQueueStore.Bucket(4), JobQueueStore.BuildRowKey(DateTime.UnixEpoch, "R", "b")); // The index — what the old re-drive trusted — still reports both as queued. var indexView = await queue.GetQueuedTaskIdsAsync("R"); @@ -72,11 +72,11 @@ public async Task AQueueRowOwnedWithNoLeaseIsNotDispatchable() { var (queue, backing) = NewQueue(); await queue.InitializeAsync(); - await queue.EnqueueBatchAsync("R", [("a", 4)], DateTime.UtcNow); + await queue.EnqueueBatchAsync("R", [("a", 4)], DateTime.UnixEpoch); // Owned with no LeaseUntil: neither "Owner eq ''" nor "LeaseUntil lt now", so the claim filter // can never match it and the pump will never dispatch it — a ghost as surely as a missing row. - var key = JobQueueStore.BuildRowKey("R", "a"); + var key = JobQueueStore.BuildRowKey(DateTime.UnixEpoch, "R", "a"); var row = await backing.GetAsync(QueueTable, JobQueueStore.Bucket(4), key); Assert.NotNull(row); row!["Owner"] = "dead-instance"; diff --git a/tests/Craft.Tests/JobQueueFifoTests.cs b/tests/Craft.Tests/JobQueueFifoTests.cs new file mode 100644 index 0000000..b731041 --- /dev/null +++ b/tests/Craft.Tests/JobQueueFifoTests.cs @@ -0,0 +1,195 @@ +using Craft.Configuration; +using Craft.Storage; +using Microsoft.Extensions.Logging.Abstractions; + +namespace Craft.Tests; + +/// +/// Schema v3: within a priority bucket the queue drains oldest run first. v2 keyed rows {run}|{task}, +/// so a bucket drained alphabetically by run name and a run whose name sorted late never ran while +/// earlier-sorting runs kept arriving (MailboxRules/UpdatePermissions starved for 16 days behind a steady +/// stream of AuditLog/DomainAnalyser runs). These pin the order, that the key stays one row per task, and +/// the migration of a live v2 backlog. +/// +public class JobQueueFifoTests +{ + private static readonly TimeSpan Lease = TimeSpan.FromMinutes(20); + private static DateTime At(int minute) => new(2026, 10, 5, 2, minute, 0, DateTimeKind.Utc); + + private sealed record Q(JobQueueStore Queue, RunRemainingCounterTests.ConditionalStore Store, string QueueTable, string IndexTable); + + private static Q NewQueue() + { + var settings = new CraftSettings(); + var store = new RunRemainingCounterTests.ConditionalStore(); + var queue = new JobQueueStore(NullLogger.Instance, settings, store); + return new Q(queue, store, $"{settings.Orchestrator.TablePrefix}Queue", $"{settings.Orchestrator.TablePrefix}QueueIndex"); + } + + private static async Task> DrainOrderAsync(JobQueueStore queue) + { + var order = new List(); + while (true) + { + var claimed = await queue.ClaimBatchAsync("w", 1, Lease); + if (claimed.Count == 0) return order; + order.Add($"{claimed[0].RunName}/{claimed[0].TaskId}"); + await queue.RemoveAsync(claimed[0]); + } + } + + private static async Task> RowsAsync(Q q, string table) + { + var rows = new List(); + await foreach (var r in q.Store.QueryTableAsync(table)) rows.Add(r); + return rows; + } + + [Fact] + public async Task ABucketDrainsOldestRunFirst_NotAlphabetically() + { + var q = NewQueue(); + await q.Queue.InitializeAsync(); + + await q.Queue.EnqueueBatchAsync("UpdatePermissions-1", [("u0", 4), ("u1", 4)], At(0)); + await q.Queue.EnqueueBatchAsync("AuditLog-2", [("a0", 4)], At(5)); + await q.Queue.EnqueueBatchAsync("DomainAnalyser-3", [("d0", 4)], At(9)); + + Assert.Equal(["UpdatePermissions-1/u0", "UpdatePermissions-1/u1", "AuditLog-2/a0", "DomainAnalyser-3/d0"], + await DrainOrderAsync(q.Queue)); + } + + [Fact] + public async Task PriorityStillBeatsAge() + { + var q = NewQueue(); + await q.Queue.InitializeAsync(); + + await q.Queue.EnqueueBatchAsync("Old", [("o0", 4)], At(0)); + await q.Queue.EnqueueBatchAsync("Urgent", [("x0", 1)], At(30)); + + Assert.Equal(["Urgent/x0", "Old/o0"], await DrainOrderAsync(q.Queue)); + } + + [Fact] + public async Task AReEnqueueHoursLater_KeepsTheRunsPlaceAndOneRow() + { + var q = NewQueue(); + await q.Queue.InitializeAsync(); + + await q.Queue.EnqueueBatchAsync("Zeta", [("z0", 4), ("z1", 4)], At(0)); + await q.Queue.EnqueueBatchAsync("Alpha", [("a0", 4)], At(10)); + // The re-drive re-queues a task with the time it noticed, long after the run started. + await q.Queue.EnqueueAsync("Zeta", "z1", 4, At(59)); + + Assert.Equal(3, (await RowsAsync(q, q.QueueTable)).Count); + Assert.Equal(["Zeta/z0", "Zeta/z1", "Alpha/a0"], await DrainOrderAsync(q.Queue)); + } + + [Fact] + public async Task TheEpochRowIsInvisibleToIndexReaders_AndGoesWithTheRun() + { + var q = NewQueue(); + await q.Queue.InitializeAsync(); + await q.Queue.EnqueueBatchAsync("R", [("a", 4)], At(0)); + + Assert.Equal(["a"], await q.Queue.GetQueuedTaskIdsAsync("R")); + Assert.Equal(["a"], await q.Queue.GetDispatchableTaskIdsAsync("R", ["a"])); + Assert.Equal(0, await q.Queue.ReleaseRunClaimsAsync("R")); + + await q.Queue.RemoveRunAsync("R"); + + Assert.Empty(await RowsAsync(q, q.QueueTable)); + Assert.DoesNotContain(await RowsAsync(q, q.IndexTable), r => r.PartitionKey == "R"); + } + + [Fact] + public async Task AnEpochSurvivesTheSecondsRoundTrip() + { + var q = NewQueue(); + await q.Queue.InitializeAsync(); + var odd = At(0).AddTicks(1_234_567); + + await q.Queue.EnqueueBatchAsync("R", [("a", 4)], odd); + await q.Queue.EnqueueAsync("R", "a", 4, odd); + + var row = Assert.Single(await RowsAsync(q, q.QueueTable)); + Assert.Equal(JobQueueStore.BuildRowKey(At(0), "R", "a"), row.RowKey); + } + + // ── migration ──────────────────────────────────────────────────────────────────────────────── + + /// A queue row and its index entry exactly as v2 wrote them, plus the v2 schema marker. + private static async Task SeedV2Async(Q q, string run, string task, DateTime queuedUtc, int priority = 4, + string owner = "", DateTimeOffset? lease = null) + { + var bucket = JobQueueStore.Bucket(priority); + var key = $"{run}|{task}"; + await q.Store.UpsertAsync(q.QueueTable, new StoreRow(bucket, key) + { + Properties = + { + ["RunName"] = run, ["TaskId"] = task, ["Priority"] = priority, ["Owner"] = owner, + ["LeaseUntil"] = lease, ["QueuedUtc"] = new DateTimeOffset(queuedUtc, TimeSpan.Zero), + } + }); + await q.Store.UpsertAsync(q.IndexTable, new StoreRow(run, $"{bucket}|{key}") + { + Properties = { ["TaskId"] = task, ["RunName"] = run } + }); + await q.Store.UpsertAsync(q.IndexTable, new StoreRow("$schema", "queue-index") { Properties = { ["Version"] = 2 } }); + } + + [Fact] + public async Task MigratesAV2Backlog_ToOldestRunFirst_KeepingClaimsAndTheIndex() + { + var q = NewQueue(); + // Alphabetically the v2 order was AuditLog, MailboxRules; by age MailboxRules is 16 days older. + await SeedV2Async(q, "MailboxRules_t1", "b1", At(0).AddDays(-16)); + await SeedV2Async(q, "MailboxRules_t1", "b2", At(0).AddDays(-16)); + await SeedV2Async(q, "AuditLog_t1", "s1", At(0)); + var leaseUntil = DateTimeOffset.UtcNow.AddMinutes(10); + await SeedV2Async(q, "AuditLog_t1", "s2", At(0), owner: "w-other", lease: leaseUntil); + + await q.Queue.InitializeAsync(); + + var queueRows = await RowsAsync(q, q.QueueTable); + Assert.Equal(4, queueRows.Count); + Assert.All(queueRows, r => Assert.NotNull(JobQueueStore.ParseEpoch(r.RowKey))); + var claimedRow = Assert.Single(queueRows, r => r.GetString("TaskId") == "s2"); + Assert.Equal("w-other", claimedRow.GetString("Owner")); + Assert.Equal(leaseUntil, claimedRow.GetDateTimeOffset("LeaseUntil")); + + // Index entries point at the new keys: every run-scoped read still sees its tasks as dispatchable. + Assert.Equal(["b1", "b2"], (await q.Queue.GetDispatchableTaskIdsAsync("MailboxRules_t1", ["b1", "b2"])).Order()); + Assert.Equal(["s1", "s2"], (await q.Queue.GetDispatchableTaskIdsAsync("AuditLog_t1", ["s1", "s2"])).Order()); + Assert.Equal(6, (await RowsAsync(q, q.IndexTable)).Count(r => r.PartitionKey != "$schema")); + + // A later enqueue of a migrated task lands on its existing row, not a second one. + await q.Queue.EnqueueAsync("MailboxRules_t1", "b2", 4, DateTime.UtcNow); + Assert.Equal(4, (await RowsAsync(q, q.QueueTable)).Count); + + Assert.Equal(["MailboxRules_t1/b1", "MailboxRules_t1/b2", "AuditLog_t1/s1"], await DrainOrderAsync(q.Queue)); + } + + [Fact] + public async Task AMigrationInterruptedPartWay_ConvergesOnRerun() + { + var q = NewQueue(); + await SeedV2Async(q, "Run", "t1", At(0)); + await SeedV2Async(q, "Run", "t2", At(0)); + // t1 already re-keyed by a pass that crashed before deleting its old row or finishing t2. + var epoch = At(0).AddMinutes(-3); + var v3Key = JobQueueStore.BuildRowKey(epoch, "Run", "t1"); + await q.Store.UpsertAsync(q.QueueTable, new StoreRow("P04", v3Key) + { + Properties = { ["RunName"] = "Run", ["TaskId"] = "t1", ["Priority"] = 4, ["Owner"] = "", ["QueuedUtc"] = new DateTimeOffset(At(0), TimeSpan.Zero) } + }); + + await q.Queue.InitializeAsync(); + + var rows = await RowsAsync(q, q.QueueTable); + Assert.Equal(2, rows.Count); + Assert.All(rows, r => Assert.Equal(epoch, JobQueueStore.ParseEpoch(r.RowKey))); + } +} diff --git a/tests/Craft.Tests/JobQueueIndexBackfillTests.cs b/tests/Craft.Tests/JobQueueIndexBackfillTests.cs index 3041822..243b0ec 100644 --- a/tests/Craft.Tests/JobQueueIndexBackfillTests.cs +++ b/tests/Craft.Tests/JobQueueIndexBackfillTests.cs @@ -87,7 +87,7 @@ public async Task MigratesLegacyRowsToTheDeterministicKeyScheme() // The legacy row is gone; one row remains, keyed deterministically and carrying QueuedUtc. var only = Assert.Single(rows); - Assert.Equal(JobQueueStore.BuildRowKey("run-a", "task-0"), only.RowKey); + Assert.Equal(JobQueueStore.BuildRowKey(At(3), "run-a", "task-0"), only.RowKey); Assert.Equal(new DateTimeOffset(At(3), TimeSpan.Zero), only.GetDateTimeOffset("QueuedUtc")); // And it is still claimable, exactly once. diff --git a/tests/Craft.Tests/JobQueueStoreTests.cs b/tests/Craft.Tests/JobQueueStoreTests.cs index 7ed2264..9c29531 100644 --- a/tests/Craft.Tests/JobQueueStoreTests.cs +++ b/tests/Craft.Tests/JobQueueStoreTests.cs @@ -196,9 +196,9 @@ public void RowKeyIsDeterministicPerRunAndTask() { // Schema v2: the key is a function of (run, task) only, so re-dispatching a task upserts its one // row instead of writing a second, time-prefixed one — the duplicate-execution class. - Assert.Equal(JobQueueStore.BuildRowKey("r", "t"), JobQueueStore.BuildRowKey("r", "t")); - Assert.NotEqual(JobQueueStore.BuildRowKey("r", "t1"), JobQueueStore.BuildRowKey("r", "t2")); - Assert.NotEqual(JobQueueStore.BuildRowKey("r1", "t"), JobQueueStore.BuildRowKey("r2", "t")); + Assert.Equal(JobQueueStore.BuildRowKey(DateTime.UnixEpoch, "r", "t"), JobQueueStore.BuildRowKey(DateTime.UnixEpoch, "r", "t")); + Assert.NotEqual(JobQueueStore.BuildRowKey(DateTime.UnixEpoch, "r", "t1"), JobQueueStore.BuildRowKey(DateTime.UnixEpoch, "r", "t2")); + Assert.NotEqual(JobQueueStore.BuildRowKey(DateTime.UnixEpoch, "r1", "t"), JobQueueStore.BuildRowKey(DateTime.UnixEpoch, "r2", "t")); } [Fact] @@ -217,10 +217,10 @@ public void RowKeyEscapingKeepsDistinctPairsDistinctAndKeysLegal() { // The '|' separator and '%' escape are themselves escaped, so "a|b"+"c" and "a"+"b|c" cannot // collide onto one row. - Assert.NotEqual(JobQueueStore.BuildRowKey("a|b", "c"), JobQueueStore.BuildRowKey("a", "b|c")); + Assert.NotEqual(JobQueueStore.BuildRowKey(DateTime.UnixEpoch, "a|b", "c"), JobQueueStore.BuildRowKey(DateTime.UnixEpoch, "a", "b|c")); // An illegal character in a component is escaped away, keeping the key legal for Azure Table. - var key = JobQueueStore.BuildRowKey("run", "Owner/Repo - No tenant"); + var key = JobQueueStore.BuildRowKey(DateTime.UnixEpoch, "run", "Owner/Repo - No tenant"); Assert.DoesNotContain(key, c => c is '/' or '\\' or '#' or '?' || char.IsControl(c)); } diff --git a/tests/Craft.Tests/RunRemainingCounterTests.cs b/tests/Craft.Tests/RunRemainingCounterTests.cs index 2319abe..8661aca 100644 --- a/tests/Craft.Tests/RunRemainingCounterTests.cs +++ b/tests/Craft.Tests/RunRemainingCounterTests.cs @@ -26,7 +26,7 @@ public class RunRemainingCounterTests /// internal sealed class ConditionalStore : ICraftTableStore { - private readonly Dictionary> _tables = new(); + private readonly System.Collections.Concurrent.ConcurrentDictionary> _tables = new(); private long _etag; public int ConditionalWrites { get; private set; } @@ -41,8 +41,8 @@ internal sealed class ConditionalStore : ICraftTableStore /// Awaited at the start of a partition query, with the table name — lets a test hold a read open. public Func? OnPartitionQuery { get; set; } - private Dictionary<(string, string), StoreRow> Table(string t) => - _tables.TryGetValue(t, out var x) ? x : _tables[t] = new(); + private System.Collections.Concurrent.ConcurrentDictionary<(string, string), StoreRow> Table(string t) => + _tables.GetOrAdd(t, _ => new()); private StoreRow Stamp(StoreRow row) => new(row.PartitionKey, row.RowKey) { @@ -136,13 +136,13 @@ public async IAsyncEnumerable QueryTableAsync(string table, public Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) { - Table(table).Remove((partitionKey, rowKey)); + Table(table).TryRemove((partitionKey, rowKey), out _); return Task.CompletedTask; } public Task DeletePartitionAsync(string table, string partitionKey, CancellationToken ct = default) { - foreach (var k in Table(table).Keys.Where(k => k.Item1 == partitionKey).ToList()) Table(table).Remove(k); + foreach (var k in Table(table).Keys.Where(k => k.Item1 == partitionKey).ToList()) Table(table).TryRemove(k, out _); return Task.CompletedTask; } } diff --git a/tests/Craft.Tests/TableKeyTests.cs b/tests/Craft.Tests/TableKeyTests.cs index c8e469f..081fe81 100644 --- a/tests/Craft.Tests/TableKeyTests.cs +++ b/tests/Craft.Tests/TableKeyTests.cs @@ -99,7 +99,7 @@ public void QueueRowKeyIsLegalForARepoNamedTask() } """); - var rowKey = JobQueueStore.BuildRowKey("UserTaskOrchestrator_No tenant", id); + var rowKey = JobQueueStore.BuildRowKey(DateTime.UnixEpoch, "UserTaskOrchestrator_No tenant", id); Assert.True(TableKeys.IsSafe(rowKey)); } From c98a0916714cea92b01993a4cc60e59697ed35df Mon Sep 17 00:00:00 2001 From: Zacgoose <107489668+Zacgoose@users.noreply.github.com> Date: Mon, 5 Oct 2026 22:31:23 +0800 Subject: [PATCH 04/24] fix(queue): key runs by start time and keep queue reads flat at scale - the queue epoch is the run's StartedUtc, so concurrent enqueues of a run can no longer write two rows for a task - schema v3 migration keys runs by their Run row's start time and skips rows already re-keyed - claim reads one page sized to the batch - run-scoped reads (dispatch check, claim release) read the run's contiguous key range instead of one GET per task - RemoveRunAsync is scoped to one outing of a recurring run name and deletes in batches - status snapshot streams with a projection and keeps only the head rows; re-scans wait 4x the last scan - CancelRunAsync cancels the live run and drops its queue rows; re-drive requeues orphans in one batch --- .../Orchestration/JobQueueStatusReader.cs | 73 ++-- Services/Orchestration/OrchestratorService.cs | 90 ++--- Services/Storage/AzureTableStore.cs | 16 + Services/Storage/ICraftTableStore.cs | 20 ++ Services/Storage/JobQueueStore.cs | 337 +++++++++--------- tests/Craft.Tests/JobQueueAzuriteTests.cs | 79 ++++ tests/Craft.Tests/JobQueueFifoTests.cs | 79 +++- .../Craft.Tests/JobQueueStatusReaderTests.cs | 19 + tests/Craft.Tests/JobQueueStoreTests.cs | 8 +- .../Craft.Tests/OrchestratorCancelRunTests.cs | 78 ++++ tests/Craft.Tests/RunRemainingCounterTests.cs | 4 + 11 files changed, 525 insertions(+), 278 deletions(-) create mode 100644 tests/Craft.Tests/OrchestratorCancelRunTests.cs diff --git a/Services/Orchestration/JobQueueStatusReader.cs b/Services/Orchestration/JobQueueStatusReader.cs index aabd61d..413d11d 100644 --- a/Services/Orchestration/JobQueueStatusReader.cs +++ b/Services/Orchestration/JobQueueStatusReader.cs @@ -29,7 +29,9 @@ namespace Craft.Orchestration; /// each (see BuildSnapshotAsync). A customer reporting sluggishness had "kept the Worker Health page /// open" — which is what was driving it. /// -/// So: stamp on completion, and never re-scan more often than the last scan took. +/// So: stamp on completion, and keep the scan to a fifth of the time — the next one waits at least four +/// times as long as the last one took. The scan also streams: only aggregates and the head of the queue are +/// kept, so a snapshot's memory does not grow with the backlog. /// public class JobQueueStatusReader : IDisposable { @@ -48,6 +50,9 @@ public class JobQueueStatusReader : IDisposable /// private const int MaxCounterLookups = 200; + /// Rows kept from the head of the queue (claim order) for the job listings; the rest are only counted. + internal const int HeadRows = 2_000; + private volatile QueueSnapshot? _cached; private readonly SemaphoreSlim _refreshGate = new(1, 1); @@ -82,12 +87,14 @@ public JobQueueStatusReader(ILogger logger, JobManager job public sealed record RunQueueInfo(int Unclaimed, int Claimed, int MinPriority, DateTime? OldestQueuedUtc); - /// One scan of the queue table, aggregated the way the status APIs consume it. + /// + /// One scan of the queue table, aggregated the way the status APIs consume it. is the + /// head of the queue only (up to , in claim order); the counts cover every row. + /// public sealed record QueueSnapshot(DateTime TakenUtc, IReadOnlyList Rows, - int Unclaimed, int Claimed, DateTime? OldestUnclaimedUtc, + int Total, int Unclaimed, int Claimed, DateTime? OldestUnclaimedUtc, IReadOnlyDictionary ByRun) { - public int Total => Rows.Count; public double AgeSeconds => (DateTime.UtcNow - TakenUtc).TotalSeconds; } @@ -99,14 +106,14 @@ public sealed record QueueSnapshot(DateTime TakenUtc, IReadOnlyList /// /// The shortest interval we will re-scan at: never less than the requested age, and never less than - /// the last scan took. On a healthy instance the scan is milliseconds and this is just the 5s TTL; - /// on a degraded one it is what stops the refresh loop from consuming the storage account. + /// four times the last scan took. On a healthy instance the scan is milliseconds and this is just the + /// 5s TTL; on a degraded one it is what stops the refresh loop from consuming the storage account. /// private TimeSpan EffectiveTtl(TimeSpan? maxAge) { var requested = maxAge ?? DefaultTtl; - var lastBuild = TimeSpan.FromTicks(Interlocked.Read(ref _lastBuildTicks)); - return lastBuild > requested ? lastBuild : requested; + var floor = TimeSpan.FromTicks(Interlocked.Read(ref _lastBuildTicks) * 4); + return floor > requested ? floor : requested; } public QueueSnapshot? GetCached(TimeSpan? maxAge = null) @@ -159,47 +166,47 @@ private TimeSpan EffectiveTtl(TimeSpan? maxAge) } /// - /// One scan of the queue table, aggregated. No per-run storage reads: everything here comes from - /// the rows the scan already returned, so the cost is one scan regardless of how many runs the - /// backlog spans. + /// One scan of the queue table, aggregated as it streams. No per-run storage reads, and only the head + /// rows are held, so the cost is one scan and the memory is bounded by the number of runs. /// private async Task BuildSnapshotAsync(CancellationToken ct) { - var rows = await _queue.ListQueuedAsync(ct); - + var head = new List(); + var total = 0; var unclaimed = 0; DateTime? oldestUnclaimed = null; var byRun = new Dictionary(StringComparer.Ordinal); - foreach (var group in rows.GroupBy(r => r.RunName)) + await foreach (var row in _queue.StreamQueuedAsync(ct)) { - var runUnclaimed = 0; - var runClaimed = 0; - var minPriority = int.MaxValue; - DateTime? oldest = null; + total++; + if (head.Count < HeadRows) head.Add(row); + + var info = byRun.TryGetValue(row.RunName, out var existing) + ? existing + : new RunQueueInfo(0, 0, row.Priority, null); - foreach (var row in group) + if (row.Claimed) { - if (row.Claimed) runClaimed++; - else + info = info with { Claimed = info.Claimed + 1 }; + } + else + { + unclaimed++; + if (oldestUnclaimed == null || row.QueuedUtc < oldestUnclaimed) oldestUnclaimed = row.QueuedUtc; + info = info with { - runUnclaimed++; - if (oldest == null || row.QueuedUtc < oldest) oldest = row.QueuedUtc; - } - if (row.Priority < minPriority) minPriority = row.Priority; + Unclaimed = info.Unclaimed + 1, + OldestQueuedUtc = info.OldestQueuedUtc is { } o && o <= row.QueuedUtc ? o : row.QueuedUtc, + }; } - - unclaimed += runUnclaimed; - if (oldest != null && (oldestUnclaimed == null || oldest < oldestUnclaimed)) - oldestUnclaimed = oldest; - - byRun[group.Key] = new RunQueueInfo(runUnclaimed, runClaimed, - minPriority == int.MaxValue ? 0 : minPriority, oldest); + if (row.Priority < info.MinPriority) info = info with { MinPriority = row.Priority }; + byRun[row.RunName] = info; } // Stamped on COMPLETION, not on entry. A scan that took longer than the TTL would otherwise // return a snapshot that is already expired, and the next poll would start another immediately. - return new QueueSnapshot(DateTime.UtcNow, rows, unclaimed, rows.Count - unclaimed, + return new QueueSnapshot(DateTime.UtcNow, head, total, unclaimed, total - unclaimed, oldestUnclaimed, byRun); } diff --git a/Services/Orchestration/OrchestratorService.cs b/Services/Orchestration/OrchestratorService.cs index f5a6edb..319b4d6 100644 --- a/Services/Orchestration/OrchestratorService.cs +++ b/Services/Orchestration/OrchestratorService.cs @@ -33,8 +33,7 @@ public class OrchestratorService : IJobDescriptorStateWriter, IDisposable private readonly JobManager _jobManager; private readonly OrchestratorTableStore _store; - /// The durable job queue. Its table is created alongside the orchestrator's; nothing - /// dispatches from it yet, so an existing deployment gains an empty table and nothing else. + /// The durable job queue: every task is enqueued here and dispatched by the pump. private readonly JobQueueStore _queue; private readonly OrchestratorStatusWriter _writer; private readonly CraftSettings _settings; @@ -546,7 +545,7 @@ public async Task ResumeInterruptedRunsAsync(CancellationToken ct) run.PostExecStatus = "Abandoned"; await _store.UpsertRunAsync(run); await _store.CleanupRunAsync(run.Name); - await _queue.RemoveRunAsync(run.Name, ct); + await _queue.RemoveRunAsync(run.Name, run.StartedUtc, ct); postExecGaveUp++; continue; } @@ -688,7 +687,7 @@ public async Task RunRetentionSweepAsync(Cancellation { try { - await _queue.RemoveRunAsync(name, ct); + await _queue.RemoveRunAsync(name, ct: ct); } catch (Exception ex) { @@ -959,21 +958,9 @@ private async Task DispatchPendingTasksAsync(OrchestratorRun run, string taskPat var pending = run.Tasks.Where(t => t.Status == "Pending").ToList(); - // Skip anything that already has a queue row. - // - // A queue RowKey is BuildRowKey(queuedUtc, run, task), so it is idempotent only for a GIVEN - // timestamp — re-dispatching the same task later writes a SECOND row rather than updating the - // first. That is exactly what crash recovery does: ResumeInterruptedRunsAsync flips interrupted - // tasks back to Pending and calls this, while every pre-crash row for those tasks is still in the - // queue. Measured on a killed 140-task fanout: 102 tasks re-dispatched on top of 102 survivors. - // - // Most duplicates are harmless — the second row resolves to a task that has since finished and is - // dropped as a stale descriptor. But if both rows are claimed while the task is still RUNNING, - // the resolver's terminal-status guard does not apply and the task runs twice. That happened: - // Intune_dev.mspadvisors.com was claimed again 5 minutes into its own execution and ran a second - // time. Not writing the duplicate is the fix; the resolver check below is the backstop. - // - // A read of the run's rows costs one table scan per dispatch, against a write per task avoided. + // Skip anything that already has a queue row. Re-writing it would reset a live claim (Owner and + // LeaseUntil) on a task another worker may be running, which is how a task ran twice. One index + // partition read per dispatch. HashSet alreadyQueued; try { @@ -1028,7 +1015,7 @@ private async Task DispatchPendingTasksAsync(OrchestratorRun run, string taskPat // the JobManager only ever sees the batch JobQueuePump claims from it. await _queue.EnqueueBatchAsync(run.Name, toQueue.Select(t => (t.Id, t.Priority ?? priority)).ToList(), - DateTime.UtcNow, ct); + run.StartedUtc, ct); // quiet = called from crash recovery, where a per-run line per resumed run is the flood the // aggregate summary replaces — drop to Debug. A normal orchestration start logs it at Info (one line). @@ -1269,20 +1256,19 @@ private void CancelRemainingSequentialTasks(OrchestratorRun run) } /// - /// Put one task back on the durable queue. Fire-and-forget because every caller is on a lock or a - /// timer callback, and a failure is recoverable: the task is still Pending in storage, so the next - /// re-drive finds it again. + /// Put tasks of one run back on the durable queue, in one batch. Fire-and-forget because every caller + /// is on a lock or a timer callback, and a failure is recoverable: the tasks are still Pending in + /// storage, so the next re-drive finds them again. /// - private void RequeueToTable(OrchestratorRun run, OrchestratorTaskItem task) + private void RequeueToTable(OrchestratorRun run, IReadOnlyList tasks) { - var priority = task.Priority ?? run.Priority; _ = Task.Run(async () => { - var key = DeferralKey(run.Name, task.Id); try { - await _queue.EnqueueAsync(run.Name, task.Id, priority, DateTime.UtcNow); - _requeueFailures.TryRemove(key, out _); + await _queue.EnqueueBatchAsync(run.Name, + tasks.Select(t => (t.Id, t.Priority ?? run.Priority)).ToList(), run.StartedUtc); + foreach (var task in tasks) _requeueFailures.TryRemove(DeferralKey(run.Name, task.Id), out _); } catch (Exception ex) { @@ -1290,17 +1276,18 @@ private void RequeueToTable(OrchestratorRun run, OrchestratorTaskItem task) // oversized property), and the re-drive resets the deferral counter on every pass — // without this cap the retry loop is infinite and the run it belongs to can never // finalize. Consecutive failures only: a success above clears the count. - var failures = _requeueFailures.AddOrUpdate(key, 1, (_, c) => c + 1); - if (failures >= MaxRequeueFailures) + foreach (var task in tasks) { + var key = DeferralKey(run.Name, task.Id); + var failures = _requeueFailures.AddOrUpdate(key, 1, (_, c) => c + 1); + if (failures < MaxRequeueFailures) continue; _requeueFailures.TryRemove(key, out _); FailTaskTerminally(run, task, $"Could not re-queue after {failures} consecutive attempts: {ex.Message}"); - return; } _logger.LogWarning(ex, - "[Scheduler] Could not re-queue {Task} in {Run} (attempt {Count}/{Max}) — the re-drive will retry", - task.Id, run.Name, failures, MaxRequeueFailures); + "[Scheduler] Could not re-queue {Count} task(s) in {Run} — the re-drive will retry", + tasks.Count, run.Name); } }); } @@ -1706,7 +1693,7 @@ private void DeferTask(OrchestratorRun run, OrchestratorTaskItem task, Exception // Back to the QUEUE, not to memory. The pump drops a claimed row once the JobManager is done with // the job, so an in-memory re-queue here would leave the retry with no durable row behind it — and // nothing to pick it up again if this instance went away. - RequeueToTable(run, task); + RequeueToTable(run, [task]); } /// @@ -1751,11 +1738,7 @@ private void DeferTask(OrchestratorRun run, OrchestratorTaskItem task, Exception /// worker-pool-sized buffer and leaves the rest in storage, so a 124-task run against eight workers /// has most of its tasks Pending and absent from the JobManager for minutes. /// - /// The consequence was severe and silent. Every 60 seconds this re-queued the entire un-started - /// backlog — measured live at 92, then 60, 60, 52, 44, 36 tasks on consecutive ticks — and because - /// RequeueToTable stamps UtcNow into the RowKey, each pass created an ADDITIONAL row for the same - /// task instead of updating the existing one. Every copy was independently claimable, so tasks ran - /// once per copy: one Intune collection executed six times, from six rows exactly 60s apart. + /// Treating that as orphaned re-queued the entire un-started backlog every 60 seconds. /// private async Task RedrivePendingTasksAsync(OrchestratorRun run) { @@ -1813,14 +1796,8 @@ private async Task RedrivePendingTasksAsync(OrchestratorRun run) return; } - // Storage decides — but the queue TABLE decides, not the index. Asking the index (the old - // GetQueuedTaskIdsAsync here) reports a task queued whenever its index row exists, and an index - // row can outlive the queue row it points at. Such a task is invisible to the pump yet looks - // "queued" to this check, so it is never re-driven and its run stalls indefinitely with the task - // Pending — this watchdog keeps ticking and finds nothing orphaned. GetDispatchableTaskIdsAsync - // verifies each candidate against the queue table (one point read apiece; the candidate set is - // small), returning only tasks the pump can actually still claim. Anything else is a ghost to - // re-enqueue. + // Storage decides — the queue TABLE, not the index, which can outlive the rows it points at (see + // GetDispatchableTaskIdsAsync). Anything the pump cannot still claim is a ghost to re-enqueue. // The sweep starts this for every live run each tick without awaiting it, so unguarded a slow // verification overlapped the next tick's for the same run and the reads piled up in-process. if (!_redriveInFlight.TryAdd(run.Name, 0)) return; @@ -1876,12 +1853,9 @@ private async Task RedrivePendingTasksAsync(OrchestratorRun run) // re-drive them. if (_redriveBackoffEnabled) _redriveBackoff[run.Name] = (now + _redriveBase, _redriveBase); - foreach (var task in orphaned) - { - // Clear the exhausted counter, or DeferTask would abandon it again on its first attempt. - _deferrals.TryRemove(DeferralKey(run.Name, task.Id), out _); - RequeueToTable(run, task); - } + // Clear the exhausted counters, or DeferTask would abandon them again on their first attempt. + foreach (var task in orphaned) _deferrals.TryRemove(DeferralKey(run.Name, task.Id), out _); + RequeueToTable(run, orphaned); _logger.LogWarning( "[Scheduler] Re-drove {Count} orphaned Pending task(s) in {Run} — no runnable queue row and not queued or running", @@ -2157,7 +2131,7 @@ private async Task FinalizeRunCoreAsync(OrchestratorRun run) // individual tasks up to 4 times each. The durable queue only ever carries TASKS — the // post-execution job is enqueued in-memory on the JobManager and, after a crash, is re-derived // from PostExecStatus — so dropping these rows here cannot cost the post-execution its retry. - _ = _queue.RemoveRunAsync(run.Name); + _ = _queue.RemoveRunAsync(run.Name, run.StartedUtc); // Dispatch PostExecution if configured if (!string.IsNullOrEmpty(run.PostExecFunctionName)) @@ -2249,7 +2223,7 @@ private void DispatchPostExecution(OrchestratorRun run) // Cleanup after successful PostExec await _store.CleanupRunAsync(run.Name); - await _queue.RemoveRunAsync(run.Name, jobCt); + await _queue.RemoveRunAsync(run.Name, run.StartedUtc, jobCt); } catch (Exception ex) { @@ -2517,7 +2491,9 @@ internal static void AddTaskFromElement(List tasks, HashSe /// public async Task<(bool found, int cancelledCount)> CancelRunAsync(string name) { - var run = await _store.GetRunAsync(name); + // The live graph when this node holds the run: cancelling a copy left the live tasks Pending, so + // the re-drive kept re-queueing them and the live run never finalized. + var run = _activeRuns.TryGetValue(name, out var live) ? live : await _store.GetRunAsync(name); if (run == null) return (false, 0); // Mark this run as cancelled so dispatched-but-not-yet-started tasks skip execution @@ -2559,6 +2535,8 @@ internal static void AddTaskFromElement(List tasks, HashSe } } + await _queue.RemoveRunAsync(run.Name, run.StartedUtc); + // Check if the run is now fully done (Running tasks will finalize themselves) var remaining = run.Tasks.Count(t => t.Status is "Running"); if (remaining == 0) diff --git a/Services/Storage/AzureTableStore.cs b/Services/Storage/AzureTableStore.cs index 476ce3b..9072934 100644 --- a/Services/Storage/AzureTableStore.cs +++ b/Services/Storage/AzureTableStore.cs @@ -463,6 +463,22 @@ public async IAsyncEnumerable QueryTableAsync(string table, string? fi yield return ToRow(entity); } + public async IAsyncEnumerable QueryTableAsync(string table, string? filter, int maxPerPage, + [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) + { + await foreach (var entity in StreamReassembledAsync(table, () => Client(table).QueryAsync(filter: filter, maxPerPage: maxPerPage, cancellationToken: ct), ct)) + yield return ToRow(entity); + } + + public async IAsyncEnumerable QueryRowKeyRangeAsync(string table, string partitionKey, string fromRowKey, + string toRowKey, IReadOnlyList? properties = null, + [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) + { + var filter = $"PartitionKey eq '{Escape(partitionKey)}' and RowKey ge '{Escape(fromRowKey)}' and RowKey lt '{Escape(toRowKey)}'"; + await foreach (var entity in EnumerateAsync(table, () => Client(table).QueryAsync(filter: filter, select: properties, cancellationToken: ct), ct)) + yield return ToRow(entity); + } + public async Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) { try diff --git a/Services/Storage/ICraftTableStore.cs b/Services/Storage/ICraftTableStore.cs index 2d7ab0d..436a03d 100644 --- a/Services/Storage/ICraftTableStore.cs +++ b/Services/Storage/ICraftTableStore.cs @@ -83,6 +83,26 @@ IAsyncEnumerable QueryTableAsync(string table, string? filter, IReadOn CancellationToken ct = default) => QueryTableAsync(table, filter, ct); + /// The filtered scan, fetched rows per request, for callers that + /// stop after the first few matches. Same contract as the filter: a backend may ignore both. + IAsyncEnumerable QueryTableAsync(string table, string? filter, int maxPerPage, + CancellationToken ct = default) + => QueryTableAsync(table, filter, ct); + + /// + /// Rows of one partition with <= RowKey < + /// (ordinal), optionally projected (name the keys too if you read them). Split entities are not + /// reassembled, so use it only on tables whose rows are never split. + /// + async IAsyncEnumerable QueryRowKeyRangeAsync(string table, string partitionKey, string fromRowKey, + string toRowKey, IReadOnlyList? properties = null, + [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) + { + await foreach (var row in QueryPartitionAsync(table, partitionKey, ct)) + if (string.CompareOrdinal(row.RowKey, fromRowKey) >= 0 && string.CompareOrdinal(row.RowKey, toRowKey) < 0) + yield return row; + } + /// Delete a single row. A missing row is not an error. Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default); diff --git a/Services/Storage/JobQueueStore.cs b/Services/Storage/JobQueueStore.cs index 3d0cc0b..de4c432 100644 --- a/Services/Storage/JobQueueStore.cs +++ b/Services/Storage/JobQueueStore.cs @@ -13,13 +13,13 @@ namespace Craft.Storage; /// Key design, and it is doing real work: /// /// PartitionKey "P04" — the priority bucket, zero-padded so it sorts numerically. -/// RowKey "{runEpochTicks:D19}|{run}|{task}" — oldest run first within a bucket, one row per task. +/// RowKey "{runEpochTicks:D19}|{run}|{task}" — the run's start time first, so a bucket drains oldest run +/// first, and one row per task because a run's start never +/// changes. A run's rows in a bucket are also contiguous, so +/// run-scoped queue reads are one RowKey range. /// -/// Azure Table returns rows ordered by partition key then row key, so a single unfiltered read yields -/// the highest-priority, oldest-first work FIRST, across every run. That is the cross-run priority -/// ordering the in-memory PriorityQueue provides today, for one round-trip and without probing each -/// bucket in turn — probing 16 buckets per refill would have cost more round-trips than the whole -/// batching exercise saves. +/// Azure Table returns rows ordered by partition key then row key, so a single read yields the +/// highest-priority, oldest-run work FIRST, across every run, for one round-trip. /// /// A claim is one conditional transaction over rows sharing a bucket: one round-trip per BATCH, not per /// task. Guarded by each row's ETag, so an external edit — or another instance claiming first — makes @@ -56,6 +56,7 @@ public sealed class JobQueueStore : IDisposable private readonly ICraftTableStore _store; private readonly string _queueTable; private readonly string _indexTable; + private readonly string _runsTable; private bool _initialized; /// @@ -98,22 +99,14 @@ public Task WaitForWorkAsync(TimeSpan pollInterval, CancellationToken ct = private const string SchemaPartition = "$schema"; private const string SchemaRowKey = "queue-index"; - /// - /// The row in a run's index partition holding the run's queue epoch: the time its first task was queued, - /// in whole seconds. Every queue key for the run is prefixed with it, so a bucket drains oldest run - /// first, and a re-enqueue of a task builds the same key however much later it happens. - /// - private const string EpochRowKey = "$epoch"; - /// /// Current on-disk schema version, applied once per storage account by : /// 1 — the run index exists (see the RUN INDEX note above). /// 2 — queue RowKeys are deterministic per (run, task) — {run}|{task} — instead of /// time-prefixed, so re-dispatching a task UPDATES its row instead of writing a second one /// (the duplicate-execution class). Enqueue time moves to the QueuedUtc property. - /// 3 — the key gains the run's epoch as a prefix — {epoch}|{run}|{task} — so a bucket drains - /// oldest run first instead of alphabetically by run name, which starved late-sorting runs. - /// Still deterministic per (run, task), because the epoch is stored once per run. + /// 3 — the key gains the run's start time as a prefix — {epoch}|{run}|{task} — so a bucket + /// drains oldest run first instead of alphabetically by run name, which starved late-sorting runs. /// A single forward migration takes any older account straight to this version; there is no /// backward-compatible dual-read — after the migration only the new key scheme is used. /// @@ -136,6 +129,7 @@ public JobQueueStore(ILogger logger, CraftSettings settings, ICra _store = store; _queueTable = $"{settings.Orchestrator.TablePrefix}Queue"; _indexTable = $"{settings.Orchestrator.TablePrefix}QueueIndex"; + _runsTable = $"{settings.Orchestrator.TablePrefix}Runs"; } public async Task InitializeAsync(CancellationToken ct = default) @@ -151,16 +145,26 @@ internal static string Bucket(int priority) => "P" + Math.Clamp(priority, 0, MaxPriorityBucket).ToString("D2", CultureInfo.InvariantCulture); /// - /// The queue RowKey: the run's epoch, then the run and task. Deterministic per (run, task) for a given - /// run epoch, so re-dispatching a task upserts its one row instead of writing a second (schema v2), and - /// ordered oldest run first within a bucket (schema v3). v2's bare {run}|{task} ordered a bucket - /// alphabetically, so a run whose name sorted late never ran while earlier-sorting runs kept arriving. - /// - /// Both components are escaped so the key is legal and the '|' separator is unambiguous; the row - /// itself carries RunName/TaskId as properties, so nothing needs to parse the key back apart. + /// The queue RowKey. The epoch is the run's start time, which every caller reads off the run, so the + /// key is the same on every enqueue of a task and a re-dispatch upserts its one row. /// - internal static string BuildRowKey(DateTime runEpochUtc, string runName, string taskId) => - $"{runEpochUtc.Ticks.ToString("D19", CultureInfo.InvariantCulture)}|{EscapeKeyComponent(runName)}|{EscapeKeyComponent(taskId)}"; + internal static string BuildRowKey(DateTime runStartedUtc, string runName, string taskId) => + $"{RunKeyPrefix(runStartedUtc, runName)}{EscapeKeyComponent(taskId)}"; + + /// The shared start of every queue key of one run: {epoch}|{run}|. + private static string RunKeyPrefix(DateTime runStartedUtc, string runName) => + $"{EpochOf(runStartedUtc).Ticks.ToString("D19", CultureInfo.InvariantCulture)}|{EscapeKeyComponent(runName)}|"; + + /// The run prefix of a v3 queue key, or null for an older key. + private static string? RunKeyPrefixOf(string rowKey) + { + if (ParseEpoch(rowKey) == null) return null; + var end = rowKey.IndexOf('|', 20); + return end < 0 ? null : rowKey[..(end + 1)]; + } + + /// Everything that starts with : '|' is followed by '}' in ordinal order. + private static string PrefixUpperBound(string prefix) => prefix[..^1] + '}'; /// Whole seconds, so an epoch builds the same key before and after a storage round trip. internal static DateTime EpochOf(DateTime utc) => @@ -176,29 +180,6 @@ internal static DateTime EpochOf(DateTime utc) => return new DateTime(ticks, DateTimeKind.Utc); } - /// - /// The run's stored epoch, or recorded as it on the run's first enqueue. - /// ponytail: read-then-write rather than insert-if-absent, so two concurrent FIRST enqueues of one run - /// could record different epochs. A run is dispatched from one place and both dispatch and re-drive - /// check the index before enqueuing, so this is a narrow window, not a duplicate path. - /// - private async Task RunEpochAsync(string runName, DateTime queuedUtc, CancellationToken ct) - { - var stored = (await _store.GetAsync(_indexTable, IndexPartition(runName), EpochRowKey, ct)) - ?.GetDateTimeOffset("EpochUtc"); - if (stored != null) return EpochOf(stored.Value.UtcDateTime); - - var epoch = EpochOf(queuedUtc); - await _store.UpsertAsync(_indexTable, EpochRow(runName, epoch), ct); - return epoch; - } - - private static StoreRow EpochRow(string runName, DateTime epoch) => - new(IndexPartition(runName), EpochRowKey) - { - Properties = { ["RunName"] = runName, ["EpochUtc"] = new DateTimeOffset(epoch, TimeSpan.Zero) } - }; - /// /// Escape a run or task id for use inside a queue RowKey: the Azure-illegal key characters plus '|' /// (the separator) and '%' (the escape marker itself), percent-encoded. Reversible and injective, so @@ -281,58 +262,31 @@ private static StoreRow IndexRow(string runName, string taskId, string bucket, s Properties = { ["TaskId"] = taskId, ["RunName"] = runName } }; - /// Add one task to the queue. Idempotent per (run, task): the key comes from the run's stored epoch. - /// - /// Queue row first, index row second. The queue row is what makes the task actually run; the index - /// only accelerates lookups. If the process dies between the two the task still executes, and the - /// missing index entry is repaired by the next enqueue of the same (queuedUtc, run, task), which - /// rewrites both keys unchanged. The other order would leave the index claiming a task is queued - /// when no row exists — the orphan re-drive trusts the index, would decline to re-queue, and the - /// run would sit Pending with nothing running. - /// - public async Task EnqueueAsync(string runName, string taskId, int priority, DateTime queuedUtc, - CancellationToken ct = default) - { - var bucket = Bucket(priority); - var rowKey = BuildRowKey(await RunEpochAsync(runName, queuedUtc, ct), runName, taskId); - - await _store.UpsertAsync(_queueTable, new StoreRow(bucket, rowKey) - { - Properties = - { - ["RunName"] = runName, - ["TaskId"] = taskId, - ["Priority"] = priority, - ["Owner"] = "", - ["LeaseUntil"] = (DateTimeOffset?)null, - // Enqueue time is a property now that it is no longer in the key (schema v2), so age and - // status reporting keep working while the key stays deterministic per (run, task). - ["QueuedUtc"] = new DateTimeOffset(queuedUtc, TimeSpan.Zero), - } - }, ct); - - await _store.UpsertAsync(_indexTable, IndexRow(runName, taskId, bucket, rowKey), ct); - - WakePump(); - } + /// Add one task to the queue. Idempotent per (run, task). + public Task EnqueueAsync(string runName, string taskId, int priority, DateTime runStartedUtc, + CancellationToken ct = default) => + EnqueueBatchAsync(runName, [(taskId, priority)], runStartedUtc, ct); - /// Queue many tasks for one run. Chunked by the caller's priority into per-bucket batches. + /// Queue tasks of one run, keyed by the run's start time and grouped into per-bucket batches. /// - /// The index rows for one run all share a partition, so however many buckets the tasks span the - /// index costs exactly one transaction. Ordering is as . + /// Queue rows first, index rows second. The queue row is what makes the task actually run; the index + /// only accelerates lookups. If the process dies between the two the task still executes, and the + /// missing index entry is repaired by the next enqueue of the same task, which rewrites both keys + /// unchanged. The other order would leave the index claiming a task is queued when no row exists — the + /// orphan re-drive trusts the index, would decline to re-queue, and the run would sit Pending with + /// nothing running. The index rows share the run's partition, so they cost one transaction. /// public async Task EnqueueBatchAsync(string runName, IReadOnlyList<(string TaskId, int Priority)> tasks, - DateTime queuedUtc, CancellationToken ct = default) + DateTime runStartedUtc, CancellationToken ct = default) { if (tasks.Count == 0) return; var indexRows = new List(tasks.Count); - var epoch = await RunEpochAsync(runName, queuedUtc, ct); + var queuedOffset = new DateTimeOffset(DateTime.SpecifyKind(runStartedUtc, DateTimeKind.Utc)); foreach (var byBucket in tasks.GroupBy(t => Bucket(t.Priority))) { - var queuedOffset = new DateTimeOffset(queuedUtc, TimeSpan.Zero); - var rows = byBucket.Select(t => new StoreRow(byBucket.Key, BuildRowKey(epoch, runName, t.TaskId)) + var rows = byBucket.Select(t => new StoreRow(byBucket.Key, BuildRowKey(runStartedUtc, runName, t.TaskId)) { Properties = { @@ -347,12 +301,10 @@ public async Task EnqueueBatchAsync(string runName, IReadOnlyList<(string TaskId await _store.UpsertBatchAsync(_queueTable, byBucket.Key, rows, ct); - indexRows.AddRange(byBucket.Select(t => - IndexRow(runName, t.TaskId, byBucket.Key, BuildRowKey(epoch, runName, t.TaskId)))); + indexRows.AddRange(rows.Select(r => IndexRow(runName, r.GetString("TaskId")!, byBucket.Key, r.RowKey))); } - if (indexRows.Count > 0) - await _store.UpsertBatchAsync(_indexTable, IndexPartition(runName), indexRows, ct); + await _store.UpsertBatchAsync(_indexTable, IndexPartition(runName), indexRows, ct); WakePump(); } @@ -387,7 +339,7 @@ public async Task> ClaimBatchAsync(string owner, int m // paged to the client on every pump tick just to find the few free rows at its head. It is an // optimisation ONLY — a store that ignores it still returns everything — so IsClaimable below // stays as the authority. Nothing here may assume the filter was applied. - await foreach (var row in _store.QueryTableAsync(_queueTable, ClaimableFilter(now), ct)) + await foreach (var row in _store.QueryTableAsync(_queueTable, ClaimableFilter(now), max, ct)) { if (!IsClaimable(row, now)) continue; @@ -548,14 +500,27 @@ public async Task RenewAsync(IReadOnlyList jobs, string owner, return ok; } - /// Drop every queued row for a run — used when a run is cancelled or cleaned up. - public async Task RemoveRunAsync(string runName, CancellationToken ct = default) + /// + /// Drop a run's queued rows. With , only that outing's rows: a recurring + /// run name shares one index partition across outings, and a late removal of the previous outing must + /// not take the next one's rows with it. + /// + public async Task RemoveRunAsync(string runName, DateTime? runStartedUtc = null, CancellationToken ct = default) { - foreach (var e in await ReadIndexAsync(runName, ct)) - await _store.DeleteAsync(_queueTable, e.Bucket, e.QueueRowKey, ct); + var entries = await ReadIndexAsync(runName, ct); + if (runStartedUtc is { } started) + { + var prefix = RunKeyPrefix(started, runName); + entries = entries.Where(e => e.QueueRowKey.StartsWith(prefix, StringComparison.Ordinal)).ToList(); + } + + foreach (var byBucket in entries.GroupBy(e => e.Bucket)) + await _store.DeleteBatchAsync(_queueTable, byBucket.Key, byBucket.Select(e => e.QueueRowKey).ToList(), ct); - // One call, and it also takes any entry whose queue row was already gone. - await _store.DeletePartitionAsync(_indexTable, IndexPartition(runName), ct); + if (runStartedUtc == null) + await _store.DeletePartitionAsync(_indexTable, IndexPartition(runName), ct); + else if (entries.Count > 0) + await _store.DeleteBatchAsync(_indexTable, IndexPartition(runName), entries.Select(e => e.IndexRowKey).ToList(), ct); } /// @@ -625,16 +590,17 @@ public async Task ReleaseRunClaimsAsync(string runName, CancellationToken c { var released = 0; - foreach (var e in await ReadIndexAsync(runName, ct)) + foreach (var (bucket, rows) in await ReadRunQueueRowsAsync(await ReadIndexAsync(runName, ct), null, ct)) { - var row = await _store.GetAsync(_queueTable, e.Bucket, e.QueueRowKey, ct); - if (row == null) continue; // finished and removed - if (string.IsNullOrEmpty(row.GetString("Owner"))) continue; // already free - - row["Owner"] = ""; - row["LeaseUntil"] = (DateTimeOffset?)null; - await _store.UpsertAsync(_queueTable, row, ct); - released++; + var owned = rows.Where(r => !string.IsNullOrEmpty(r.GetString("Owner"))).ToList(); + foreach (var row in owned) + { + row["Owner"] = ""; + row["LeaseUntil"] = (DateTimeOffset?)null; + } + if (owned.Count == 0) continue; + await _store.UpsertBatchAsync(_queueTable, bucket, owned, ct); + released += owned.Count; } // Freed claims are claimable again — wake the pump to pick them up rather than waiting for the @@ -644,6 +610,42 @@ public async Task ReleaseRunClaimsAsync(string runName, CancellationToken c return released; } + /// Above this many tasks a run's queue rows are read as one key range rather than one GET each. + private const int PointReadLimit = 32; + + /// + /// The queue rows behind , per bucket. A few are point reads; more are one + /// range read over the run's contiguous keys, so the cost follows the run's size, never the queue's. + /// + private async Task Rows)>> ReadRunQueueRowsAsync( + List<(string TaskId, string Bucket, string QueueRowKey, string IndexRowKey)> entries, + IReadOnlyList? properties, CancellationToken ct) + { + var result = new List<(string, List)>(); + foreach (var byBucket in entries.GroupBy(e => e.Bucket)) + { + var rows = new List(); + var keys = byBucket.Select(e => e.QueueRowKey).ToHashSet(StringComparer.Ordinal); + var prefixes = keys.Select(RunKeyPrefixOf).Distinct().ToList(); + + if (keys.Count <= PointReadLimit || prefixes.Contains(null)) + { + foreach (var key in keys) + if (await _store.GetAsync(_queueTable, byBucket.Key, key, ct) is { } row) rows.Add(row); + } + else + { + foreach (var prefix in prefixes) + await foreach (var row in _store.QueryRowKeyRangeAsync(_queueTable, byBucket.Key, prefix!, + PrefixUpperBound(prefix!), properties, ct)) + if (keys.Contains(row.RowKey)) rows.Add(row); + } + + result.Add((byBucket.Key, rows)); + } + return result; + } + /// A queued row as the status APIs see it: identity, priority, age and claim state. public sealed record QueuedRow(string RunName, string TaskId, int Priority, DateTime QueuedUtc, bool Claimed, string Owner, string Bucket, string RowKey); @@ -657,23 +659,23 @@ private static DateTime QueuedUtcOf(StoreRow row) => ?? ParseLegacyQueuedUtc(row.RowKey) ?? DateTime.UtcNow; + /// A queue row's columns. Keys are named because a projection returns only what it lists. + private static readonly string[] s_queuedRowProperties = + ["PartitionKey", "RowKey", "RunName", "TaskId", "Priority", "Owner", "LeaseUntil", "QueuedUtc"]; + /// - /// Every row currently in the queue, in storage order (highest priority bucket first, oldest first - /// within it). This is the durable backlog the status APIs report — the in-memory JobManager only - /// ever holds a worker-pool-sized buffer of it. - /// - /// Claimed means owned under a live lease, i.e. buffered or running on some instance; everything - /// else is waiting for a pump to take it. One unfiltered scan, so the cost is proportional to the - /// backlog — callers are expected to cache the result rather than call this per poll. + /// Every row currently in the queue, streamed in storage order (highest priority bucket first, oldest + /// run first within it) and projected to what the status APIs read. Claimed means owned under a live + /// lease. One scan, proportional to the backlog, so callers aggregate as it streams and cache the + /// result rather than holding the rows or calling this per poll. /// - public async Task> ListQueuedAsync(CancellationToken ct = default) + public async IAsyncEnumerable StreamQueuedAsync( + [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) { var now = DateTimeOffset.UtcNow; - var rows = new List(); - - await foreach (var row in _store.QueryTableAsync(_queueTable, ct)) + await foreach (var row in _store.QueryTableAsync(_queueTable, null, s_queuedRowProperties, ct)) { - rows.Add(new QueuedRow( + yield return new QueuedRow( row.GetString("RunName") ?? "", row.GetString("TaskId") ?? "", row.GetInt32("Priority") ?? 0, @@ -681,9 +683,15 @@ public async Task> ListQueuedAsync(CancellationToken ct !IsClaimable(row, now), row.GetString("Owner") ?? "", row.PartitionKey, - row.RowKey)); + row.RowKey); } + } + /// materialized, for small queues and tests. + public async Task> ListQueuedAsync(CancellationToken ct = default) + { + var rows = new List(); + await foreach (var row in StreamQueuedAsync(ct)) rows.Add(row); return rows; } @@ -710,8 +718,8 @@ public async Task RemoveTaskAsync(string runName, string taskId, Cancellati } /// - /// Move a task's queue rows to a new priority bucket, keeping their enqueue timestamp so the task - /// keeps its place in line within the new priority. Returns how many rows moved. + /// Move a task's queue rows to a new priority bucket, keeping the run's epoch so the task keeps its + /// place in line within the new priority. Returns how many rows moved. /// /// Delete-then-add, in that order: a crash in between loses the row, which the orphan re-drive /// repairs by re-queueing the task. The other order leaves TWO claimable rows for one task, and a @@ -745,8 +753,7 @@ public async Task ReprioritizeTaskAsync(string runName, string taskId, int await _store.DeleteAsync(_indexTable, partition, IndexRowKey(row.PartitionKey, row.RowKey), ct); await _store.DeleteAsync(_queueTable, row.PartitionKey, row.RowKey, ct); - var queuedUtc = QueuedUtcOf(row); - await EnqueueAsync(runName, taskId, newPriority, queuedUtc, ct); + await EnqueueAsync(runName, taskId, newPriority, ParseEpoch(row.RowKey) ?? QueuedUtcOf(row), ct); moved++; } @@ -760,9 +767,7 @@ public async Task ReprioritizeTaskAsync(string runName, string taskId, int /// the run graph, sitting in this queue, not yet claimed by the pump — and one whose row is /// genuinely gone. Under the pump, waiting is the normal state of a backlog: a 124-task run against /// eight workers has most of its tasks Pending and absent from the JobManager for minutes at a time. - /// Treating that as orphaned re-queues the whole backlog on a timer, and because a RowKey is - /// prefixed with the enqueue timestamp, each pass adds a SECOND row for the same task rather than - /// updating the first — so the task is claimed and executed once per copy. + /// Treating that as orphaned re-queues the whole backlog on a timer. /// public async Task> GetQueuedTaskIdsAsync(string runName, CancellationToken ct = default) { @@ -799,8 +804,8 @@ public async Task> GetQueuedTaskIdsAsync(string runName, Cancell /// /// A task in any of those states is invisible to the pump AND reported "queued" by the index, so the /// re-drive that trusts the index never re-enqueues it and the run stalls indefinitely with it, its - /// watchdog never firing. Verifying against the queue table costs one point read per id, so the caller - /// passes a SMALL candidate set (the re-drive's aged-Pending tasks), never the whole run. + /// watchdog never firing. One index read plus , so the cost follows + /// the run's size. /// public async Task> GetDispatchableTaskIdsAsync( string runName, IReadOnlyCollection taskIds, CancellationToken ct = default) @@ -809,25 +814,19 @@ public async Task> GetDispatchableTaskIdsAsync( if (taskIds.Count == 0) return result; var wanted = taskIds as HashSet ?? new HashSet(taskIds, StringComparer.Ordinal); - var now = DateTimeOffset.UtcNow; + var entries = (await ReadIndexAsync(runName, ct)).Where(e => wanted.Contains(e.TaskId)).ToList(); - // The index carries the bucket + queue row key to address each row directly — one partition read, - // then a point read only for the ids the caller asked about. - foreach (var e in await ReadIndexAsync(runName, ct)) + foreach (var (_, rows) in await ReadRunQueueRowsAsync(entries, s_queuedRowProperties, ct)) { - if (!wanted.Contains(e.TaskId)) continue; - - var row = await _store.GetAsync(_queueTable, e.Bucket, e.QueueRowKey, ct); - if (row == null) continue; // index points at a queue row that is gone — a ghost - - // Owned with no lease is what the server-side claim filter cannot match (it is neither - // Owner eq '' nor LeaseUntil lt now), so the pump would never dispatch it however long it - // waits — a ghost as surely as a missing row. A free row, or one under a lease live or - // lapsed, the pump will get. - if (!string.IsNullOrEmpty(row.GetString("Owner")) && row.GetDateTimeOffset("LeaseUntil") == null) - continue; - - result.Add(e.TaskId); + foreach (var row in rows) + { + // Owned with no lease is what the server-side claim filter cannot match (it is neither + // Owner eq '' nor LeaseUntil lt now), so the pump would never dispatch it — a ghost as + // surely as a missing row. + if (!string.IsNullOrEmpty(row.GetString("Owner")) && row.GetDateTimeOffset("LeaseUntil") == null) + continue; + if (row.GetString("TaskId") is { } taskId) result.Add(taskId); + } } return result; @@ -837,14 +836,12 @@ public async Task> GetDispatchableTaskIdsAsync( /// Bring the queue tables up to , once per storage account. The marker row /// written at the end is checked first, so every later start is a single point read. /// - /// v1 built the run index; v2 additionally re-keys every queue row to the deterministic - /// {run}|{task} scheme () and moves the enqueue timestamp into the - /// QueuedUtc property. One forward pass takes any older account straight to the current - /// version — there is no dual-read, and after the pass only the new key scheme is used. + /// Re-keys every queue row to , taking each run's start time from the Runs + /// table, and rebuilds the index entries. One forward pass takes any older account straight to the + /// current version — there is no dual-read, and after the pass only the new key scheme is used. /// /// Rows are rewritten new-key-first, old-key-deleted-after, so a crash mid-pass leaves the marker - /// unset and the next start finishes the job (a row already in the new scheme is re-written harmlessly - /// and not deleted). The pump awaits before it claims — and every + /// unset and the next start finishes the job, skipping rows already on their key. The pump awaits before it claims — and every /// enqueue path calls it too — so no row is ever claimed or written while the migration is only /// half-applied. /// @@ -869,6 +866,9 @@ private async Task MigrateSchemaAsync(CancellationToken ct) var rekeyed = 0; var skipped = 0; var epochs = new Dictionary(StringComparer.Ordinal); + await foreach (var run in _store.QueryTableAsync(_runsTable, "PartitionKey eq 'Run'", ["PartitionKey", "RowKey", "StartedUtc"], ct)) + if (run.PartitionKey == "Run" && run.GetDateTimeOffset("StartedUtc") is { } runStarted) + epochs[run.RowKey] = runStarted.UtcDateTime; static void AddRow(Dictionary> map, string key, StoreRow row) { @@ -912,15 +912,19 @@ async Task FlushAsync() var bucket = row.PartitionKey; var runPartition = IndexPartition(runName); - // One epoch per run. Rows already on a v3 key (a re-run after a crash) sort ahead of the run's - // unmigrated rows in the scan, so their epoch is seen first and the rest follow it. + // The run's start time, as every later enqueue will key it. A run with no Run row falls back to + // the first of its rows the scan meets, and keeps that for the rest of them. if (!epochs.TryGetValue(runName, out var epoch)) + epochs[runName] = epoch = ParseEpoch(row.RowKey) ?? QueuedUtcOf(row); + var newKey = BuildRowKey(epoch, runName, taskId); + buffered++; + + // Already on its key (a re-run after a crash): only make sure the index points at it. + if (row.RowKey == newKey) { - epoch = ParseEpoch(row.RowKey) ?? EpochOf(QueuedUtcOf(row)); - epochs[runName] = epoch; - AddRow(newIndex, runPartition, EpochRow(runName, epoch)); + AddRow(newIndex, runPartition, IndexRow(runName, taskId, bucket, newKey)); + continue; } - var newKey = BuildRowKey(epoch, runName, taskId); // The new-scheme row: same bucket + claim state + priority, key deterministic, enqueue time as // a property (from the row, or the legacy key, or now). @@ -937,16 +941,9 @@ async Task FlushAsync() } }); AddRow(newIndex, runPartition, IndexRow(runName, taskId, bucket, newKey)); - - // Delete the legacy row + index entry, UNLESS its key is already the new scheme (a re-run over - // already-migrated rows just re-writes them — deleting would drop what we just wrote). - if (row.RowKey != newKey) - { - rekeyed++; - AddKey(oldQueue, bucket, row.RowKey); - AddKey(oldIndex, runPartition, IndexRowKey(bucket, row.RowKey)); - } - buffered++; + rekeyed++; + AddKey(oldQueue, bucket, row.RowKey); + AddKey(oldIndex, runPartition, IndexRowKey(bucket, row.RowKey)); if (buffered >= BackfillFlushThreshold) { diff --git a/tests/Craft.Tests/JobQueueAzuriteTests.cs b/tests/Craft.Tests/JobQueueAzuriteTests.cs index 074461d..9b34da8 100644 --- a/tests/Craft.Tests/JobQueueAzuriteTests.cs +++ b/tests/Craft.Tests/JobQueueAzuriteTests.cs @@ -119,6 +119,85 @@ public async Task ServerSideRunFilter_ScopesToOneRun() Assert.Equal(["b1"], await queue.GetQueuedTaskIdsAsync("run-b")); } + /// + /// A large run's queue rows are read as one RowKey range, built from the run's key prefix. The fakes + /// filter ranges in memory, so only here is the service's own range filter — and the page-sized claim + /// beside it — evaluated. A wrong bound is silent: the run's work reads as gone, or a neighbour's rows + /// leak in. + /// + [Fact] + public async Task RunKeyRange_ScopesToOneRun_AndAPageSizedClaimStillTakesTheHead() + { + var queue = await TryConnectAsync(); + if (queue == null) return; + + var ids = Enumerable.Range(0, 40).Select(i => $"t{i:D2}").ToList(); + await queue.EnqueueBatchAsync("Big", ids.Select(i => (i, 4)).ToList(), At(1)); + await queue.EnqueueBatchAsync("Bigger", [("x", 4)], At(2)); + await queue.EnqueueBatchAsync("Early", [("e", 4)], At(0)); + + Assert.Equal(ids, (await queue.GetDispatchableTaskIdsAsync("Big", ids)).Order()); + + var claimed = await queue.ClaimBatchAsync("dead", 3, TimeSpan.FromMinutes(30)); + Assert.Equal(["e", "t00", "t01"], claimed.Select(c => c.TaskId)); + + var head = await queue.ListQueuedAsync(); + Assert.Equal(42, head.Count); + Assert.All(head, r => Assert.StartsWith("P04", r.Bucket)); + Assert.Equal(3, head.Count(r => r.Claimed)); + + Assert.Equal(2, await queue.ReleaseRunClaimsAsync("Big")); + await queue.RemoveRunAsync("Big", At(1)); + Assert.Empty(await queue.GetQueuedTaskIdsAsync("Big")); + Assert.Equal(["x"], await queue.GetQueuedTaskIdsAsync("Bigger")); + } + + /// + /// The v3 migration keys each run by the StartedUtc on its Run row, read with a projection. A projection + /// returns only the columns it names, so a missing key column here reads every run as unknown and keys + /// it off its queue rows instead — silently, since the fakes return whole rows. + /// + [Fact] + public async Task Migration_KeysARunByItsRunRowsStartTime() + { + var settings = new CraftSettings { Storage = { AllowDevelopmentStorage = true } }; + var connection = Environment.GetEnvironmentVariable("CRAFT_TEST_TABLE_CONNECTION"); + if (!string.IsNullOrWhiteSpace(connection)) settings.Auth.UserStorageConnection = connection; + settings.Orchestrator.TablePrefix = "azqm" + Guid.NewGuid().ToString("N")[..8]; + var store = new AzureTableStore(settings); + try + { + using var cts = new CancellationTokenSource(TimeSpan.FromSeconds(3)); + await store.PingAsync(cts.Token); + } + catch { return; } + + var prefix = settings.Orchestrator.TablePrefix; + var started = At(0).AddDays(-3); + await store.EnsureTableAsync($"{prefix}Runs"); + await store.EnsureTableAsync($"{prefix}Queue"); + await store.EnsureTableAsync($"{prefix}QueueIndex"); + await store.UpsertAsync($"{prefix}Runs", new StoreRow("Run", "MailboxRules_t1") + { + Properties = { ["StartedUtc"] = new DateTimeOffset(started) } + }); + await store.UpsertAsync($"{prefix}Queue", new StoreRow("P04", "MailboxRules_t1|b1") + { + Properties = + { + ["RunName"] = "MailboxRules_t1", ["TaskId"] = "b1", ["Priority"] = 4, ["Owner"] = "", + ["QueuedUtc"] = new DateTimeOffset(At(0)), + } + }); + await store.UpsertAsync($"{prefix}QueueIndex", new StoreRow("$schema", "queue-index") { Properties = { ["Version"] = 2 } }); + + var queue = new JobQueueStore(NullLogger.Instance, settings, store); + await queue.InitializeAsync(); + + var row = Assert.Single(await queue.ListQueuedAsync()); + Assert.Equal(JobQueueStore.BuildRowKey(started, "MailboxRules_t1", "b1"), row.RowKey); + } + /// A run name containing a quote must not break the filter or leak into it. [Fact] public async Task ServerSideRunFilter_HandlesAQuoteInTheRunName() diff --git a/tests/Craft.Tests/JobQueueFifoTests.cs b/tests/Craft.Tests/JobQueueFifoTests.cs index b731041..c6ca829 100644 --- a/tests/Craft.Tests/JobQueueFifoTests.cs +++ b/tests/Craft.Tests/JobQueueFifoTests.cs @@ -16,14 +16,16 @@ public class JobQueueFifoTests private static readonly TimeSpan Lease = TimeSpan.FromMinutes(20); private static DateTime At(int minute) => new(2026, 10, 5, 2, minute, 0, DateTimeKind.Utc); - private sealed record Q(JobQueueStore Queue, RunRemainingCounterTests.ConditionalStore Store, string QueueTable, string IndexTable); + private sealed record Q(JobQueueStore Queue, RunRemainingCounterTests.ConditionalStore Store, string QueueTable, + string IndexTable, string RunsTable); private static Q NewQueue() { var settings = new CraftSettings(); var store = new RunRemainingCounterTests.ConditionalStore(); var queue = new JobQueueStore(NullLogger.Instance, settings, store); - return new Q(queue, store, $"{settings.Orchestrator.TablePrefix}Queue", $"{settings.Orchestrator.TablePrefix}QueueIndex"); + var prefix = settings.Orchestrator.TablePrefix; + return new Q(queue, store, $"{prefix}Queue", $"{prefix}QueueIndex", $"{prefix}Runs"); } private static async Task> DrainOrderAsync(JobQueueStore queue) @@ -79,28 +81,69 @@ public async Task AReEnqueueHoursLater_KeepsTheRunsPlaceAndOneRow() await q.Queue.EnqueueBatchAsync("Zeta", [("z0", 4), ("z1", 4)], At(0)); await q.Queue.EnqueueBatchAsync("Alpha", [("a0", 4)], At(10)); - // The re-drive re-queues a task with the time it noticed, long after the run started. - await q.Queue.EnqueueAsync("Zeta", "z1", 4, At(59)); + // The re-drive re-queues with the run's start time, however long after the run started. + await q.Queue.EnqueueAsync("Zeta", "z1", 4, At(0)); Assert.Equal(3, (await RowsAsync(q, q.QueueTable)).Count); Assert.Equal(["Zeta/z0", "Zeta/z1", "Alpha/a0"], await DrainOrderAsync(q.Queue)); } [Fact] - public async Task TheEpochRowIsInvisibleToIndexReaders_AndGoesWithTheRun() + public async Task RemovingOneOutingOfARecurringRun_LeavesTheNextOutingsRows() { var q = NewQueue(); await q.Queue.InitializeAsync(); - await q.Queue.EnqueueBatchAsync("R", [("a", 4)], At(0)); + await q.Queue.EnqueueBatchAsync("CIPPDBCacheOrchestrator", [("old", 4)], At(0)); + await q.Queue.EnqueueBatchAsync("CIPPDBCacheOrchestrator", [("new", 4)], At(30)); - Assert.Equal(["a"], await q.Queue.GetQueuedTaskIdsAsync("R")); - Assert.Equal(["a"], await q.Queue.GetDispatchableTaskIdsAsync("R", ["a"])); - Assert.Equal(0, await q.Queue.ReleaseRunClaimsAsync("R")); + // The previous outing's finalize removes its rows after the next outing has enqueued. + await q.Queue.RemoveRunAsync("CIPPDBCacheOrchestrator", At(0)); - await q.Queue.RemoveRunAsync("R"); + Assert.Equal(["new"], await q.Queue.GetQueuedTaskIdsAsync("CIPPDBCacheOrchestrator")); + Assert.Equal("new", Assert.Single(await RowsAsync(q, q.QueueTable)).GetString("TaskId")); + await q.Queue.RemoveRunAsync("CIPPDBCacheOrchestrator"); Assert.Empty(await RowsAsync(q, q.QueueTable)); - Assert.DoesNotContain(await RowsAsync(q, q.IndexTable), r => r.PartitionKey == "R"); + Assert.DoesNotContain(await RowsAsync(q, q.IndexTable), r => r.PartitionKey == "CIPPDBCacheOrchestrator"); + } + + [Fact] + public async Task ALargeRunsDispatchCheck_ReadsItsKeyRange_NotOneRowAtATime() + { + var q = NewQueue(); + await q.Queue.InitializeAsync(); + var ids = Enumerable.Range(0, 40).Select(i => $"t{i:D2}").ToList(); + await q.Queue.EnqueueBatchAsync("Big", ids.Select(i => (i, 4)).ToList(), At(0)); + await q.Queue.EnqueueBatchAsync("Other", [("o", 4)], At(1)); + + var rows = await RowsAsync(q, q.QueueTable); + // A ghost (index entry, no queue row) and an owned row with no lease, which the pump never claims. + await q.Store.DeleteAsync(q.QueueTable, "P04", rows.Single(r => r.GetString("TaskId") == "t00").RowKey); + var stuck = rows.Single(r => r.GetString("TaskId") == "t01"); + stuck["Owner"] = "gone"; + await q.Store.UpsertAsync(q.QueueTable, stuck); + + q.Store.Gets.Clear(); + var dispatchable = await q.Queue.GetDispatchableTaskIdsAsync("Big", ids); + + Assert.Equal(ids.Skip(2), dispatchable.Order()); + Assert.False(q.Store.Gets.ContainsKey(q.QueueTable)); + } + + [Fact] + public async Task ReleasingALargeRunsClaims_FreesEveryRowOfThatRunOnly() + { + var q = NewQueue(); + await q.Queue.InitializeAsync(); + await q.Queue.EnqueueBatchAsync("Big", Enumerable.Range(0, 40).Select(i => ($"t{i:D2}", 4)).ToList(), At(0)); + await q.Queue.EnqueueBatchAsync("Other", [("o", 4)], At(1)); + Assert.Equal(41, (await q.Queue.ClaimBatchAsync("dead", 100, Lease)).Count); + + q.Store.Gets.Clear(); + Assert.Equal(40, await q.Queue.ReleaseRunClaimsAsync("Big")); + + Assert.False(q.Store.Gets.ContainsKey(q.QueueTable)); + Assert.Equal(40, (await q.Queue.ClaimBatchAsync("next", 100, Lease)).Count); } [Fact] @@ -144,7 +187,13 @@ private static async Task SeedV2Async(Q q, string run, string task, DateTime que public async Task MigratesAV2Backlog_ToOldestRunFirst_KeepingClaimsAndTheIndex() { var q = NewQueue(); - // Alphabetically the v2 order was AuditLog, MailboxRules; by age MailboxRules is 16 days older. + // Alphabetically the v2 order was AuditLog, MailboxRules; by age MailboxRules is 16 days older. Its + // Run row holds the start time later enqueues key by; AuditLog has none and falls back to its rows. + var mailboxStarted = At(0).AddDays(-17); + await q.Store.UpsertAsync(q.RunsTable, new StoreRow("Run", "MailboxRules_t1") + { + Properties = { ["StartedUtc"] = new DateTimeOffset(mailboxStarted) } + }); await SeedV2Async(q, "MailboxRules_t1", "b1", At(0).AddDays(-16)); await SeedV2Async(q, "MailboxRules_t1", "b2", At(0).AddDays(-16)); await SeedV2Async(q, "AuditLog_t1", "s1", At(0)); @@ -163,10 +212,10 @@ public async Task MigratesAV2Backlog_ToOldestRunFirst_KeepingClaimsAndTheIndex() // Index entries point at the new keys: every run-scoped read still sees its tasks as dispatchable. Assert.Equal(["b1", "b2"], (await q.Queue.GetDispatchableTaskIdsAsync("MailboxRules_t1", ["b1", "b2"])).Order()); Assert.Equal(["s1", "s2"], (await q.Queue.GetDispatchableTaskIdsAsync("AuditLog_t1", ["s1", "s2"])).Order()); - Assert.Equal(6, (await RowsAsync(q, q.IndexTable)).Count(r => r.PartitionKey != "$schema")); + Assert.Equal(4, (await RowsAsync(q, q.IndexTable)).Count(r => r.PartitionKey != "$schema")); - // A later enqueue of a migrated task lands on its existing row, not a second one. - await q.Queue.EnqueueAsync("MailboxRules_t1", "b2", 4, DateTime.UtcNow); + // A later enqueue of a migrated task, keyed by the run's start, lands on its existing row. + await q.Queue.EnqueueAsync("MailboxRules_t1", "b2", 4, mailboxStarted); Assert.Equal(4, (await RowsAsync(q, q.QueueTable)).Count); Assert.Equal(["MailboxRules_t1/b1", "MailboxRules_t1/b2", "AuditLog_t1/s1"], await DrainOrderAsync(q.Queue)); diff --git a/tests/Craft.Tests/JobQueueStatusReaderTests.cs b/tests/Craft.Tests/JobQueueStatusReaderTests.cs index 0485c44..0dea7d8 100644 --- a/tests/Craft.Tests/JobQueueStatusReaderTests.cs +++ b/tests/Craft.Tests/JobQueueStatusReaderTests.cs @@ -72,6 +72,25 @@ await f.Queue.EnqueueBatchAsync("StandardsApply", Assert.Equal(At(5), summary.OldestQueuedUtc); } + [Fact] + public async Task ASnapshotHoldsOnlyTheHeadOfABigQueue_ButCountsAllOfIt() + { + var f = await NewFixtureAsync(); + var total = JobQueueStatusReader.HeadRows + 500; + await f.Queue.EnqueueBatchAsync("Late", [("l", 4)], At(9)); + await f.Queue.EnqueueBatchAsync("Big", + Enumerable.Range(0, total - 1).Select(i => ($"b{i:D5}", 4)).ToList(), At(1)); + + var snap = await f.Reader.GetAsync(Fresh); + + Assert.Equal(JobQueueStatusReader.HeadRows, snap!.Rows.Count); + Assert.All(snap.Rows, r => Assert.Equal("Big", r.RunName)); + Assert.Equal(total, snap.Total); + Assert.Equal(total, snap.Unclaimed); + Assert.Equal(total - 1, snap.ByRun["Big"].Unclaimed); + Assert.Equal(1, snap.ByRun["Late"].Unclaimed); + } + [Fact] public async Task ClaimedRowsAreNotCountedAsQueued() { diff --git a/tests/Craft.Tests/JobQueueStoreTests.cs b/tests/Craft.Tests/JobQueueStoreTests.cs index 9c29531..bfe7eaf 100644 --- a/tests/Craft.Tests/JobQueueStoreTests.cs +++ b/tests/Craft.Tests/JobQueueStoreTests.cs @@ -55,11 +55,11 @@ public async Task ReEnqueuingATaskUpsertsOneRowRatherThanDuplicating() var (queue, _) = NewQueue(); await queue.InitializeAsync(); - // The re-dispatch case (crash recovery, orphan re-drive). Schema v2 keys deterministically per - // (run, task), so the second enqueue UPDATES the first row instead of adding a duplicate — the - // duplicate that used to get claimed and executed a second time. + // The re-dispatch case (crash recovery, orphan re-drive). The key is the run's start plus the task, + // so the second enqueue UPDATES the first row instead of adding a duplicate — the duplicate that + // used to get claimed and executed a second time. + await queue.EnqueueAsync("run", "task-0", 4, At(1)); await queue.EnqueueAsync("run", "task-0", 4, At(1)); - await queue.EnqueueAsync("run", "task-0", 4, At(10)); Assert.Single(await queue.GetQueuedTaskIdsAsync("run")); diff --git a/tests/Craft.Tests/OrchestratorCancelRunTests.cs b/tests/Craft.Tests/OrchestratorCancelRunTests.cs new file mode 100644 index 0000000..0a3ad84 --- /dev/null +++ b/tests/Craft.Tests/OrchestratorCancelRunTests.cs @@ -0,0 +1,78 @@ +using System.Collections.Concurrent; +using System.Reflection; +using Craft.Configuration; +using Craft.Orchestration; +using Craft.Storage; +using Microsoft.Extensions.Logging.Abstractions; + +namespace Craft.Tests; + +/// +/// Cancelling a run this node holds must cancel the live tasks, not a copy loaded from storage: the live +/// ones are what the re-drive and completion check read, so a cancelled copy left them Pending and the +/// re-drive kept putting them back on the queue. +/// +public class OrchestratorCancelRunTests +{ + [Fact] + public async Task CancellingALiveRun_CancelsItsLiveTasks_AndDropsItsQueueRows() + { + var settings = new CraftSettings { Orchestrator = { TablePrefix = "cnl" + Guid.NewGuid().ToString("N")[..8] } }; + settings.Orchestrator.BatchStatusWrites = false; + var backing = new RunRemainingCounterTests.ConditionalStore(); + var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); + var queue = new JobQueueStore(NullLogger.Instance, settings, backing); + await store.InitializeAsync(); + await queue.InitializeAsync(); + var svc = NewService(settings, store, queue); + + var run = new OrchestratorRun + { + Name = "Mailbox_t1", + Status = "Running", + Priority = 4, + StartedUtc = DateTime.UtcNow, + Tasks = [Pending("a"), Pending("b"), Pending("c")] + }; + await store.UpsertRunAsync(run); + await store.UpsertTaskBatchAsync(run.Name, run.Tasks); + await store.InitRemainingAsync(run.Name, run.Tasks.Count); + await queue.EnqueueBatchAsync(run.Name, run.Tasks.Select(t => (t.Id, 4)).ToList(), run.StartedUtc); + run.Tasks[0].Status = "Running"; + await store.UpsertTaskAsync(run.Name, run.Tasks[0]); + Get>(svc, "_activeRuns")[run.Name] = run; + + var (found, cancelled) = await svc.CancelRunAsync(run.Name); + + Assert.True(found); + Assert.Equal(2, cancelled); + Assert.Equal(["Running", "Cancelled", "Cancelled"], run.Tasks.Select(t => t.Status)); + Assert.Empty(await queue.GetQueuedTaskIdsAsync(run.Name)); + } + + private static OrchestratorTaskItem Pending(string id) => new() { Id = id, Status = "Pending" }; + + private static OrchestratorService NewService(CraftSettings settings, OrchestratorTableStore store, JobQueueStore queue) + { + var svc = (OrchestratorService)System.Runtime.CompilerServices.RuntimeHelpers + .GetUninitializedObject(typeof(OrchestratorService)); + Set(svc, "_logger", NullLogger.Instance); + Set(svc, "_store", store); + Set(svc, "_queue", queue); + Set(svc, "_writer", new OrchestratorStatusWriter(store, NullLogger.Instance, settings)); + Set(svc, "_settings", settings); + Set(svc, "_lock", new object()); + Set(svc, "_activeRuns", new ConcurrentDictionary()); + Set(svc, "_cancelledRuns", new ConcurrentDictionary()); + Set(svc, "_finalizingRuns", new ConcurrentDictionary()); + return svc; + } + + private static void Set(object target, string field, object? value) => + typeof(OrchestratorService).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)! + .SetValue(target, value); + + private static T Get(object target, string field) => + (T)typeof(OrchestratorService).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)! + .GetValue(target)!; +} diff --git a/tests/Craft.Tests/RunRemainingCounterTests.cs b/tests/Craft.Tests/RunRemainingCounterTests.cs index 8661aca..3c22edf 100644 --- a/tests/Craft.Tests/RunRemainingCounterTests.cs +++ b/tests/Craft.Tests/RunRemainingCounterTests.cs @@ -105,8 +105,12 @@ public Task TryReplaceBatchAsync(string table, string partitionKey, IReadO return Task.FromResult(true); } + /// Point reads served, by table. + public System.Collections.Concurrent.ConcurrentDictionary Gets { get; } = new(); + public Task GetAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) { + Gets.AddOrUpdate(table, 1, (_, n) => n + 1); // Hand back a copy: a caller mutating what it read must not mutate the store in place. if (!Table(table).TryGetValue((partitionKey, rowKey), out var r)) return Task.FromResult(null); return Task.FromResult(new StoreRow(r.PartitionKey, r.RowKey) From 5055813ceb1dfef9849426fa01631bee70c49254 Mon Sep 17 00:00:00 2001 From: Zacgoose <107489668+Zacgoose@users.noreply.github.com> Date: Mon, 5 Oct 2026 23:55:11 +0800 Subject: [PATCH 05/24] feat(orchestration): keep all run and task state in storage - each run is one partition of a Work table; a task's state is its row, and claim, finish and the aggregation barrier are single partition transactions - the pump claims from a Ready list ordered by band then run age; lapsed leases are reclaimed and a task interrupted MaxRetries times is failed - PostExecution is an ordinary leased work item; child runs are placeholders the parent waits on - a run name still in progress is not started again - removes the re-drive, recovery pass, status writer, queue index and remaining counter, and their settings; the previous tables are dropped at startup - per-partition write limiter at 1,900 ops/s --- Services/Bridges/OrchestratorBridge.cs | 106 +- Services/Bridges/WorkerMetricsBridge.cs | 24 +- .../Configuration/OrchestratorSettings.cs | 97 +- .../Hosting/CraftHostBuilderExtensions.cs | 18 +- Services/Orchestration/FinishBatcher.cs | 52 + Services/Orchestration/JobDescriptor.cs | 14 +- Services/Orchestration/JobManager.cs | 8 +- Services/Orchestration/JobQueuePump.cs | 268 -- .../Orchestration/JobQueueStatusReader.cs | 305 +- .../MarkerNotPersistedException.cs | 16 - Services/Orchestration/OrchestratorRun.cs | 40 - Services/Orchestration/OrchestratorService.cs | 2559 +++-------------- .../Orchestration/OrchestratorStatusWriter.cs | 388 --- .../Orchestration/OrchestratorTaskItem.cs | 29 +- Services/Orchestration/WorkPump.cs | 195 ++ Services/Storage/AzureTableStore.cs | 39 + Services/Storage/ICraftTableStore.cs | 46 + Services/Storage/JobQueueStore.cs | 972 ------- Services/Storage/OrchestratorCleanupResult.cs | 15 - Services/Storage/OrchestratorRunSummary.cs | 12 - Services/Storage/OrchestratorTableStore.cs | 997 ------- Services/Storage/ResultStore.cs | 315 ++ Services/Storage/ResultWrite.cs | 10 - Services/Storage/TaskStatusWrite.cs | 12 - Services/Storage/WorkStore.cs | 782 +++++ docs/configuration.md | 32 +- perf-harness/docker-compose.bg.yml | 5 - .../AzureTableStoreLargeEntityTests.cs | 6 +- .../JobDescriptorRehydrationTests.cs | 143 +- tests/Craft.Tests/JobDurabilityTests.cs | 62 - tests/Craft.Tests/JobManagerReEnqueueTests.cs | 9 +- tests/Craft.Tests/JobQueueAzuriteTests.cs | 305 -- .../Craft.Tests/JobQueueDispatchableTests.cs | 106 - tests/Craft.Tests/JobQueueFifoTests.cs | 244 -- .../Craft.Tests/JobQueueIndexBackfillTests.cs | 201 -- tests/Craft.Tests/JobQueuePumpBackoffTests.cs | 276 -- tests/Craft.Tests/JobQueuePumpTests.cs | 224 -- tests/Craft.Tests/JobQueueRetentionTests.cs | 280 -- tests/Craft.Tests/JobQueueRunLookupTests.cs | 142 - .../Craft.Tests/JobQueueStatusReaderTests.cs | 236 +- .../JobQueueStatusRefreshCostTests.cs | 215 -- tests/Craft.Tests/JobQueueStoreTests.cs | 336 --- tests/Craft.Tests/MemoryTableStore.cs | 148 + .../Craft.Tests/OrchestrationContractTests.cs | 298 ++ .../OrchestratorBridgeLineageTests.cs | 49 +- .../Craft.Tests/OrchestratorCancelRunTests.cs | 78 - .../OrchestratorChildRunGuardTests.cs | 147 - .../OrchestratorFinalizedRunTests.cs | 346 --- .../OrchestratorRedriveGuardTests.cs | 163 -- .../OrchestratorResultStreamingTests.cs | 19 +- .../OrchestratorResultsAzuriteTests.cs | 10 +- .../OrchestratorRetentionAzuriteTests.cs | 162 -- .../Craft.Tests/OrchestratorRetentionTests.cs | 314 -- .../OrchestratorRunPersistenceTests.cs | 249 -- .../OrchestratorSequentialTests.cs | 529 ---- .../OrchestratorStaleRunningTests.cs | 209 -- ...estratorTaskLargeParametersAzuriteTests.cs | 71 +- .../OrchestratorTaskPathRehydrationTests.cs | 123 - tests/Craft.Tests/RunRemainingCounterTests.cs | 353 --- tests/Craft.Tests/StartupClaimGateTests.cs | 140 - .../StatusWriterDurabilityTests.cs | 499 ---- tests/Craft.Tests/TableKeyTests.cs | 8 +- tests/Craft.Tests/WorkStoreTests.cs | 265 ++ 63 files changed, 2865 insertions(+), 11456 deletions(-) create mode 100644 Services/Orchestration/FinishBatcher.cs delete mode 100644 Services/Orchestration/JobQueuePump.cs delete mode 100644 Services/Orchestration/MarkerNotPersistedException.cs delete mode 100644 Services/Orchestration/OrchestratorRun.cs delete mode 100644 Services/Orchestration/OrchestratorStatusWriter.cs create mode 100644 Services/Orchestration/WorkPump.cs delete mode 100644 Services/Storage/JobQueueStore.cs delete mode 100644 Services/Storage/OrchestratorCleanupResult.cs delete mode 100644 Services/Storage/OrchestratorRunSummary.cs delete mode 100644 Services/Storage/OrchestratorTableStore.cs create mode 100644 Services/Storage/ResultStore.cs delete mode 100644 Services/Storage/ResultWrite.cs delete mode 100644 Services/Storage/TaskStatusWrite.cs create mode 100644 Services/Storage/WorkStore.cs delete mode 100644 tests/Craft.Tests/JobQueueAzuriteTests.cs delete mode 100644 tests/Craft.Tests/JobQueueDispatchableTests.cs delete mode 100644 tests/Craft.Tests/JobQueueFifoTests.cs delete mode 100644 tests/Craft.Tests/JobQueueIndexBackfillTests.cs delete mode 100644 tests/Craft.Tests/JobQueuePumpBackoffTests.cs delete mode 100644 tests/Craft.Tests/JobQueuePumpTests.cs delete mode 100644 tests/Craft.Tests/JobQueueRetentionTests.cs delete mode 100644 tests/Craft.Tests/JobQueueRunLookupTests.cs delete mode 100644 tests/Craft.Tests/JobQueueStatusRefreshCostTests.cs delete mode 100644 tests/Craft.Tests/JobQueueStoreTests.cs create mode 100644 tests/Craft.Tests/MemoryTableStore.cs create mode 100644 tests/Craft.Tests/OrchestrationContractTests.cs delete mode 100644 tests/Craft.Tests/OrchestratorCancelRunTests.cs delete mode 100644 tests/Craft.Tests/OrchestratorChildRunGuardTests.cs delete mode 100644 tests/Craft.Tests/OrchestratorFinalizedRunTests.cs delete mode 100644 tests/Craft.Tests/OrchestratorRedriveGuardTests.cs delete mode 100644 tests/Craft.Tests/OrchestratorRetentionAzuriteTests.cs delete mode 100644 tests/Craft.Tests/OrchestratorRetentionTests.cs delete mode 100644 tests/Craft.Tests/OrchestratorRunPersistenceTests.cs delete mode 100644 tests/Craft.Tests/OrchestratorSequentialTests.cs delete mode 100644 tests/Craft.Tests/OrchestratorStaleRunningTests.cs delete mode 100644 tests/Craft.Tests/OrchestratorTaskPathRehydrationTests.cs delete mode 100644 tests/Craft.Tests/RunRemainingCounterTests.cs delete mode 100644 tests/Craft.Tests/StartupClaimGateTests.cs delete mode 100644 tests/Craft.Tests/StatusWriterDurabilityTests.cs create mode 100644 tests/Craft.Tests/WorkStoreTests.cs diff --git a/Services/Bridges/OrchestratorBridge.cs b/Services/Bridges/OrchestratorBridge.cs index 17ccc6d..24988ac 100644 --- a/Services/Bridges/OrchestratorBridge.cs +++ b/Services/Bridges/OrchestratorBridge.cs @@ -32,10 +32,10 @@ public static void QueueOrchestration(string name, string batchJson, int priorit // character would register a child link no live run ever matches. name = TableKeys.Sanitize(name); parentRunName = ResolveParentRunName(name, parentRunName); - var gated = RegisterPendingChild(parentRunName, name); + var child = RegisterPendingChild(parentRunName, name); s_pending.Enqueue(new PendingOrchestration(name, batchJson, priority, postExecFunctionName, postExecParametersJson, parentRunName, reference, - PendingChildRegistered: gated, Sequential: sequential)); + Sequential: sequential, ParentRunKey: child?.ParentRunKey, ChildKey: child?.ChildKey)); } /// @@ -55,10 +55,10 @@ public static void QueueOrchestrationFromFile(string name, string batchFilePath, { name = TableKeys.Sanitize(name); parentRunName = ResolveParentRunName(name, parentRunName); - var gated = RegisterPendingChild(parentRunName, name); + var child = RegisterPendingChild(parentRunName, name); s_pending.Enqueue(new PendingOrchestration(name, string.Empty, priority, postExecFunctionName, postExecParametersJson, parentRunName, reference, batchFilePath, - PendingChildRegistered: gated, Sequential: sequential)); + Sequential: sequential, ParentRunKey: child?.ParentRunKey, ChildKey: child?.ChildKey)); } /// @@ -88,76 +88,51 @@ public static void QueueOrchestrationFromFile(string name, string batchFilePath, } /// - /// Register the child link at ENQUEUE time — while the parent task's script is still executing, - /// so the parent cannot pass its completion check before the gate exists. Registering after - /// StartFromBatchAsync (the old shape) loses that race for the parent's LAST task: the drain - /// runs in a background Task.Run while the enqueuing task is marked terminal immediately, so - /// the parent would finalize — and dispatch PostExecution — before its child was visible. - /// Returns whether a gate was taken, so the drain releases exactly what was registered. + /// Make the parent wait for this child, at ENQUEUE time — while the parent's task is still executing, so + /// the parent cannot reach its barrier first. The returned keys travel with the queued run: the child + /// fills the placeholder when it finishes, and a child that is never created releases it on the drain. /// - private static bool RegisterPendingChild(string? parentRunName, string childName) => - !string.IsNullOrEmpty(parentRunName) && - s_service?.TryRegisterPendingChildRun(parentRunName, childName) == true; + private static (string ParentRunKey, string ChildKey)? RegisterPendingChild(string? parentRunName, string childName) => + string.IsNullOrEmpty(parentRunName) ? null : s_service?.RegisterPendingChild(parentRunName, childName); - /// - /// Synchronous drain — blocks until all pending orchestrations are started. - /// Safe to call from any context (no SynchronizationContext on background workers). - /// + /// Synchronous drain — blocks until all pending orchestrations are started. public static void DrainPending() { while (s_pending.TryDequeue(out var p)) - { - try - { - if (s_service == null) { DiscardUndispatchable(p); continue; } - s_service.StartFromBatchAsync(p.Name, p.BatchJson, p.Priority, - p.PostExecFunctionName, p.PostExecParametersJson, CancellationToken.None, - p.ParentRunName, p.Reference, p.BatchFilePath, p.Sequential) - .GetAwaiter().GetResult(); - } - catch (Exception ex) - { - s_service?._logger.LogError(ex, "[Orchestrator] DrainPending failed for {Name}", p.Name); - } - finally - { - // The enqueue-time gate lifts on EVERY path once the start attempt is over: a - // started child is in _activeRuns by now (which takes over blocking the parent), - // and one that failed to start must stop blocking — a leaked gate would defer the - // parent's finalize forever, re-checked every 60s for the process lifetime. - if (p.PendingChildRegistered) - s_service?.ReleasePendingChildRun(p.Name); - } - } + Task.Run(() => StartAsync(p)).GetAwaiter().GetResult(); DrainPendingPlanners(); } - /// - /// Async drain — preferred from async call sites (PostExec lambdas, ExecuteScript). - /// + /// Async drain — preferred from async call sites (PostExec, ExecuteScript). public static async Task DrainPendingAsync() { while (s_pending.TryDequeue(out var p)) + await StartAsync(p); + await DrainPendingPlannersAsync(); + } + + private static async Task StartAsync(PendingOrchestration p) + { + var created = false; + try { - try - { - if (s_service == null) { DiscardUndispatchable(p); continue; } - await s_service.StartFromBatchAsync(p.Name, p.BatchJson, p.Priority, - p.PostExecFunctionName, p.PostExecParametersJson, CancellationToken.None, - p.ParentRunName, p.Reference, p.BatchFilePath, p.Sequential); - } - catch (Exception ex) - { - s_service?._logger.LogError(ex, "[Orchestrator] DrainPending failed for {Name}", p.Name); - } - finally + if (s_service == null) { DiscardUndispatchable(p); return; } + created = await s_service.StartFromBatchAsync(p.Name, p.BatchJson, p.Priority, + p.PostExecFunctionName, p.PostExecParametersJson, CancellationToken.None, + p.ParentRunName, p.Reference, p.BatchFilePath, p.Sequential, p.ParentRunKey, p.ChildKey); + } + catch (Exception ex) + { + s_service?._logger.LogError(ex, "[Orchestrator] DrainPending failed for {Name}", p.Name); + } + finally + { + if (!created && p.ParentRunKey != null && p.ChildKey != null && s_service != null) { - // See DrainPending: the gate lifts whatever the outcome of the start attempt. - if (p.PendingChildRegistered) - s_service?.ReleasePendingChildRun(p.Name); + try { await s_service.AbandonPendingChildAsync(p.ParentRunKey, p.ChildKey); } + catch (Exception ex) { s_service._logger.LogWarning(ex, "[Orchestrator] Could not release {Name} from its parent", p.Name); } } } - await DrainPendingPlannersAsync(); } /// @@ -179,15 +154,16 @@ private static void DiscardUndispatchable(PendingOrchestration p) /// /// A queued run. Exactly one of and - /// carries the batch; the file path wins when both are set. - /// records whether enqueue took a pending-child gate on - /// the orchestrator, so the drain releases exactly the gates that were taken — releasing on a - /// refused registration could lift a gate held by ANOTHER queued entry of the same child name. + /// carries the batch; the file path wins when both are set. and + /// are set when the parent was made to wait for this run. /// public record PendingOrchestration(string Name, string BatchJson, int Priority, string? PostExecFunctionName, string? PostExecParametersJson, string? ParentRunName, - string? Reference = null, string? BatchFilePath = null, - bool PendingChildRegistered = false, bool Sequential = false); + string? Reference = null, string? BatchFilePath = null, bool Sequential = false, + string? ParentRunKey = null, string? ChildKey = null) + { + public bool PendingChildRegistered => ChildKey != null; + } private static readonly ConcurrentQueue s_pendingPlanners = new(); diff --git a/Services/Bridges/WorkerMetricsBridge.cs b/Services/Bridges/WorkerMetricsBridge.cs index c095126..c40d3ce 100644 --- a/Services/Bridges/WorkerMetricsBridge.cs +++ b/Services/Bridges/WorkerMetricsBridge.cs @@ -693,10 +693,8 @@ public static bool DeleteJob(string jobId) => s_jobManager?.DeleteJob(jobId) ?? false; /// - /// Empty the durable job queue — a maintenance/reset primitive. Returns the number of queue rows - /// removed, or -1 if the orchestrator is unavailable or the clear failed. In-flight work is - /// unaffected and Pending tasks may be re-driven, so pair with when the - /// intent is to STOP work rather than clear a wedged or corrupted queue. + /// Empty the durable job queue — cancel every task still waiting in storage. Returns how many were + /// cancelled, or -1 if the orchestrator is unavailable or the clear failed. Running tasks finish. /// PS usage: [Craft.Services.WorkerMetricsBridge]::ClearQueue(). /// public static int ClearQueue() @@ -714,28 +712,20 @@ public static int ClearQueue() } /// - /// Change a queued job's priority. In the local buffer this re-enqueues at the new priority; for an - /// unclaimed durable row it moves the row to the new priority bucket (keeping its age) and records - /// the override on the task so a restart re-queues it at the operator's priority. + /// Change a queued job's priority. In the local buffer this re-enqueues at the new priority; for a task + /// still in storage it moves the task's whole run to the new priority band, since a run's tasks share + /// one queue position. /// public static bool ChangePriority(string jobId, int newPriority) { if (s_jobManager?.ChangePriority(jobId, newPriority) == true) return true; - if (s_queueReader == null) return false; + if (s_queueReader == null || s_orchestrator == null) return false; return RunBridged(async ct => { var snap = await s_queueReader.GetAsync(TimeSpan.FromSeconds(2), ct); var row = snap?.Rows.FirstOrDefault(r => !r.Claimed && $"{r.RunName}-{r.TaskId}" == jobId); - if (row == null) return false; - - var moved = await s_queueReader.Queue.ReprioritizeTaskAsync(row.RunName, row.TaskId, newPriority, ct); - if (moved == 0) return false; - - // Best-effort durability of the override itself: only effective where the run is live, - // which on the dispatching node it is. The row move above is what changes dispatch order. - s_orchestrator?.PriorityChanged(new JobDescriptor(row.RunName, row.TaskId, row.Priority), newPriority); - return true; + return row != null && await s_orchestrator.ReprioritizeRunAsync(row.RunName, newPriority); }); } diff --git a/Services/Configuration/OrchestratorSettings.cs b/Services/Configuration/OrchestratorSettings.cs index 8c97e36..acb79cb 100644 --- a/Services/Configuration/OrchestratorSettings.cs +++ b/Services/Configuration/OrchestratorSettings.cs @@ -1,74 +1,16 @@ namespace Craft.Configuration; /// -/// Orchestrator settings — fan-out/fan-in task execution with crash recovery. +/// Orchestrator settings — fan-out/fan-in task execution, with all state in storage. /// public class OrchestratorSettings { /// - /// Prefix for the three Azure Tables used by the orchestrator. - /// Tables created: {Prefix}Runs, {Prefix}Tasks, {Prefix}Results. + /// Prefix for the orchestrator's Azure Tables. + /// Tables created: {Prefix}Work, {Prefix}Ready, {Prefix}Names, {Prefix}Finished, {Prefix}TaskResults. /// public string TablePrefix { get; set; } = "Orchestrator"; - /// - /// Batch and coalesce per-task/run status writes through OrchestratorStatusWriter instead of writing each - /// individually. Removes the per-task Azure Table write from the fan-out critical path (the throughput - /// ceiling — see docs/orch-analysis.md). Default true. Results are never batched (their chunking path is - /// untouched). Set false to fall back to the original per-task writes (for A/B). - /// - public bool BatchStatusWrites { get; set; } = true; - - /// - /// Coalesce SMALL task results (those that fit one Azure Table property) through the batched status - /// writer instead of a per-task upsert on the fan-out critical path. Each result is written BEFORE - /// its task's terminal marker in the same flush, so a result is always durable before the task is - /// counted done (and therefore before finalize/post-execution reads it). Large results keep the - /// directly-awaited chunked path. Default true; only applies when is - /// also true. Set false to fall back to the original per-task awaited result write. - /// - public bool BatchResultWrites { get; set; } = true; - - /// - /// When batching status writes, write the pre-invoke "Running" marker under a synchronous barrier so it - /// is durable BEFORE the task invokes (batched with other concurrently-starting tasks). Preserves the - /// AttemptCount/MaxRetries poison-task guarantee. Default true. False = eventual (faster, weaker: the - /// marker rides the periodic flush, so a host crash within the flush window may not advance AttemptCount). - /// - public bool DurableRunningBarrier { get; set; } = true; - - /// How often (ms) the status writer flushes coalesced writes. Also the barrier latency ceiling. - /// Default 25. - public int StatusFlushIntervalMs { get; set; } = 25; - - /// - /// How long a task will wait for its durable "Running" marker before giving up (seconds, default 90). - /// - /// This wait sits between the JobManager dispatching a task and that task checking out a worker, so - /// an unbounded one is a whole-host outage: production wedged for 101 and 75 minutes with all 8 - /// limiter slots held by tasks blocked here, every BG worker idle, and the heap at 24% of its cap. - /// On timeout the task is deferred, NOT failed — the marker never landed, so storage still has it - /// Pending and it is retried. Must exceed , since a waiter - /// may need the in-flight flush to finish plus one more. - /// - public int RunningBarrierTimeoutSeconds { get; set; } = 90; - - /// - /// Ceiling on one flush of coalesced status writes (seconds, default 30). The drain loop is a single - /// loop, so an unbounded storage call inside a flush stops every status write and every barrier - /// waiter in the process. Writes that do not complete are put back and retried on the next flush. - /// - public int StatusFlushTimeoutSeconds { get; set; } = 30; - - /// - /// How many per-run status writes may be in flight within one flush (default 8). - /// - /// Writes are grouped by run because a batch must share a partition key. The workload that broke - /// this was ~600 runs of ONE task each (one per tenant), which turned a "batch" into hundreds of - /// sequential round-trips inside a single flush. 1 restores the old sequential behaviour. - /// - public int StatusFlushConcurrency { get; set; } = 8; - /// /// PowerShell function used to execute individual orchestrator tasks. /// Receives a hashtable with task parameters. @@ -87,44 +29,25 @@ public class OrchestratorSettings /// public string PostExecFunction { get; set; } = "Invoke-CraftPostExecution"; - /// Maximum number of times a task can be interrupted before being marked Failed. + /// Maximum number of times a task can be interrupted before being marked Failed, and how many + /// times a run's PostExecution is attempted. public int MaxRetries { get; set; } = 3; /// - /// How long a run's rows outlive it (hours, default 48). A run that finished — or that nothing is - /// driving and that last wrote to storage — longer ago than this is removed from all three tables, - /// together with any Tasks/Results partition whose Run row is already gone. Craft itself needs the - /// rows only while a run is live; they stay this long for operators reading recent history. + /// How long a finished run's rows outlive it (hours, default 48). Craft needs them only while the run + /// is live; they stay this long for operators reading recent history. /// public int RetentionHours { get; set; } = 48; /// /// How often the retention sweep runs after the one at startup (hours, default 4). 0 disables the - /// periodic sweep; the startup pass, which follows crash recovery, still runs. + /// periodic sweep; the startup pass still runs. /// public int CleanupIntervalHours { get; set; } = 4; /// - /// The per-run status/re-drive tick cadence (seconds, default 60). Each live run has a timer firing at - /// this interval to log status, re-drive orphaned tasks, and re-check completion. It is also the floor - /// of the re-drive backoff. Raising it cuts per-run overhead at high live-run counts; the perf harness - /// lowers it to exercise the backoff in compressed time. Minimum 1s. + /// How often each active run's status line is considered for logging (seconds, default 60). A line is + /// written when the run's counts change, or every ten minutes when they do not. Minimum 1s. /// public int StatusTimerIntervalSeconds { get; set; } = 60; - - /// - /// Whether the per-run re-drive backs off geometrically once it has verified a run has no orphaned - /// tasks (default true). When false the re-drive verifies against storage on every tick — the old - /// behaviour, retained as a safety switch and for A/B measurement of the backoff's effect. - /// - public bool RedriveBackoff { get; set; } = true; - - /// - /// Whether a task sheds its Parameters payload from the in-memory run graph once it is durably - /// persisted and enqueued, rehydrating it from the Tasks table at dispatch (default true). This bounds - /// the retained memory of a large pending backlog — thousands of runs each holding every task's payload - /// is what drives the live-set toward the GC heap ceiling — at the cost of one point read per task at - /// dispatch. False keeps the payload resident the whole time (the old behaviour), for A/B or safety. - /// - public bool ShedPendingParameters { get; set; } = true; } diff --git a/Services/Hosting/CraftHostBuilderExtensions.cs b/Services/Hosting/CraftHostBuilderExtensions.cs index ab3f637..834b642 100644 --- a/Services/Hosting/CraftHostBuilderExtensions.cs +++ b/Services/Hosting/CraftHostBuilderExtensions.cs @@ -275,15 +275,15 @@ public static IServiceCollection AddCraftServices(this IServiceCollection servic services.AddSingleton(); services.AddSingleton(); - services.AddSingleton(); - // The durable job queue. Registered alongside the orchestrator store because it shares the - // same ICraftTableStore and therefore the same bounded connection pool. - services.AddSingleton(); + // Durable orchestration state and task results, over the shared ICraftTableStore (and its + // bounded connection pool). + services.AddSingleton(_ => new PartitionRateLimiter()); + services.AddSingleton(); + services.AddSingleton(); // Table-backed queue view for the status APIs — the JobManager only buffers a worker-pool-sized // slice of the backlog, so status must read the tables. All roles: an HTTP-only node serves the // worker-health endpoint for work that runs elsewhere. services.AddSingleton(); - services.AddSingleton(); services.AddSingleton(); services.AddSingleton(); services.AddSingleton(); @@ -297,10 +297,10 @@ public static IServiceCollection AddCraftServices(this IServiceCollection servic if (roles.Background) { services.AddHostedService(sp => sp.GetRequiredService()); - // Feeds the JobManager from the durable queue a batch at a time, so the backlog lives - // in storage rather than in this process. - services.AddSingleton(); - services.AddHostedService(sp => sp.GetRequiredService()); + // Feeds the JobManager from storage a batch at a time, so the backlog lives in storage + // rather than in this process. + services.AddSingleton(); + services.AddHostedService(sp => sp.GetRequiredService()); services.AddHostedService(sp => sp.GetRequiredService()); services.AddHostedService(sp => sp.GetRequiredService()); } diff --git a/Services/Orchestration/FinishBatcher.cs b/Services/Orchestration/FinishBatcher.cs new file mode 100644 index 0000000..ecbb49e --- /dev/null +++ b/Services/Orchestration/FinishBatcher.cs @@ -0,0 +1,52 @@ +using Craft.Storage; + +namespace Craft.Orchestration; + +/// +/// Coalesces task finishes per run, so tasks of one run finishing together share one transaction instead of +/// each paying for its own (and racing each other on the run header). Callers await their own outcome: a +/// finish is durable when the await returns, which is what lets a worker slot go. +/// +public sealed class FinishBatcher(WorkStore store, ILogger logger, TimeSpan? window = null) +{ + private readonly TimeSpan _window = window ?? TimeSpan.FromMilliseconds(15); + private readonly object _lock = new(); + private readonly Dictionary Done)>> _pending = new(StringComparer.Ordinal); + + public Task FinishAsync(string runKey, WorkStore.Finish finish) + { + var done = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + bool first; + lock (_lock) + { + first = !_pending.TryGetValue(runKey, out var list); + if (first) _pending[runKey] = list = []; + list!.Add((finish, done)); + } + if (first) _ = FlushAfterWindowAsync(runKey); + return done.Task; + } + + private async Task FlushAfterWindowAsync(string runKey) + { + await Task.Delay(_window); + List<(WorkStore.Finish Finish, TaskCompletionSource Done)> batch; + lock (_lock) + { + batch = _pending[runKey]; + _pending.Remove(runKey); + } + + try + { + var outcome = await store.FinishAsync(runKey, batch.Select(b => b.Finish).ToList()); + foreach (var (_, done) in batch) done.TrySetResult(outcome); + } + catch (Exception ex) + { + logger.LogWarning(ex, "[Orchestrator] Could not record {Count} finished task(s) of {Run}; their leases lapse and they run again", + batch.Count, runKey); + foreach (var (_, done) in batch) done.TrySetException(ex); + } + } +} diff --git a/Services/Orchestration/JobDescriptor.cs b/Services/Orchestration/JobDescriptor.cs index e5d5d5d..6af62bf 100644 --- a/Services/Orchestration/JobDescriptor.cs +++ b/Services/Orchestration/JobDescriptor.cs @@ -4,14 +4,24 @@ namespace Craft.Orchestration; /// The identity of a queued orchestrator task — everything the queue needs to hold, and nothing more. /// /// Orchestrator fan-out is where queue depth comes from (production peaked at 783 queued with 3.7-hour -/// waits), and it used to enqueue a closure capturing the whole graph, the +/// waits), and it used to enqueue a closure capturing the whole run graph, the /// task, the script path and the service. A descriptor replaces all of that with two string references /// and an int; turns it back into runnable work at dispatch time. /// /// Run this task belongs to — the storage partition key. /// Task id within the run — the storage row key. /// Dispatch priority (lower = higher). -public readonly record struct JobDescriptor(string RunName, string TaskId, int Priority); +public readonly record struct JobDescriptor(string RunName, string TaskId, int Priority) +{ + /// The run's storage key (its partition); null for a descriptor that names a run only. + public string? RunKey { get; init; } + + /// The task's position in its run, which addresses its row. + public int Seq { get; init; } + + /// Which execution of the task this is (1 for the first). + public int Attempt { get; init; } +} /// /// Rehydrates a into runnable work. Returns null when the descriptor is diff --git a/Services/Orchestration/JobManager.cs b/Services/Orchestration/JobManager.cs index 9ab490e..12f5b36 100644 --- a/Services/Orchestration/JobManager.cs +++ b/Services/Orchestration/JobManager.cs @@ -200,11 +200,9 @@ public string Enqueue(JobDescriptor descriptor, string name, string? id = null) // writes status onto the queue item's own record), so the TRACKED record sat frozen at the // previous outing's "Completed" while a live copy of the job was queued or running. // - // IsQueuedOrRunning reads this dictionary, and JobQueuePump.ReleaseFinishedAsync treats a "no" - // as permission to DELETE that task's durable queue row. A stale record therefore had the pump - // dropping rows out from under running work — observed live releasing 7-9 "finished" jobs per - // second against 8 slots. RedrivePendingTasks consults the same predicate, so it was misreading - // task state for the same reason. + // IsQueuedOrRunning reads this dictionary, and WorkPump treats a "no" as the job being done and stops + // renewing its claim. A stale record once had the pump dropping work out from under running jobs — + // observed live releasing 7-9 "finished" jobs per second against 8 slots. _jobs[jobId] = record; lock (_queueLock) diff --git a/Services/Orchestration/JobQueuePump.cs b/Services/Orchestration/JobQueuePump.cs deleted file mode 100644 index a5e030c..0000000 --- a/Services/Orchestration/JobQueuePump.cs +++ /dev/null @@ -1,268 +0,0 @@ -using Craft.Configuration; -using Craft.Storage; - -namespace Craft.Orchestration; - -/// -/// Keeps the in-memory job queue small by feeding it from storage a batch at a time. -/// -/// The point is that the backlog lives in the table, not in this process: the JobManager holds at most -/// a worker-pool-sized buffer plus whatever is in flight, and tops up only when it runs low. A run of -/// 7,336 tasks is 7,336 rows in storage and a handful of objects here. -/// -/// Deliberately a separate pump rather than a change to the dispatch loop. That loop owns the limiter -/// slot lifecycle, and its invariants were written to close a leak that wedged a production instance for -/// 28 hours (see BackgroundTaskLimiter's accounting notes). Feeding it through the enqueue path it -/// already has means none of that is disturbed — this component can only ever add work. -/// -/// Lease handling is what makes the claim safe to hold in memory: a claimed row stays owned by this -/// instance for LeaseSeconds, renewed while the job is still in flight, and reclaimable by anyone once -/// it lapses. An instance that dies mid-batch gives its work back without anything having to notice. -/// -public class JobQueuePump : BackgroundService -{ - private readonly ILogger _logger; - private readonly JobQueueStore _queue; - private readonly JobManager _jobs; - private readonly string _owner; - private readonly int _batchSize; - private readonly int _lowWater; - private readonly TimeSpan _lease; - private readonly TimeSpan _pollInterval; - private readonly TimeSpan _idlePollInterval; - - /// Renew a claim only once its lease has less than this left. The lease is set comfortably - /// longer than a task can run, so most claims finish without ever needing a renewal; this is the tail - /// (a deep buffer, or a genuinely long task) that has been held long enough to approach expiry. - private readonly TimeSpan _renewWhenWithin; - - /// Rows claimed by this instance, by the job id they were handed to the JobManager under. - private readonly Dictionary _inFlight = new(StringComparer.Ordinal); - - /// UTC lease expiry per in-flight job id, so renewal can skip claims with plenty of lease - /// left instead of re-reading every row every tick. - private readonly Dictionary _leaseExpiry = new(StringComparer.Ordinal); - - /// Completes when startup recovery is done; nothing is claimed before it. Null (tests, a host - /// without an orchestrator) claims immediately. - private readonly Task? _claimGate; - - public JobQueuePump(ILogger logger, JobQueueStore queue, JobManager jobs, - IConfiguration configuration, CraftSettings settings, OrchestratorService? orchestrator = null) - { - _logger = logger; - _queue = queue; - _jobs = jobs; - _claimGate = orchestrator?.RecoveryDone; - - // Identifies this instance's claims. The container id is stable for the life of the process and - // distinct per instance, which is exactly the scope a lease needs. - _owner = Environment.GetEnvironmentVariable("HOSTNAME") - ?? $"instance-{Environment.ProcessId}"; - - // One batch per worker, so a refill hands the pool exactly enough to stay busy. - _batchSize = Math.Max(1, configuration.GetValue("JobQueueBatchSize", Math.Max(1, settings.Worker.BgPoolSize))); - - // Top up before the buffer empties, so workers never wait on a storage round-trip. - _lowWater = Math.Max(0, configuration.GetValue("JobQueueLowWaterMark", 2)); - - // Comfortably longer than the 1200s task timeout: a lease that lapses under a task still running - // would hand its work to a second worker. - _lease = TimeSpan.FromSeconds(Math.Max(60, configuration.GetValue("JobQueueLeaseSeconds", 1800))); - - // Renew in the last third of the lease. Below the lease length by construction, so a claim is - // never renewed on the same tick it was taken. - _renewWhenWithin = TimeSpan.FromTicks(_lease.Ticks / 3); - - _pollInterval = TimeSpan.FromMilliseconds( - Math.Max(100, configuration.GetValue("JobQueuePollIntervalMs", 1000))); - - // Ceiling for the idle backoff. Clamped to at least the base interval, or "backing off" would - // speed the loop up. - _idlePollInterval = TimeSpan.FromMilliseconds(Math.Max( - _pollInterval.TotalMilliseconds, - configuration.GetValue("JobQueueIdlePollIntervalMs", 10_000))); - } - - protected override async Task ExecuteAsync(CancellationToken stoppingToken) - { - _logger.LogInformation( - "[JobQueuePump] Started: owner={Owner} batch={Batch} lowWater={Low} lease={Lease}s poll={Poll}ms idlePoll={IdlePoll}ms", - _owner, _batchSize, _lowWater, _lease.TotalSeconds, - _pollInterval.TotalMilliseconds, _idlePollInterval.TotalMilliseconds); - - // No claim before startup recovery has run. A row claimed earlier rehydrates its run from storage — - // stale Running markers from the previous process included — into the live graph ahead of recovery, - // whose reset then lands on a copy that loses the _activeRuns race. Rows enqueued meanwhile (timer or - // HTTP-started runs) just wait in the table and are claimed on the first cycle after. - if (_claimGate is { IsCompleted: false }) - { - _logger.LogInformation("[JobQueuePump] Waiting for startup recovery before claiming"); - try { await _claimGate.WaitAsync(stoppingToken); } - catch (OperationCanceledException) { return; } - _logger.LogInformation("[JobQueuePump] Startup recovery done — claiming"); - } - - var idleTicks = 0; - - while (!stoppingToken.IsCancellationRequested) - { - var claimedAny = false; - try - { - // Ensure the queue schema is migrated before this pump ever claims. It is a cheap bool - // check after the first success; before it, claiming could hand out a row the one-time - // key migration is still rewriting. Idempotent and shared with the enqueue paths. - await _queue.InitializeAsync(stoppingToken); - - await ReleaseFinishedAsync(stoppingToken); - claimedAny = await RefillAsync(stoppingToken); - await RenewAsync(stoppingToken); - } - catch (OperationCanceledException) when (stoppingToken.IsCancellationRequested) - { - break; - } - catch (Exception ex) - { - // One bad cycle must not end the pump; the next tick tries again. A pump that dies - // silently would look exactly like an empty queue. - // - // The log itself is guarded: under a pegged GC hard limit even LogError allocates and can - // throw OOM, and this catch is the last frame before ExecuteAsync — an escape here faults - // the service and, with the host's default StopHost behaviour, restarts the container - // mid-run. A failed log is never worth the pump. (BackgroundTaskLimiter guards its logs - // past the point of commitment for the same reason.) - try { _logger.LogError(ex, "[JobQueuePump] Cycle failed; continuing"); } - catch { /* logging is never worth the loop */ } - } - - // Idle means this pump has nothing: it claimed nothing AND holds nothing. Note what is - // deliberately NOT idle — a full buffer. RefillAsync returns early without claiming while - // QueuedCount is above the low-water mark, and treating that as idle would back the loop off - // exactly when a busy run is about to need its next batch. A refill delivers at most - // batchSize jobs per tick, so the poll interval is a hard throughput ceiling of - // batchSize/interval: at a flat 10s that is 0.8 tasks/sec, which on a 7,336-task fan-out is - // hours of pure waiting however fast the tasks are. Backing off only when genuinely empty - // keeps that ceiling at the base interval whenever it could bind. - if (claimedAny || _inFlight.Count > 0) idleTicks = 0; - else idleTicks++; - - // Wait for the poll interval OR an enqueue signal, whichever comes first. The signal is what - // makes a freshly-queued run start now instead of on the next tick — a cold system had backed - // the interval off toward its idle ceiling, so without this the first task of a quiet-time - // orchestration waited up to that ceiling just to be claimed. The interval stays as the - // backstop (a missed signal, cross-instance work, freed leases), so this only removes the wait. - try { await _queue.WaitForWorkAsync(NextDelay(idleTicks), stoppingToken); } - catch (OperationCanceledException) { break; } - } - - _logger.LogInformation("[JobQueuePump] Stopped ({InFlight} claims still held)", _inFlight.Count); - } - - /// - /// Drop the queue rows for jobs the JobManager has finished with. - /// - /// Removal is deliberately AFTER the work is done, not at claim time: a row deleted on claim would - /// take the task with it if this instance died holding it. Until then the lease is what stops anyone - /// else running it. - /// - private async Task ReleaseFinishedAsync(CancellationToken ct) - { - if (_inFlight.Count == 0) return; - - var finished = _inFlight.Where(kv => !_jobs.IsQueuedOrRunning(kv.Key)).ToList(); - if (finished.Count == 0) return; - - // One transaction per partition rather than two point deletes per task. Tracking is cleared only - // after the storage delete lands, so a failure here throws, the cycle guard logs it, and the same - // claims are retried next tick (the delete is idempotent — a row already gone is tolerated). - await _queue.RemoveBatchAsync(finished.Select(kv => kv.Value).ToList(), ct); - - foreach (var (jobId, _) in finished) - { - _inFlight.Remove(jobId); - _leaseExpiry.Remove(jobId); - } - - _logger.LogDebug("[JobQueuePump] Released {Count} finished job(s)", finished.Count); - } - - /// - /// How long to wait before the next cycle: the base interval while there is anything to do, doubling - /// toward once the pump has gone quiet. - /// - /// The doubling matters more than the ceiling — it means a queue that goes briefly empty between - /// batches barely slows down, while one that is empty for minutes stops scanning storage every - /// second. Any tick that claims or holds work resets it, so the pump is back at full speed on the - /// cycle after work appears. - /// - private TimeSpan NextDelay(int idleTicks) - { - if (idleTicks <= 0) return _pollInterval; - - // Clamped before the shift so the multiplier cannot overflow on a long idle stretch. - var factor = 1L << Math.Min(idleTicks, 20); - var ms = Math.Min(_idlePollInterval.TotalMilliseconds, _pollInterval.TotalMilliseconds * factor); - return TimeSpan.FromMilliseconds(ms); - } - - /// - /// Claim another batch once the buffer has drawn down to the low-water mark. Returns whether - /// anything was claimed, which is what tells the loop this tick was not idle. - /// - private async Task RefillAsync(CancellationToken ct) - { - if (_jobs.QueuedCount > _lowWater) return false; - - var claimed = await _queue.ClaimBatchAsync(_owner, _batchSize, _lease, ct); - if (claimed.Count == 0) return false; - - var leaseExpiry = DateTime.UtcNow + _lease; - foreach (var job in claimed) - { - // Enqueued by identity, the way the orchestrator already does it, so the work is rebuilt at - // dispatch time and nothing but the descriptor is retained here. - var name = $"{job.RunName}-{job.TaskId}"; - var jobId = _jobs.Enqueue(new JobDescriptor(job.RunName, job.TaskId, job.Priority), name); - _inFlight[jobId] = job; - // Slightly earlier than the lease storage actually recorded (claimed a moment before this), - // so renewal errs toward being early rather than late. - _leaseExpiry[jobId] = leaseExpiry; - } - - _logger.LogDebug("[JobQueuePump] Claimed {Count} job(s) ({Queued} queued after refill)", - claimed.Count, _jobs.QueuedCount); - - return true; - } - - /// - /// Extend the lease on everything still in flight. A failure here means a claim lapsed and the work - /// may already have been taken by someone else, so it is logged loudly — but not acted on, because - /// the task itself is the JobManager's to finish or fail. - /// - private async Task RenewAsync(CancellationToken ct) - { - if (_inFlight.Count == 0) return; - - // Only the claims whose lease is actually running low. Everything else has ample lease left and - // does not need a storage round-trip this tick — the old code re-read every in-flight row every - // tick regardless of how much lease remained. - var now = DateTime.UtcNow; - var dueIds = _inFlight.Keys - .Where(id => !_leaseExpiry.TryGetValue(id, out var expiry) || expiry - now < _renewWhenWithin) - .ToList(); - if (dueIds.Count == 0) return; - - var due = dueIds.Select(id => _inFlight[id]).ToList(); - if (!await _queue.RenewAsync(due, _owner, _lease, ct)) - { - _logger.LogWarning("[JobQueuePump] One or more leases could not be renewed — work may have been reclaimed"); - return; - } - - var renewedExpiry = now + _lease; - foreach (var id in dueIds) _leaseExpiry[id] = renewedExpiry; - } -} diff --git a/Services/Orchestration/JobQueueStatusReader.cs b/Services/Orchestration/JobQueueStatusReader.cs index 413d11d..a6914f2 100644 --- a/Services/Orchestration/JobQueueStatusReader.cs +++ b/Services/Orchestration/JobQueueStatusReader.cs @@ -4,111 +4,47 @@ namespace Craft.Orchestration; /// -/// The table-backed view of the job queue, for the status APIs. -/// -/// Since ownership of queued tasks moved into the {prefix}Queue table, the in-memory JobManager holds -/// only a worker-pool-sized buffer of claims plus whatever closure jobs were enqueued directly. Every -/// consumer that used to read it as "the queue" — the worker-health page, /API/jobs/*, the stats -/// history — was therefore reporting the buffer as if it were the backlog: a 7,000-task fan-out showed -/// eight queued jobs. This type merges the two truths: the JobManager for what THIS instance is doing, -/// the tables for what exists. -/// -/// Reads are cached with a short TTL and refreshed single-flight, because the queue scan is -/// proportional to the backlog and the snapshot consumers poll — the stats sampler on its timer, the -/// dashboard at a few hertz, the perf harness at 4 Hz. One scan per TTL window serves all of them. -/// A refresh failure keeps the previous snapshot: stale numbers with an honest timestamp beat an -/// exception on a health endpoint. -/// -/// THE TTL ADAPTS TO WHAT THE SCAN ACTUALLY COSTS, and that is not a refinement. The snapshot used to -/// be stamped with the time the refresh STARTED, so on a large backlog an 80-second scan produced a -/// snapshot already 80 seconds past a 5-second TTL the moment it was stored. The next poll — the -/// worker-health page sits at a few hertz — saw it stale and started another. The single-flight gate -/// kept it to one at a time, but they ran back to back, permanently, against the largest table in the -/// account, on the instance least able to afford it. Measured on the instance that motivated this: -/// full scans of a 743,000-row queue looping continuously, alongside 12,028 per-run counter reads -/// each (see BuildSnapshotAsync). A customer reporting sluggishness had "kept the Worker Health page -/// open" — which is what was driving it. -/// -/// So: stamp on completion, and keep the scan to a fifth of the time — the next one waits at least four -/// times as long as the last one took. The scan also streams: only aggregates and the head of the queue are -/// kept, so a snapshot's memory does not grow with the backlog. +/// The status APIs' view of queued work: the JobManager for what this process holds, the Ready list for what +/// exists durably. One Ready scan (one small row per active run) feeds every count, and only the head of the +/// queue is read row by row, so a snapshot costs the same whatever the backlog. Snapshots are cached and +/// re-scanned no more often than four times the last scan took. /// public class JobQueueStatusReader : IDisposable { private readonly ILogger _logger; private readonly JobManager _jobs; - private readonly JobQueueStore _queue; - private readonly OrchestratorTableStore _store; + private readonly WorkStore _store; private static readonly TimeSpan DefaultTtl = TimeSpan.FromSeconds(5); - /// - /// How many runs will resolve counter rows for. Each is a point - /// read, cheap alone and not in the thousands. A run without one falls back to local numbers, which - /// is the same graceful path a pre-counter run already takes — so this degrades detail, not - /// correctness. Truncation is logged rather than silent. - /// - private const int MaxCounterLookups = 200; - - /// Rows kept from the head of the queue (claim order) for the job listings; the rest are only counted. + /// Rows listed from the head of the queue; runs past it are counted, not listed. internal const int HeadRows = 2_000; + private const int HeadRuns = 50; private volatile QueueSnapshot? _cached; private readonly SemaphoreSlim _refreshGate = new(1, 1); - - /// - /// How long the last successful snapshot took to build. The floor under the effective TTL, so a - /// scan that costs 80s is not re-run 5s later. Volatile read/write of a long via Interlocked — - /// ticks rather than TimeSpan so it stays atomic. - /// private long _lastBuildTicks; - public JobQueueStatusReader(ILogger logger, JobManager jobs, - JobQueueStore queue, OrchestratorTableStore store) + public JobQueueStatusReader(ILogger logger, JobManager jobs, WorkStore store) { _logger = logger; _jobs = jobs; - _queue = queue; _store = store; } - /// The underlying durable queue, for callers that need its maintenance operations. - public JobQueueStore Queue => _queue; + /// A task waiting in storage, as the job listings show it. + public sealed record QueuedRow(string RunName, string TaskId, int Priority, DateTime QueuedUtc, bool Claimed); - /// - /// Per-run slice of the durable queue, derived entirely from the rows the scan already returned. - /// - /// Deliberately carries no counter data. The counter lives in a different table, partitioned per - /// run, so folding it in here meant one point read per distinct run on EVERY refresh — 12,028 of - /// them on the instance that motivated this — even though only - /// ever read those fields. The two paths that poll hardest, and - /// , never used them at all. - /// - public sealed record RunQueueInfo(int Unclaimed, int Claimed, int MinPriority, - DateTime? OldestQueuedUtc); + public sealed record RunQueueInfo(int Unclaimed, int Claimed, int MinPriority, DateTime? OldestQueuedUtc, + int Total, int Done, string? Reference); - /// - /// One scan of the queue table, aggregated the way the status APIs consume it. is the - /// head of the queue only (up to , in claim order); the counts cover every row. - /// - public sealed record QueueSnapshot(DateTime TakenUtc, IReadOnlyList Rows, - int Total, int Unclaimed, int Claimed, DateTime? OldestUnclaimedUtc, - IReadOnlyDictionary ByRun) + /// One Ready scan. is the head of the queue only; the counts cover every run. + public sealed record QueueSnapshot(DateTime TakenUtc, IReadOnlyList Rows, int Total, int Unclaimed, + int Claimed, DateTime? OldestUnclaimedUtc, IReadOnlyDictionary ByRun) { public double AgeSeconds => (DateTime.UtcNow - TakenUtc).TotalSeconds; } - /// - /// The most recent snapshot without ever blocking on storage — for the paths that must stay cheap - /// and non-blocking (GetSnapshot on a PS worker, the stats sampler, the 4 Hz allocation poll). - /// A stale or missing snapshot kicks off a background refresh and returns what exists NOW; the - /// refreshed data is simply what the next call sees. - /// - /// - /// The shortest interval we will re-scan at: never less than the requested age, and never less than - /// four times the last scan took. On a healthy instance the scan is milliseconds and this is just the - /// 5s TTL; on a degraded one it is what stops the refresh loop from consuming the storage account. - /// private TimeSpan EffectiveTtl(TimeSpan? maxAge) { var requested = maxAge ?? DefaultTtl; @@ -116,43 +52,30 @@ private TimeSpan EffectiveTtl(TimeSpan? maxAge) return floor > requested ? floor : requested; } + /// The latest snapshot without blocking; a stale one starts a refresh in the background. public QueueSnapshot? GetCached(TimeSpan? maxAge = null) { var cached = _cached; if (cached == null || DateTime.UtcNow - cached.TakenUtc > EffectiveTtl(maxAge)) - { _ = Task.Run(() => GetAsync(maxAge, CancellationToken.None)); - } return cached; } - /// - /// A snapshot no older than , refreshing if needed. Returns the previous - /// snapshot when the refresh fails, and null only when storage has never answered at all. - /// public async Task GetAsync(TimeSpan? maxAge = null, CancellationToken ct = default) { - var ttl = EffectiveTtl(maxAge); var cached = _cached; - if (cached != null && DateTime.UtcNow - cached.TakenUtc <= ttl) return cached; + if (cached != null && DateTime.UtcNow - cached.TakenUtc <= EffectiveTtl(maxAge)) return cached; await _refreshGate.WaitAsync(ct); try { - // Re-check under the gate, against a TTL recomputed AFTER the wait: whoever held the gate - // may also have just told us how expensive this scan is now. cached = _cached; if (cached != null && DateTime.UtcNow - cached.TakenUtc <= EffectiveTtl(maxAge)) return cached; - - var startedUtc = DateTime.UtcNow; - var snapshot = await BuildSnapshotAsync(ct); - Interlocked.Exchange(ref _lastBuildTicks, (DateTime.UtcNow - startedUtc).Ticks); - _cached = snapshot; - } - catch (OperationCanceledException) when (ct.IsCancellationRequested) - { - throw; + var started = DateTime.UtcNow; + _cached = await BuildSnapshotAsync(ct); + Interlocked.Exchange(ref _lastBuildTicks, (DateTime.UtcNow - started).Ticks); } + catch (OperationCanceledException) when (ct.IsCancellationRequested) { throw; } catch (Exception ex) { _logger.LogWarning(ex, "[JobQueueStatus] Queue snapshot refresh failed — serving previous data"); @@ -161,114 +84,76 @@ private TimeSpan EffectiveTtl(TimeSpan? maxAge) { _refreshGate.Release(); } - return _cached; } - /// - /// One scan of the queue table, aggregated as it streams. No per-run storage reads, and only the head - /// rows are held, so the cost is one scan and the memory is bounded by the number of runs. - /// private async Task BuildSnapshotAsync(CancellationToken ct) { - var head = new List(); - var total = 0; - var unclaimed = 0; - DateTime? oldestUnclaimed = null; + var local = _jobs.GetJobs().Where(j => j.RunName != null && j.Status is "Queued" or "Running") + .GroupBy(j => j.RunName!).ToDictionary(g => g.Key, g => g.Count(), StringComparer.Ordinal); + var byRun = new Dictionary(StringComparer.Ordinal); + var head = new List(); + int total = 0, unclaimed = 0, runsListed = 0; + DateTime? oldest = null; - await foreach (var row in _queue.StreamQueuedAsync(ct)) + await foreach (var e in _store.ReadReadyAsync(1000, ct)) { - total++; - if (head.Count < HeadRows) head.Add(row); - - var info = byRun.TryGetValue(row.RunName, out var existing) - ? existing - : new RunQueueInfo(0, 0, row.Priority, null); - - if (row.Claimed) + var outstanding = Math.Max(0, e.Total - e.Done); + var claimed = Math.Min(outstanding, local.GetValueOrDefault(e.Name)); + total += outstanding; + unclaimed += outstanding - claimed; + if (outstanding > claimed && (oldest == null || e.StartedUtc < oldest)) oldest = e.StartedUtc; + byRun[e.Name] = new RunQueueInfo(outstanding - claimed, claimed, e.Band, e.StartedUtc, e.Total, e.Done, e.Reference); + + if (head.Count < HeadRows && runsListed < HeadRuns && outstanding > claimed) { - info = info with { Claimed = info.Claimed + 1 }; - } - else - { - unclaimed++; - if (oldestUnclaimed == null || row.QueuedUtc < oldestUnclaimed) oldestUnclaimed = row.QueuedUtc; - info = info with + runsListed++; + foreach (var t in await _store.GetTasksAsync(e.RunKey, 'P', ct)) { - Unclaimed = info.Unclaimed + 1, - OldestQueuedUtc = info.OldestQueuedUtc is { } o && o <= row.QueuedUtc ? o : row.QueuedUtc, - }; + if (head.Count >= HeadRows) break; + head.Add(new QueuedRow(e.Name, t.TaskId, e.Band, e.StartedUtc, false)); + } } - if (row.Priority < info.MinPriority) info = info with { MinPriority = row.Priority }; - byRun[row.RunName] = info; } - // Stamped on COMPLETION, not on entry. A scan that took longer than the TTL would otherwise - // return a snapshot that is already expired, and the next poll would start another immediately. - return new QueueSnapshot(DateTime.UtcNow, head, total, unclaimed, total - unclaimed, - oldestUnclaimed, byRun); + return new QueueSnapshot(DateTime.UtcNow, head, total, unclaimed, total - unclaimed, oldest, byRun); } // ─── Merged views ─── - /// - /// The JobManager summary with the durable backlog folded in: Queued counts local jobs PLUS - /// unclaimed queue rows (claimed rows are already represented by local records on the instance - /// that holds them), and OldestQueuedUtc considers both sources. - /// + /// The JobManager summary with the durable backlog folded in. public async Task GetSummaryAsync(CancellationToken ct = default) { var summary = _jobs.GetSummary(); summary.QueuedLocal = summary.Queued; - var snap = await GetAsync(ct: ct); if (snap == null) return summary; summary.QueuedDurable = snap.Unclaimed; summary.Queued += snap.Unclaimed; - - if (snap.OldestUnclaimedUtc is { } oldest - && (summary.OldestQueuedUtc == null || oldest < summary.OldestQueuedUtc)) - { + if (snap.OldestUnclaimedUtc is { } oldest && (summary.OldestQueuedUtc == null || oldest < summary.OldestQueuedUtc)) summary.OldestQueuedUtc = oldest; - } - return summary; } - /// - /// The job listing the worker-health page shows: local records merged with the unclaimed durable - /// backlog. Durable rows only participate when the status filter admits "Queued" — every other - /// status describes work an instance has already claimed, which local records cover. - /// + /// Local job records plus the head of the durable queue, for the job listing. public async Task> GetJobDetailsAsync(string? runName = null, string? status = null, int limit = 100, CancellationToken ct = default) { var local = _jobs.GetJobDetails(runName, status, limit); - - var includeDurable = string.IsNullOrEmpty(status) - || status.Equals("Queued", StringComparison.OrdinalIgnoreCase); - if (!includeDurable) return local; + if (!string.IsNullOrEmpty(status) && !status.Equals("Queued", StringComparison.OrdinalIgnoreCase)) return local; var snap = await GetAsync(ct: ct); if (snap == null || snap.Rows.Count == 0) return local; var now = DateTime.UtcNow; var merged = new List(local); - foreach (var row in snap.Rows) { - // Claimed rows are queued or running inside some instance's JobManager; on this instance - // they are the local records already in the list. The id guard closes the enqueue/claim - // race window on top of that. - if (row.Claimed) continue; - if (!string.IsNullOrEmpty(runName) - && !string.Equals(row.RunName, runName, StringComparison.OrdinalIgnoreCase)) continue; - + if (!string.IsNullOrEmpty(runName) && !string.Equals(row.RunName, runName, StringComparison.OrdinalIgnoreCase)) continue; var id = $"{row.RunName}-{row.TaskId}"; if (_jobs.IsQueuedOrRunning(id)) continue; - merged.Add(new JobDetail { Id = id, @@ -280,21 +165,10 @@ public async Task> GetJobDetailsAsync(string? runName = null, st WaitSeconds = Math.Max(0, (now - row.QueuedUtc).TotalSeconds), }); } - - // Same ordering contract as JobManager.GetJobDetails, re-applied across the merged set. - return merged - .OrderBy(j => j.Priority) - .ThenBy(j => j.QueuedUtc) - .Take(limit) - .ToList(); + return merged.OrderBy(j => j.Priority).ThenBy(j => j.QueuedUtc).Take(limit).ToList(); } - /// - /// Run summaries with durable truth folded in: Total from the run's counter row, Queued including - /// the unclaimed backlog, Completed at least the durably-terminal count. Runs that exist only in - /// the table — a backlog nothing here has claimed yet — get a synthesized entry, because a run the - /// page cannot see is a run nobody can cancel. - /// + /// Run summaries: local job records with each active run's durable size and progress folded in. public async Task> GetRunSummariesAsync(CancellationToken ct = default) { var summaries = _jobs.GetRunSummaries(); @@ -302,68 +176,22 @@ public async Task> GetRunSummariesAsync(CancellationToken ct if (snap == null) return summaries; var byName = summaries.ToDictionary(s => s.Name, StringComparer.Ordinal); - foreach (var (run, info) in snap.ByRun) { - if (byName.TryGetValue(run, out var summary)) - { - summary.Queued += info.Unclaimed; - } - else - { - // Claimed rows here belong to another instance's buffer — not distinguishable from - // queued at this distance, and not terminal, so Queued is the honest bucket. - var synthesized = new JobRunSummary - { - Name = run, - Priority = info.MinPriority, - Queued = info.Unclaimed + info.Claimed, - Total = info.Unclaimed + info.Claimed, - }; - summaries.Add(synthesized); - byName[run] = synthesized; - } - } - - // Counter rows, resolved HERE rather than in the snapshot, because this is the only caller that - // reads them. Each is a point read on the run's own task partition. - // - // The counter is a run's true size and durable progress — a restart or a multi-instance claim - // pattern otherwise shrinks Total to whatever this node happened to process. It covers runs with - // queue rows and runs whose backlog is fully claimed alike, which is why this is one pass over - // the active summaries rather than two loops keyed off the snapshot. - var ordered = summaries - .OrderBy(r => r.Priority) - .ThenByDescending(r => r.StartedUtc) - .ToList(); - - var active = ordered.Where(s => s.Queued + s.Running > 0).ToList(); - var lookups = Math.Min(active.Count, MaxCounterLookups); - - for (var i = 0; i < lookups; i++) - { - try - { - if (await _store.GetCounterAsync(active[i].Name, ct) is { } counter) - Overlay(active[i], counter.Remaining, counter.Total); - } - catch (Exception ex) + if (!byName.TryGetValue(run, out var summary)) { - _logger.LogDebug(ex, "[JobQueueStatus] Counter read failed for {Run}", active[i].Name); + summary = new JobRunSummary { Name = run, Priority = info.MinPriority, StartedUtc = info.OldestQueuedUtc }; + summaries.Add(summary); + byName[run] = summary; } + summary.Reference ??= info.Reference; + summary.Queued += info.Unclaimed; + summary.Total = Math.Max(summary.Total, info.Total); + summary.Completed = Math.Max(summary.Completed, info.Done - summary.Failed); + summary.CompletedUtc = null; } - if (active.Count > lookups) - { - // Said out loud: a silent cap here would read as "these runs have no durable progress" - // rather than "we did not look", and the two are indistinguishable in the UI. - _logger.LogDebug( - "[JobQueueStatus] Resolved counters for {Looked} of {Active} active runs (cap {Cap}) — " + - "the remainder report local numbers only", - lookups, active.Count, MaxCounterLookups); - } - - return ordered; + return summaries.OrderBy(r => r.Priority).ThenByDescending(r => r.StartedUtc).ToList(); } public void Dispose() @@ -371,19 +199,4 @@ public void Dispose() GC.SuppressFinalize(this); _refreshGate.Dispose(); } - - /// - /// Fold a run's counter into its summary. Total is authoritative when present. The durably-terminal - /// count (Total − Remaining) spans Completed, Failed and Cancelled across every instance and every - /// restart; local Failed is kept (it is a lower bound) and the rest raises Completed. - /// - private static void Overlay(JobRunSummary summary, int? remaining, int? total) - { - if (total is not { } t || remaining is not { } r) return; - - if (t > summary.Total) summary.Total = t; - - var durablyDone = Math.Max(0, t - r); - summary.Completed = Math.Max(summary.Completed, durablyDone - summary.Failed); - } } diff --git a/Services/Orchestration/MarkerNotPersistedException.cs b/Services/Orchestration/MarkerNotPersistedException.cs deleted file mode 100644 index f619255..0000000 --- a/Services/Orchestration/MarkerNotPersistedException.cs +++ /dev/null @@ -1,16 +0,0 @@ -namespace Craft.Orchestration; - -/// -/// The durable "Running" marker for a task did not reach storage — it timed out, or the flush carrying -/// it failed. -/// -/// This is a DEFERRAL, not a task failure, and the distinction is the whole point of the type. Nothing -/// was written, so the task is still Pending in storage; running it anyway would defeat the -/// AttemptCount/MaxRetries bound the marker exists to provide, and failing it would lose work that never -/// ran. The correct response is to release the slot and retry. -/// -public sealed class MarkerNotPersistedException : Exception -{ - public MarkerNotPersistedException(string message) : base(message) { } - public MarkerNotPersistedException(string message, Exception inner) : base(message, inner) { } -} diff --git a/Services/Orchestration/OrchestratorRun.cs b/Services/Orchestration/OrchestratorRun.cs deleted file mode 100644 index 9808b28..0000000 --- a/Services/Orchestration/OrchestratorRun.cs +++ /dev/null @@ -1,40 +0,0 @@ -namespace Craft.Orchestration; - -public class OrchestratorRun -{ - public string Name { get; set; } = string.Empty; - public string? Reference { get; set; } - public string Status { get; set; } = "Pending"; - // 4 matches what every live enqueue path actually passes when a caller sets nothing — a run row - // rehydrated without a stored priority must not come back HIGHER than it originally ran. - public int Priority { get; set; } = 4; - public DateTime StartedUtc { get; set; } - public DateTime? CompletedUtc { get; set; } - public List Tasks { get; set; } = []; - public string? TaskScriptName { get; set; } - public string? PostExecFunctionName { get; set; } - public string? PostExecParametersJson { get; set; } - // null | "Pending" | "Running" | "Completed" | "Failed" | "Abandoned" - // "Failed" is retryable — recovery picks it back up on the next host start. "Abandoned" is the - // terminal one: retries are spent, and the run's result rows have been cleaned up. - public string? PostExecStatus { get; set; } - - /// - /// How many times post-execution has been attempted. Bounds the retry that - /// ResumeInterruptedRunsAsync performs for a "Failed" post-execution, the same way - /// bounds task recovery — without it a - /// permanently-failing aggregation would be retried on every host start forever. - /// - public int PostExecAttemptCount { get; set; } - - public string? ParentRunName { get; set; } - - /// - /// Sequential execution mode. When true the run's tasks are dispatched ONE AT A TIME, in ascending - /// (batch) order: only the current task is ever enqueued, - /// and the next is enqueued when it reaches a terminal state. Runs on any free worker (no pinning) — - /// the durable queue simply never holds more than one of this run's tasks at once. The default (false) - /// is the fan-out behaviour: every task is enqueued up front and drained in parallel by the pool. - /// - public bool Sequential { get; set; } -} diff --git a/Services/Orchestration/OrchestratorService.cs b/Services/Orchestration/OrchestratorService.cs index 319b4d6..5010776 100644 --- a/Services/Orchestration/OrchestratorService.cs +++ b/Services/Orchestration/OrchestratorService.cs @@ -1,5 +1,4 @@ using System.Collections.Concurrent; -using System.Diagnostics.CodeAnalysis; using System.Text.Json; using System.Text.Json.Serialization; using Craft.Configuration; @@ -9,178 +8,36 @@ namespace Craft.Orchestration; -// OrchestrationResults removed — results are now persisted to the -// CippOrchestratorResults Azure Table via OrchestratorTableStore. - /// -/// Lightweight replacement for Azure Durable Functions orchestration. -/// Manages fan-out/fan-in runs with crash-resilient task tracking. +/// Fan-out/fan-in runs on top of . A run is created durably, its tasks are claimed by +/// and executed here, and each finish is one transaction that also advances the run's +/// counts; the finish that completes the tasks queues the run's PostExecution as one more task. Storage holds +/// all of it, so there is nothing to recover after a restart: unfinished claims lapse and are taken again. /// -/// Flow: -/// 1. Scheduler triggers StartOrResumeRun with a planner script -/// 2. Planner runs on bg pool, returns JSON array of tasks -/// 3. Each task is dispatched through JobManager with priority ordering -/// 4. State is persisted to Azure Table Storage after every state change -/// 5. On restart, interrupted tasks resume from where they left off -/// 6. After 3 interruptions (host crash/reboot), a task is marked Failed -/// 7. PostExecStatus tracks PostExecution lifecycle for crash resilience +/// This process keeps only what is in flight: the pump's claimed buffer and the jobs running from it. /// -public class OrchestratorService : IJobDescriptorStateWriter, IDisposable +public class OrchestratorService : IJobDescriptorStateWriter { internal readonly ILogger _logger; private readonly PowerShellRunnerService _psRunner; private readonly BackgroundTaskLimiter _limiter; private readonly JobManager _jobManager; - private readonly OrchestratorTableStore _store; - - /// The durable job queue: every task is enqueued here and dispatched by the pump. - private readonly JobQueueStore _queue; - private readonly OrchestratorStatusWriter _writer; + private readonly WorkStore _store; + private readonly ResultStore _results; private readonly CraftSettings _settings; - private readonly object _lock = new(); + private readonly FinishBatcher _finisher; private readonly ConcurrentDictionary _activePlanners = new(); - private readonly ConcurrentDictionary _activeRuns = new(); - private readonly ConcurrentDictionary> _childRuns = new(); - - /// - /// Child runs seen in storage at startup but not yet processed by . - /// They count as incomplete: without this, a parent processed BEFORE its child would find the child - /// absent from (nothing has resumed it yet) and conclude it had finished. - /// Each name is cleared as its run is processed, whichever way that goes. - /// - private readonly ConcurrentDictionary _recoveringChildren = new(); - - /// - /// Child runs queued through the bridge but not yet started, keyed by child name with a count - /// of outstanding queue entries. They count as incomplete for the same reason recovering - /// children do: between a task's script enqueuing a sub-orchestration and DrainPending getting - /// it into there is otherwise nothing for the parent's completion - /// check to see — and that window contains the parent's own last-task completion, because the - /// drain runs in the background while the enqueuing task is marked terminal immediately. - /// - private readonly ConcurrentDictionary _pendingChildRuns = new(); - private readonly ConcurrentDictionary _cancelledRuns = new(); - - /// - /// Last status line emitted per run — the (completed, failed, running, pending) tuple and when. Lets - /// skip re-emitting an identical line every 60s for a run that has not - /// changed (the dominant log volume at scale — thousands of runs parked at "0 running / N pending"), - /// while a slow heartbeat still proves a long-lived run is alive. Dropped at finalize. - /// - private readonly ConcurrentDictionary _lastStatusLog = new(); - - /// - /// Per-run re-drive backoff: when the storage verification in may - /// next run, and the interval it grew to. The re-drive is a watchdog for the rare orphaned-Pending task; - /// in steady state it reads storage and finds nothing, so once it does it backs off geometrically instead - /// of paying a full index read (+ a point read per candidate) on every 60s tick for every live run — the - /// dominant per-tick storage cost at scale. Snaps back to the base interval the moment it finds an orphan. - /// Dropped at finalize. - /// - private readonly ConcurrentDictionary _redriveBackoff = new(); - - /// Runs with a re-drive verification in flight; a tick skips them instead of starting another. - private readonly ConcurrentDictionary _redriveInFlight = new(); - - /// Caps concurrent re-drive verifications across all runs; a tick that finds none free skips the run. - private readonly SemaphoreSlim _redriveSlots = new(RedriveMaxConcurrent, RedriveMaxConcurrent); + private readonly ConcurrentDictionary _scripts = new(StringComparer.OrdinalIgnoreCase); + private readonly ConcurrentDictionary _lastStatusLog = new(); - private const int RedriveMaxConcurrent = 8; + /// Identifies this process's claims, and is what a lease is checked against. + public string Owner { get; } - public void Dispose() - { - _redriveSlots.Dispose(); - GC.SuppressFinalize(this); - } - - /// Per-run status/re-drive tick cadence, from Orchestrator:StatusTimerIntervalSeconds. - private readonly TimeSpan _statusInterval; - - /// Re-drive backoff floor — one tick. The interval grows from here to . - private readonly TimeSpan _redriveBase; - - /// Whether the re-drive backoff is active, from Orchestrator:RedriveBackoff. - private readonly bool _redriveBackoffEnabled; - - /// Whether pending tasks shed their Parameters payload, from Orchestrator:ShedPendingParameters. - private readonly bool _shedParameters; - - private static long _redriveStorageReads; - - /// - /// Count of storage verifications the re-drive has performed (a - /// call: one index-partition read + a point read per candidate). Instrumentation for the perf harness — - /// the backoff's whole purpose is to hold this down at high live-run counts. - /// - public static long RedriveStorageReads => Interlocked.Read(ref _redriveStorageReads); - - /// - /// Resolved task-script path per run. One entry per RUN (not per task), so a 738-task fan-out costs - /// one string reference instead of 738 captured ones. Populated at dispatch, dropped at finalize. - /// - private readonly ConcurrentDictionary _taskScriptPaths = new(); - - /// - /// How many times a run's post-execution may be attempted before it is abandoned. Matches the - /// per-task cap so recovery behaves the same at both levels: retry a few times across restarts, - /// then stop and release the storage rather than retrying forever. - /// - private const int MaxPostExecAttempts = 3; + /// How long a claim is held before anyone may take it back. Longer than any task may run. + public TimeSpan Lease { get; } - /// - /// How often an UNCHANGED run still emits a status line, so a long-lived run proves it is alive without - /// logging the identical line on every 60s tick. A real status change always logs immediately. - /// private static readonly TimeSpan StatusHeartbeat = TimeSpan.FromMinutes(10); - /// - /// Runs whose finalize has been claimed, so it happens once. Claimed in CheckRunCompletion, - /// released on the deferral/failure paths there, and cleared in DispatchPendingTasksAsync when a - /// run becomes live again. In-memory only: after a restart nothing has been finalized yet, so an - /// empty set is the correct starting state. - /// - private readonly ConcurrentDictionary _finalizingRuns = new(); - - /// - /// Sequential runs whose single driver job is currently executing. A sequential run's steps all run on - /// one pinned worker inside one driver, and the not-yet-run steps deliberately have no queue row — so - /// while a driver is active the re-drive must not treat those steps as orphaned and enqueue them (which - /// would spawn a second driver). Set when the driver starts, cleared when it finishes; in-memory only, - /// so after a crash it is empty and the re-drive/resume correctly re-triggers the driver. - /// - private readonly ConcurrentDictionary _activeSequentialDrivers = new(); - - /// - /// Get the Reference for a given run name, or null if not found/no reference set. - /// - public string? GetRunReference(string runName) - { - return _activeRuns.TryGetValue(runName, out var run) ? run.Reference : null; - } - - /// - /// Find a run name by its Reference value (exact match). - /// - public string? FindRunByReference(string reference) - { - return _activeRuns.Values - .FirstOrDefault(r => string.Equals(r.Reference, reference, StringComparison.OrdinalIgnoreCase)) - ?.Name; - } - - /// - /// Completes once startup recovery has finished — or been abandoned, see . - /// claims nothing before it. A claim taken earlier rehydrates the run into - /// _activeRuns ahead of recovery, recovery's own copy then loses the TryAdd, and the live graph - /// keeps the dead process's stale Running markers instead of recovery's reset. - /// - public Task RecoveryDone => _recoveryDone.Task; - private readonly TaskCompletionSource _recoveryDone = new(TaskCreationOptions.RunContinuationsAsynchronously); - - /// Open the claim gate. Called from a finally, so a recovery that throws or never runs - /// (shutdown mid-startup, storage down) still releases the pump rather than wedging it. - public void MarkRecoveryDone() => _recoveryDone.TrySetResult(); - private static readonly JsonSerializerOptions s_jsonOptions = new() { WriteIndented = true, @@ -188,14 +45,22 @@ public void Dispose() DefaultIgnoreCondition = JsonIgnoreCondition.WhenWritingNull }; + /// Kept for callers that read it; there is no re-drive any more. + public static long RedriveStorageReads => 0; + + /// Completes once startup has initialized storage; claims nothing before it. + public Task RecoveryDone => _recoveryDone.Task; + private readonly TaskCompletionSource _recoveryDone = new(TaskCreationOptions.RunContinuationsAsynchronously); + public void MarkRecoveryDone() => _recoveryDone.TrySetResult(); + public OrchestratorService( ILogger logger, PowerShellRunnerService psRunner, BackgroundTaskLimiter limiter, JobManager jobManager, - OrchestratorTableStore store, - JobQueueStore queue, - OrchestratorStatusWriter writer, + WorkStore store, + ResultStore results, + IConfiguration configuration, CraftSettings settings) { _logger = logger; @@ -203,135 +68,25 @@ public OrchestratorService( _limiter = limiter; _jobManager = jobManager; _store = store; - _queue = queue; - _writer = writer; + _results = results; _settings = settings; + Owner = Environment.GetEnvironmentVariable("HOSTNAME") ?? $"instance-{Environment.ProcessId}"; + Lease = TimeSpan.FromSeconds(Math.Max(60, configuration.GetValue("JobQueueLeaseSeconds", 1800))); + _finisher = new FinishBatcher(store, logger); + _store.AfterFinish = AfterFinishAsync; - // Status/re-drive tick cadence. Configurable so a constrained deployment can slow it (fewer ticks = - // less per-run overhead at high live-run counts) and so the perf harness can compress it to exercise - // the re-drive backoff quickly. The backoff floor is one tick. - _statusInterval = TimeSpan.FromSeconds(Math.Max(1, _settings.Orchestrator.StatusTimerIntervalSeconds)); - _redriveBase = _statusInterval; - _redriveBackoffEnabled = _settings.Orchestrator.RedriveBackoff; - _shedParameters = _settings.Orchestrator.ShedPendingParameters; - - // The queue holds descriptors; this is how they become work again at dispatch time, and how - // operator changes to a queued task are made durable. _jobManager.SetWorkResolver(ResolveTaskWorkAsync); _jobManager.SetDescriptorStateWriter(this); } - // ─── IJobDescriptorStateWriter ─── - // Operator actions on a QUEUED task. Both mutate the live run graph (so the in-memory view and - // CheckRunCompletion stay consistent) and queue a durable write through the same coalescing - // status writer the task lifecycle already uses — which guarantees a flush before the run - // finalizes and a final drain on shutdown. - - /// Persist a per-task priority override so recovery re-queues at the operator's priority. - public void PriorityChanged(JobDescriptor descriptor, int newPriority) - { - if (!TryFindLive(descriptor, out _, out var task)) return; - - // Rehydrate a shed payload before the durable write below (Replace mode) overwrites the stored - // ParametersJson with null. This is a rare, operator-initiated path, so the blocking read is fine. - if (_shedParameters && task.Parameters == null) - { - var p = _store.GetTaskParametersAsync(descriptor.RunName, task.Id).GetAwaiter().GetResult() ?? []; - lock (_lock) { task.Parameters ??= p; } - } - - lock (_lock) task.Priority = newPriority; - _writer.QueueTask(descriptor.RunName, task); - } - - /// - /// Persist a cancellation. Without this the task row stays Pending and - /// re-queues it after a restart — the job comes back. - /// - public void Cancelled(JobDescriptor descriptor) - { - if (!TryFindLive(descriptor, out var run, out var task)) return; - CancelLiveTask(run, task); - } - - /// - /// Mark a live task Cancelled in the run graph and queue the durable write. Returns false when the - /// task is already terminal — nothing to cancel, nothing to persist. With - /// , a Running task is also refused: the callers that pass it - /// found the task via a queue snapshot that may be seconds old, and cancelling a task that has - /// since dispatched would race its own completion write. - /// - private bool CancelLiveTask(OrchestratorRun run, OrchestratorTaskItem task, bool requirePending = false) - { - lock (_lock) - { - if (task.Status is "Completed" or "Failed" or "Cancelled") return false; - if (requirePending && task.Status != "Pending") return false; - task.Status = "Cancelled"; - task.LastError = "Cancelled by user"; - task.CompletedUtc = DateTime.UtcNow; - task.Parameters = null!; - // Cancelling the last outstanding task can complete the run. - CheckRunCompletion(run); - } - _writer.QueueTask(run.Name, task); - return true; - } - - /// - /// Cancel a task that exists only as a durable queue row — queued in the table, not (yet) claimed - /// into this instance's JobManager, which is where the worker-health page now sees most of a - /// backlog. - /// - /// Order is load-bearing: the run graph is marked terminal and the write queued BEFORE the queue - /// row is removed, because the orphan re-drive reads "Pending with no row" as work to re-queue — - /// remove the row first and the cancel un-does itself within a minute. The row removal itself is - /// best-effort: a row left behind is claimed, found terminal at rehydration, skipped and released. - /// - /// Returns false when the run is not live on this node or the task is already terminal; callers - /// should treat that as "nothing cancellable here". - /// - public async Task TryCancelQueuedTaskAsync(string runName, string taskId) - { - if (!TryFindLive(new JobDescriptor(runName, taskId, 0), out var run, out var task)) return false; - if (!CancelLiveTask(run, task, requirePending: true)) return false; - - try - { - await _queue.RemoveTaskAsync(runName, taskId); - } - catch (Exception ex) - { - _logger.LogWarning(ex, - "[Scheduler] Cancelled {Run}/{Task} but could not remove its queue row — it will be skipped at claim", - runName, taskId); - } - - return true; - } - - /// - /// Resolve a descriptor to its LIVE run and task instances. Operator actions only apply to QUEUED - /// jobs, whose run is by definition still active, so a miss means the job is no longer - /// cancellable/reprioritizable and there is nothing to persist. - /// - private bool TryFindLive(JobDescriptor descriptor, - [NotNullWhen(true)] out OrchestratorRun? run, - [NotNullWhen(true)] out OrchestratorTaskItem? task) - { - task = null; - if (!_activeRuns.TryGetValue(descriptor.RunName, out run)) return false; - lock (_lock) task = run.Tasks.FirstOrDefault(t => t.Id == descriptor.TaskId); - return task != null; - } + // ── starting runs ── /// - /// Start a new run or resume an interrupted one. - /// Called by the SchedulerService when an Orchestrator-type task fires. + /// Start a planner-based run: the planner script prints the task list. Skipped while a run of this name is + /// still going, so a timer that fires again does not stack outings. /// public async Task StartOrResumeRun(string name, string plannerPath, string taskPath, int priority, CancellationToken ct) { - // Prevent duplicate concurrent planners for the same run if (!_activePlanners.TryAdd(name, true)) { _logger.LogInformation("[Scheduler] Run {Name} already in progress, skipping", name); @@ -340,77 +95,17 @@ public async Task StartOrResumeRun(string name, string plannerPath, string taskP try { - await _store.InitializeAsync(); - await _queue.InitializeAsync(ct); - var run = await _store.GetRunAsync(name); - - if (run != null && run.Status == "Running") + if (await IsActiveAsync(name, ct)) { - // If tasks are already dispatched for this run, skip - if (_activeRuns.ContainsKey(name)) - { - _logger.LogInformation("[Scheduler] Run {Name} tasks already dispatched, skipping", name); - return; - } - - // Recover interrupted tasks - var recovered = 0; - var tasksToUpdate = new List(); - lock (_lock) - { - foreach (var task in run.Tasks.Where(t => t.Status == "Running")) - { - task.AttemptCount++; - if (task.AttemptCount >= 3) - { - task.Status = "Failed"; - task.LastError = $"Cancelled {task.AttemptCount} times by host interruption"; - _logger.LogWarning("[Scheduler] Task {TaskId} permanently failed after {Attempts} cancellations", - task.Id, task.AttemptCount); - } - else - { - task.Status = "Pending"; - _logger.LogInformation("[Scheduler] Resuming task {TaskId} attempt {Attempt}/3", - task.Id, task.AttemptCount + 1); - } - tasksToUpdate.Add(task); - recovered++; - } - } - - var pendingCount = run.Tasks.Count(t => t.Status == "Pending"); - if (pendingCount > 0) - { - if (recovered > 0) - { - foreach (var t in tasksToUpdate) - await _store.UpsertTaskAsync(run.Name, t); - } - _logger.LogInformation("[Scheduler] Resuming run {Name}: {Pending}/{Total} pending", - name, pendingCount, run.Tasks.Count); - await DispatchPendingTasksAsync(run, taskPath, run.Priority, ct); - return; - } - - // All tasks finished — finalize, and STOP. Finalize dispatches post-execution - // asynchronously, and its success path deletes the run's partitions by name - // (CleanupRunAsync). Falling through to start a fresh same-named outing here raced - // that delete and lost the new run's rows mid-flight; the next scheduler tick starts - // the fresh outing cleanly instead. - await FinalizeRunAsync(run); + _logger.LogInformation("[Scheduler] Run {Name} tasks already dispatched, skipping", name); return; } - // Start a new run _logger.LogInformation("[Scheduler] Starting orchestrator: {Name}", name); - string output; try { - output = await _limiter.RunAsync( - () => _psRunner.ExecuteScriptWithOutput(plannerPath), - $"Planner-{name}", ct); + output = await _limiter.RunAsync(() => _psRunner.ExecuteScriptWithOutput(plannerPath), $"Planner-{name}", ct); } catch (Exception ex) { @@ -426,23 +121,8 @@ public async Task StartOrResumeRun(string name, string plannerPath, string taskP return; } - run = new OrchestratorRun - { - Name = name, - Status = "Running", - Priority = priority, - StartedUtc = DateTime.UtcNow, - Tasks = tasks, - TaskScriptName = Path.GetFileNameWithoutExtension(taskPath) - }; - - await _store.UpsertRunAsync(run); - await _store.UpsertTaskBatchAsync(name, tasks); - // Seed the durable outstanding-task count alongside the tasks themselves, so completion - // is answerable from storage instead of by walking this run graph. - await _store.InitRemainingAsync(name, tasks.Count, ct); - _logger.LogInformation("[Scheduler] Run {Name} created with {Count} tasks at P{Priority}", name, tasks.Count, priority); - await DispatchPendingTasksAsync(run, taskPath, priority, ct); + await CreateAsync(name, tasks, priority, Path.GetFileNameWithoutExtension(taskPath), null, null, null, null, null, + false, ct); } finally { @@ -450,19 +130,12 @@ public async Task StartOrResumeRun(string name, string plannerPath, string taskP } } - /// - /// Start a planner-based orchestrator run using the standard naming convention. - /// Resolves planner and task scripts from the command name, then delegates to StartOrResumeRun. - /// Called by OrchestratorBridge.QueuePlannerRun (fire-and-forget from PowerShell). - /// + /// Start a planner run by command name: planner Start-X, tasks Invoke-XTask. public async Task StartPlannerRunAsync(string command, int priority, CancellationToken ct) { var plannerFunc = _psRunner.FindScript(command); - var baseName = command.StartsWith("Start-", StringComparison.OrdinalIgnoreCase) - ? command[6..] - : command; - var taskScriptName = $"Invoke-{baseName}Task"; - var taskFunc = _psRunner.FindScript(taskScriptName); + var baseName = command.StartsWith("Start-", StringComparison.OrdinalIgnoreCase) ? command[6..] : command; + var taskFunc = _psRunner.FindScript($"Invoke-{baseName}Task"); if (plannerFunc != null && taskFunc != null) { @@ -472,1781 +145,524 @@ public async Task StartPlannerRunAsync(string command, int priority, Cancellatio else { _logger.LogWarning("[Orchestrator] Scripts not found for planner run: {Command} planner={Planner} task={Task}", - command, command, taskScriptName); + command, command, $"Invoke-{baseName}Task"); } } /// - /// Resume any runs that were interrupted by a previous process crash. - /// Called once on application startup. + /// Start a run from a pre-built batch (OrchestratorBridge). The batch is a JSON Lines file + /// (, deleted on every path) or a JSON array string. Returns whether a run + /// was created: false when the batch is empty or a run of this name is still going. /// - public async Task ResumeInterruptedRunsAsync(CancellationToken ct) + public async Task StartFromBatchAsync(string name, string batchJson, int priority, + string? postExecFunctionName, string? postExecParametersJson, CancellationToken ct, + string? parentRunName = null, string? reference = null, string? batchFilePath = null, + bool sequential = false, string? parentRunKey = null, string? childKey = null) { - await _store.InitializeAsync(); - await _queue.InitializeAsync(ct); - - // Rebuild parent→child links BEFORE processing anything. _childRuns is in-memory, so without - // this a resumed parent has no registered children, AllChildRunsComplete answers true, and the - // parent can finalize (and fire PostExecution) while its children are still running — exactly - // what the child-run guard exists to prevent. One partition scan, no task rows. - var summaries = await _store.ListRunSummariesAsync(); - var runningRuns = summaries.Where(s => s.Status == "Running").Select(s => s.Name) - .ToHashSet(StringComparer.Ordinal); - var reattached = 0; - foreach (var child in summaries) - { - if (string.IsNullOrEmpty(child.ParentRunName)) continue; - // Rows written before the self-parent guard can record a run as its own parent; - // reattaching one would rebuild the self-wait this fix removes. - if (child.ParentRunName == child.Name) continue; - if (child.Status is not "Running") continue; // terminal children cannot block a parent - // The parent must itself be resuming. Lineage can point at a run that already - // finalized — a run queued from PostExecution records its finished spawner as parent — - // and reattaching those would build bags no completion check consults, or worse, gate - // the NEXT outing of a recurring parent name on a leftover of the previous one. - if (!runningRuns.Contains(child.ParentRunName)) continue; - _childRuns.GetOrAdd(child.ParentRunName, _ => new ConcurrentBag()).Add(child.Name); - _recoveringChildren.TryAdd(child.Name, true); - reattached++; - } - if (reattached > 0) - _logger.LogInformation("[Scheduler] Reattached {Count} in-flight child runs to their parents", reattached); - - // Recovery emits ONE aggregate line, not a handful per run: at scale a crash-loop replayed thousands - // of per-run "Found/Resuming/Released/Dispatched" lines on every restart. Per-run detail is kept at - // Debug; the counts below carry the summary. Genuine problems (unresumable, post-exec abandoned) still - // log at their own level as they happen. - var resumed = 0; var pendingTotal = 0; var postExecResumed = 0; - var staleReleased = 0; var finalizedNow = 0; var unresumable = 0; var postExecGaveUp = 0; - foreach (var runName in summaries.Select(s => s.Name)) - { - try - { - var run = await _store.GetRunAsync(runName); - if (run == null) continue; - - // Check for runs whose PostExec was pending/running when we crashed, or failed outright. - // - // "Failed" belongs here: post-execution IS the aggregation, so a failed one means the - // run's whole point never happened (no cached permissions, no applied standards) and its - // Results rows were never cleaned up, because CleanupRunAsync only runs on success. This - // used to be excluded, which quietly contradicted the catch block's own comment that a - // failure "will be retried on next startup". - if (run.Status is "Completed" or "CompletedWithErrors" - && run.PostExecStatus is "Pending" or "Running" or "Failed") - { - if (run.PostExecAttemptCount >= MaxPostExecAttempts) - { - // Out of retries. Say so once and release the storage, rather than re-reading - // this run's rows on every start for the life of the deployment. - _logger.LogError( - "[Scheduler] PostExecution for {Name} failed {Count} times — giving up and cleaning up results", - run.Name, run.PostExecAttemptCount); - run.PostExecStatus = "Abandoned"; - await _store.UpsertRunAsync(run); - await _store.CleanupRunAsync(run.Name); - await _queue.RemoveRunAsync(run.Name, run.StartedUtc, ct); - postExecGaveUp++; - continue; - } - - _logger.LogDebug( - "[Scheduler] Resuming interrupted PostExecution for run: {Name} (PostExecStatus={Status}, attempt {Attempt}/{Max})", - run.Name, run.PostExecStatus, run.PostExecAttemptCount + 1, MaxPostExecAttempts); - DispatchPostExecution(run); - postExecResumed++; - continue; - } - - if (run.Status != "Running") continue; - - _logger.LogDebug("[Scheduler] Found interrupted run: {Name}", run.Name); - - // Use the stored task script name, fall back to naming convention - var taskPath = !string.IsNullOrEmpty(run.TaskScriptName) - ? _psRunner.FindScript(run.TaskScriptName) - : FindTaskScript(run.Name); - if (taskPath == null) - { - _logger.LogWarning("[Scheduler] Cannot resume {Name}: task script not found (tried {Script})", - run.Name, run.TaskScriptName ?? $"Invoke-{run.Name}Task"); - unresumable++; - continue; - } - - // Recover interrupted tasks - var tasksToUpdate = new List(); - lock (_lock) - { - foreach (var task in run.Tasks.Where(t => t.Status == "Running")) - { - task.AttemptCount++; - if (task.AttemptCount >= 3) - { - task.Status = "Failed"; - task.LastError = $"Cancelled {task.AttemptCount} times by host interruption"; - } - else - { - task.Status = "Pending"; - } - tasksToUpdate.Add(task); - } - } - - // Persist recovered task state changes - foreach (var t in tasksToUpdate) - await _store.UpsertTaskAsync(run.Name, t); - - var pending = run.Tasks.Count(t => t.Status == "Pending"); - if (pending > 0) - { - // Hand back the claims the dead process was holding. Reaching here means this run was - // interrupted, so every lease on its rows belongs to a process that no longer exists — - // but the lease itself is still live as far as storage is concerned, so nothing can - // claim those rows until it lapses. Re-dispatch below deliberately does not paper over - // that by writing duplicate rows, which is what used to hide it, so without this the - // run stalls for the remainder of the lease: measured as 12 tasks Pending with nothing - // running for the balance of a 30 minute lease after a kill. - try - { - var released = await _queue.ReleaseRunClaimsAsync(run.Name, ct); - if (released > 0) - { - staleReleased += released; - _logger.LogDebug( - "[Scheduler] Released {Count} stale claim(s) held by the previous process for {Name}", - released, run.Name); - } - } - catch (Exception ex) - { - // Not fatal: the leases still lapse on their own, just slowly. - _logger.LogWarning(ex, "[Scheduler] Could not release stale claims for {Name}", run.Name); - } - - _logger.LogDebug("[Scheduler] Resuming interrupted run {Name}: {Pending} pending", run.Name, pending); - resumed++; pendingTotal += pending; - await DispatchPendingTasksAsync(run, taskPath, run.Priority, ct, quiet: true); - } - else - { - await FinalizeRunAsync(run); - finalizedNow++; - } - } - catch (Exception ex) - { - _logger.LogError(ex, "[Scheduler] Failed to process interrupted run: {Name}", runName); - } - finally - { - // This run is no longer awaiting recovery, however it went. If it was resumed it is now - // in _activeRuns and still blocks its parent; if it finalized or could not be resumed, - // it must stop blocking. Clearing here covers every path including the failure one. - _recoveringChildren.TryRemove(runName, out _); - } - } - - if (resumed + finalizedNow + postExecResumed + unresumable + postExecGaveUp > 0) - _logger.LogInformation( - "[Scheduler] Crash recovery: resumed {Resumed} run(s) ({Pending} pending tasks re-dispatched), " + - "{PostExec} post-execution(s), {Finalized} finalized, {Stale} stale claim(s) released" + - "{Unresumable}{GaveUp}", - resumed, pendingTotal, postExecResumed, finalizedNow, staleReleased, - unresumable > 0 ? $", {unresumable} unresumable" : "", - postExecGaveUp > 0 ? $", {postExecGaveUp} post-exec abandoned" : ""); - - // First retention pass, now that every run that could be resumed is back in _activeRuns and so - // exempt from the abandoned-run rule. The scheduler keeps it going on an interval from here. try { - await RunRetentionSweepAsync(ct); - } - catch (Exception ex) - { - _logger.LogWarning(ex, "[Scheduler] Startup retention sweep failed"); - } - } - - /// - /// One retention pass over the orchestrator tables: finished runs past - /// Orchestrator:RetentionHours, runs nobody is driving that have not been written to for that - /// long, and Tasks/Results partitions whose Run row is already gone. Runs at the end of startup - /// recovery and then every Orchestrator:CleanupIntervalHours via . - /// - public async Task RunRetentionSweepAsync(CancellationToken ct) - { - var retention = TimeSpan.FromHours(Math.Max(1, _settings.Orchestrator.RetentionHours)); - var active = _activeRuns.Keys.ToHashSet(StringComparer.Ordinal); - var result = await _store.CleanupOldRunsAsync(retention, active, ct); - - // An abandoned run can still have rows in the durable queue. The pump would drop each as a - // stale descriptor when it came to claim it — but only after paying for the claim. - foreach (var name in result.AbandonedRuns) - { - try - { - await _queue.RemoveRunAsync(name, ct: ct); - } - catch (Exception ex) + name = TableKeys.Sanitize(name); + if (!_activePlanners.TryAdd(name, true) || await IsActiveAsync(name, ct)) { - _logger.LogWarning(ex, "[Scheduler] Could not remove queue rows for abandoned run {Name}", name); + _activePlanners.TryRemove(name, out _); + _logger.LogInformation("[Orchestrator] Run {Name} already active, skipping", name); + return false; } - } - - return result; - } - /// - /// Periodic retention sweeps for the life of the host, started by the scheduler once recovery has - /// run. A sweep that fails is logged and tried again next interval; CleanupIntervalHours of 0 - /// leaves only the startup pass. - /// - public async Task RunRetentionLoopAsync(CancellationToken ct) - { - var hours = _settings.Orchestrator.CleanupIntervalHours; - if (hours <= 0) - { - _logger.LogInformation( - "[Scheduler] Periodic retention sweep disabled (CleanupIntervalHours={Hours}); only the startup pass runs", hours); - return; - } - - var interval = TimeSpan.FromHours(hours); - using var timer = new PeriodicTimer(interval); - try - { - while (await timer.WaitForNextTickAsync(ct)) + try { - try + var tasks = !string.IsNullOrEmpty(batchFilePath) + ? ParseTasksFromJsonLinesFile(batchFilePath, name) + : ParseTasksFromJson(batchJson, name); + if (tasks.Count == 0) { - await RunRetentionSweepAsync(ct); + _logger.LogWarning("[Orchestrator] Batch for {Name} produced 0 tasks", name); + return false; } - catch (OperationCanceledException) when (ct.IsCancellationRequested) - { - return; - } - catch (Exception ex) + + var genericTaskFunc = _settings.Orchestrator.GenericTaskFunction; + if (string.IsNullOrEmpty(genericTaskFunc) || Script(genericTaskFunc) == null) { - _logger.LogWarning(ex, "[Scheduler] Retention sweep failed; next attempt in {Interval}", interval); + _logger.LogError("[Orchestrator] Cannot start {Name}: task function {Func} not found", name, genericTaskFunc); + return false; } - } - } - catch (OperationCanceledException) - { - // Host shutdown. - } - } - /// - /// One loop that ticks every live run's status / re-drive / completion-recheck, at - /// . It replaces the per-run that used to do this: at - /// high live-run counts that meant one Timer object per run and a continuous stream of fire-and-forget - /// callbacks onto the thread pool (~M/interval per second), whereas one sweep over - /// is a single scheduling source that allocates nothing per run. The per-run work stays cheap — - /// skips an unchanged line, is gated by - /// its backoff and returns synchronously when backed off, and - /// short-circuits — so ticking thousands of runs in one pass is fast. A throw for one run is logged and - /// neither stops the sweep nor takes the host down (a Timer callback that threw would have crashed it). - /// Started once by , alongside the retention loop. - /// - public async Task RunStatusSweepLoopAsync(CancellationToken ct) - { - using var timer = new PeriodicTimer(_statusInterval); - try - { - while (await timer.WaitForNextTickAsync(ct)) + if (string.IsNullOrEmpty(postExecFunctionName)) postExecFunctionName = null; + if (string.IsNullOrEmpty(postExecParametersJson)) postExecParametersJson = null; + await CreateAsync(name, tasks, priority, genericTaskFunc, postExecFunctionName, postExecParametersJson, + reference, parentRunKey, childKey, sequential, ct); + return true; + } + finally { - foreach (var run in _activeRuns.Values) - { - try - { - LogRunStatus(run); - RedrivePendingTasks(run); - lock (_lock) { CheckRunCompletion(run); } - } - catch (Exception ex) - { - _logger?.LogWarning(ex, "[Scheduler] Run status tick failed for {Name}", run?.Name); - } - } + _activePlanners.TryRemove(name, out _); } } - catch (OperationCanceledException) - { - // Host shutdown. - } - } - - /// - /// Start an orchestrator run from a pre-built batch. - /// Called by OrchestratorBridge.DrainPending() when PowerShell's Start-CIPPOrchestrator - /// queues a run on CIPPNG (bypassing the planner script phase). - /// - /// The batch arrives one of two ways. , when set, is a JSON Lines - /// file with one task object per line and is the preferred form: the caller writes it a task at a - /// time and this reads it a line at a time, so no single string ever holds the whole batch. - /// is the original whole-array-in-one-string form, kept for callers - /// that still use it — it costs the full batch as a string here AND again as a JsonDocument. - /// A file path wins when both are given; the file is deleted once parsed. - /// - public async Task StartFromBatchAsync(string name, string batchJson, int priority, - string? postExecFunctionName, string? postExecParametersJson, CancellationToken ct, - string? parentRunName = null, string? reference = null, string? batchFilePath = null, - bool sequential = false) - { - // The batch file is this method's to dispose of, on EVERY path — including the two - // "already running, skipping" returns below, which never look at it. Those are the common - // case for a duplicate enqueue, so leaving cleanup at the parse site would quietly fill the - // container's temp directory with the batches of runs that were skipped rather than started. - try - { - await StartFromBatchCoreAsync(name, batchJson, priority, postExecFunctionName, - postExecParametersJson, parentRunName, reference, batchFilePath, sequential, ct); - } finally { if (!string.IsNullOrEmpty(batchFilePath)) { try { if (File.Exists(batchFilePath)) File.Delete(batchFilePath); } - catch (Exception ex) - { - _logger.LogDebug(ex, "[Orchestrator] Failed to delete batch file {Path}", batchFilePath); - } + catch (Exception ex) { _logger.LogDebug(ex, "[Orchestrator] Failed to delete batch file {Path}", batchFilePath); } } } } - private async Task StartFromBatchCoreAsync(string name, string batchJson, int priority, - string? postExecFunctionName, string? postExecParametersJson, - string? parentRunName, string? reference, string? batchFilePath, bool sequential, CancellationToken ct) - { - // Run names become PartitionKeys verbatim, and batch names carry user-typed task names - // ("Alert on Entra ID P1/P2 …"). An illegal key character 400s every write for the run — - // run row, task rows, counter, queue rows — identically forever, so the run can neither - // start nor be re-driven. Sanitize at the boundary, like task ids at mint. - name = TableKeys.Sanitize(name); - if (!string.IsNullOrEmpty(parentRunName)) - parentRunName = TableKeys.Sanitize(parentRunName); - - // A run cannot be its own parent. The ambient RunName rides along when a run is re-queued - // from inside its own context; persisting it would feed the reattach loop a self-link on the - // next start and show circular lineage in the status APIs. - if (parentRunName == name) - parentRunName = null; - - if (!_activePlanners.TryAdd(name, true)) - { - _logger.LogInformation("[Orchestrator] Run {Name} already in progress, skipping", name); - return; - } - - try - { - await _store.InitializeAsync(); - await _queue.InitializeAsync(ct); + private async Task IsActiveAsync(string name, CancellationToken ct) => + await _store.GetRunByNameAsync(name, ct) is { IsFinished: false }; - var existing = await _store.GetRunAsync(name); - if (existing != null && existing.Status == "Running" && _activeRuns.ContainsKey(name)) - { - _logger.LogInformation("[Orchestrator] Run {Name} already active, skipping", name); - return; - } - - List tasks; - if (!string.IsNullOrEmpty(batchFilePath)) - { - var batchBytes = File.Exists(batchFilePath) ? new FileInfo(batchFilePath).Length : 0; - _logger.LogDebug( - "[Orchestrator] Batch for {Name} streamed from {Path} ({KB:F1} KB on disk, never held whole)", - name, batchFilePath, batchBytes / 1024.0); - - tasks = ParseTasksFromJsonLinesFile(batchFilePath, name); - } - else - { - // Sizing the legacy inbound batch: the whole run's task list as ONE string, built by - // ConvertTo-Json in the calling PowerShell, held in the bridge queue, and parsed here - // into a JsonDocument that holds it a second time. - _logger.LogDebug( - "[Orchestrator] Batch for {Name} is {Chars} chars (~{ApproxKB:F1} KB in memory)", - name, batchJson.Length, batchJson.Length * 2 / 1024.0); - - tasks = ParseTasksFromJson(batchJson, name); - } - - if (tasks.Count == 0) - { - _logger.LogWarning("[Orchestrator] Batch for {Name} produced 0 tasks", name); - return; - } - - // Stamp payload order. Sequential dispatch reads this to run tasks one at a time in the order - // they were submitted; fan-out ignores it. The parser preserves batch order, so the index is it. - for (var i = 0; i < tasks.Count; i++) tasks[i].Sequence = i; - - var genericTaskFunc = _settings.Orchestrator.GenericTaskFunction; - if (string.IsNullOrEmpty(genericTaskFunc)) - { - _logger.LogError("[Orchestrator] Cannot start {Name}: App:Orchestrator:GenericTaskFunction not configured", name); - return; - } - var taskPath = _psRunner.FindScript(genericTaskFunc); - if (taskPath == null) - { - _logger.LogError("[Orchestrator] Cannot start {Name}: {Func} not found", name, genericTaskFunc); - return; - } - - var run = new OrchestratorRun - { - Name = name, - Reference = reference, - Status = "Running", - Priority = priority, - StartedUtc = DateTime.UtcNow, - Tasks = tasks, - TaskScriptName = genericTaskFunc, - PostExecFunctionName = postExecFunctionName, - PostExecParametersJson = postExecParametersJson, - ParentRunName = parentRunName, - Sequential = sequential - }; - - await _store.UpsertRunAsync(run); - await _store.UpsertTaskBatchAsync(name, tasks); - // Seed the durable outstanding-task count alongside the tasks themselves, so completion - // is answerable from storage instead of by walking this run graph. - await _store.InitRemainingAsync(name, tasks.Count, ct); - // IsNullOrEmpty, not != null: PowerShell marshals a $null argument to an empty string, so a - // run with no post-execution arrives here with "" and a null check logs the meaningless - // "(PostExec: Push-)". Everything that acts on this field already uses IsNullOrEmpty — - // only the log disagreed. - var postExecSuffix = !string.IsNullOrEmpty(postExecFunctionName) - ? $" (PostExec: Push-{postExecFunctionName})" - : ""; - _logger.LogInformation( - "[Orchestrator] Run {Name} created from batch: {Count} tasks P{Priority}{PostExec}", - name, tasks.Count, priority, postExecSuffix); - await DispatchPendingTasksAsync(run, taskPath, priority, ct); - } - finally - { - _activePlanners.TryRemove(name, out _); - } - } - - private async Task DispatchPendingTasksAsync(OrchestratorRun run, string taskPath, int priority, - CancellationToken ct, bool quiet = false) - { - _activeRuns.TryAdd(run.Name, run); - // Registered before anything is enqueued — the resolver reads it on the dispatch side. - _taskScriptPaths[run.Name] = taskPath; - - // This run is live again, so it is allowed to finalize again. Matters for the stable run names - // (CIPPDBCacheOrchestrator, ProcessDeltaQueries) that recur within one process lifetime — without - // this, the finalize claim from their previous outing would strand the next one forever. - _finalizingRuns.TryRemove(run.Name, out _); - - // No per-run timer: the single RunStatusSweepLoopAsync ticks every run in _activeRuns (which this - // run was just added to). One sweep replaces what used to be a System.Threading.Timer per run — at - // high live-run counts that was thousands of Timer objects and a continuous drizzle of fire-and-forget - // callbacks onto the thread pool. The sweep still runs LogRunStatus + RedrivePendingTasks + - // CheckRunCompletion for each run at the same cadence (the last re-run on purpose: a finalize deferred - // while storage showed work outstanding has no task transition left to retrigger it). - - var pending = run.Tasks.Where(t => t.Status == "Pending").ToList(); - - // Skip anything that already has a queue row. Re-writing it would reset a live claim (Owner and - // LeaseUntil) on a task another worker may be running, which is how a task ran twice. One index - // partition read per dispatch. - HashSet alreadyQueued; - try - { - alreadyQueued = await _queue.GetQueuedTaskIdsAsync(run.Name, ct); - } - catch (Exception ex) - { - // Enqueuing a duplicate is recoverable; enqueuing nothing strands the run. Prefer the former. - _logger.LogWarning(ex, - "[Scheduler] Could not read existing queue rows for {Run} — dispatching without de-duplication", - run.Name); - alreadyQueued = []; - } - - List toQueue; - if (run.Sequential) - { - // Sequential mode: a single entry row starts the ONE pinned driver that then runs every step of - // the run inline (see BuildSequentialRunWork), so exactly ONE queue row ever exists for the run — - // the lowest-Sequence task still Pending — and the later steps get no rows at all. Enqueue that - // one task, and only if it does not already have a row. A second row for the same run would let a - // second driver start and break the pinning/ordering. This covers both first dispatch (enqueues - // Sequence 0 to start the driver) and resume (re-enqueues the current step only if its row is - // gone, so a fresh driver picks the run back up where it left off). - var current = pending.OrderBy(t => t.Sequence).FirstOrDefault(); - toQueue = current != null && !alreadyQueued.Contains(current.Id) ? [current] : []; - } - else - { - toQueue = pending.Where(t => !alreadyQueued.Contains(t.Id)).ToList(); - } - - // Shed the Parameters payload BEFORE the tasks become claimable. The caller has already persisted - // them (UpsertTaskBatchAsync), so the Tasks table is authoritative; the live graph keeps each task - // object (its identity + Status drive completion tracking) but drops the payload it does not need - // while it waits, and BuildTaskWork rehydrates it from storage at dispatch. This is what bounds the - // retained memory of a large pending backlog — thousands of runs each holding every task's payload is - // what walks the live-set into the GC heap ceiling. Ordering matters: shedding AFTER the enqueue - // could null a task the pump had already claimed and whose BuildTaskWork had just rehydrated it, so - // shed here, before EnqueueBatchAsync makes the rows claimable. Only toQueue (the rows enqueued in - // THIS call) is touched — a task already queued from a prior call may be mid-dispatch. Sequential - // runs are exempt: they are small (a handful of ordered steps) and their pinned driver holds every - // step's payload in the live graph while it runs, so shedding would buy no memory and only add reads. - if (_shedParameters && !run.Sequential) - { - lock (_lock) - foreach (var t in toQueue) - if (t.Status == "Pending") t.Parameters = null!; - } - - // One batched write per priority bucket rather than one per task. The queue is the backlog now; - // the JobManager only ever sees the batch JobQueuePump claims from it. - await _queue.EnqueueBatchAsync(run.Name, - toQueue.Select(t => (t.Id, t.Priority ?? priority)).ToList(), - run.StartedUtc, ct); - - // quiet = called from crash recovery, where a per-run line per resumed run is the flood the - // aggregate summary replaces — drop to Debug. A normal orchestration start logs it at Info (one line). - var level = quiet ? LogLevel.Debug : LogLevel.Information; - if (toQueue.Count == pending.Count) - { - _logger.Log(level, "[Scheduler] Dispatched {Count} tasks for {Name} at P{Priority}", - toQueue.Count, run.Name, priority); - } - else - { - _logger.Log(level, - "[Scheduler] Dispatched {Count} tasks for {Name} at P{Priority} ({Existing} already queued)", - toQueue.Count, run.Name, priority, pending.Count - toQueue.Count); - } - } - - /// - /// Build the work for a SEQUENTIAL run: a single delegate that checks out ONE background worker and - /// runs every step of the run on it, in payload (Sequence) order, one at a time, then reclaims the - /// worker once at the end. This is what "pin one worker for the whole run" means — the run starts on a - /// worker and stays there until it finishes, never going back to the pool between steps to be - /// re-scheduled onto a different one. The run is dispatched as a SINGLE queue row (the entry task; see - /// DispatchPendingTasksAsync), so exactly one JobManager slot and one worker are held for the run's - /// whole duration and the mid-run steps never get their own rows. - /// - /// Failure policy is best-effort: a step that throws is recorded Failed and the driver moves on to the - /// next, so one bad step cannot strand the rest and the run ends CompletedWithErrors. Each step runs - /// through InvokeAsync, whose finally resets the runspace after success, failure OR cancellation, so - /// steps stay isolated on the shared worker. Registered in for - /// its whole life so the re-drive leaves the not-yet-reached (deliberately row-less) steps alone while - /// the driver is progressing. - /// - private Func BuildSequentialRunWork(OrchestratorRun run, string taskPath) - { - return async (jobCt) => - { - // One driver per run. Registered for the driver's whole life so the re-drive does not mistake - // the run's row-less pending steps for orphans and enqueue them. TryAdd (not an unconditional - // set) also closes the duplicate-entry-row race: if two rows for the same run are claimed before - // either driver marks the entry step Running, the loser here does nothing and the winner runs - // every step. Because this guard is OUTSIDE the try/finally, the loser never runs the finally - // that would otherwise remove the winner's registration. - if (!_activeSequentialDrivers.TryAdd(run.Name, true)) - { - _logger.LogDebug( - "[Scheduler] Sequential run {Run} already has an active driver — skipping duplicate entry", run.Name); - return; - } - PowerShellWorker? worker = null; - var workerFaulted = false; - try - { - // One worker for the whole run: checked out once here, reclaimed once in the finally. - worker = CheckoutSequentialWorker(jobCt); - - while (true) - { - OrchestratorTaskItem? task; - lock (_lock) - { - task = run.Tasks.Where(t => t.Status == "Pending") - .OrderBy(t => t.Sequence) - .FirstOrDefault(); - } - if (task == null) break; // every step is terminal — the run is done - - // Run cancelled while we were working through it: mark this and every remaining step - // Cancelled and stop. - if (_cancelledRuns.ContainsKey(run.Name)) - { - CancelRemainingSequentialTasks(run); - break; - } - - // Rehydrate a shed payload. Sequential runs are exempt from shedding (see - // DispatchPendingTasksAsync), so this is normally a no-op — kept for a resumed run whose - // in-memory payload was dropped. A missing row means the durable state was lost: fail - // that step closed rather than run it blank, then carry on with the next. - if (_shedParameters && task.Parameters == null) - { - var rehydrated = await _store.GetTaskParametersAsync(run.Name, task.Id, jobCt); - if (rehydrated == null) - { - FailTaskTerminally(run, task, - "Parameters could not be rehydrated at dispatch — the Tasks-table row is missing. " + - "The task's payload was shed from memory and storage no longer has it."); - continue; - } - lock (_lock) { task.Parameters ??= rehydrated; } - } - - lock (_lock) { task.Status = "Running"; task.OwnedHere = true; } - // Durable "Running" marker — awaited before the invoke, same as the parallel path. - try - { - await _writer.MarkRunningAsync(run.Name, task, jobCt); - } - catch (MarkerNotPersistedException ex) - { - // The marker never landed, so storage still has this step Pending. Unlike the - // parallel path we cannot just give a slot back and let a re-drive retry the one - // task — we own the worker and the whole run — and running later steps out of order - // is not allowed. Put the step back to Pending and STOP the driver. The entry job - // then completes with pending work left and no driver active, so the re-drive - // re-enqueues the current step and a fresh driver resumes the run here. - lock (_lock) { task.Status = "Pending"; } - _logger.LogWarning(ex, - "[Scheduler] Sequential run {Run}: could not persist the Running marker for {Task} — " + - "leaving the run for the re-drive to resume", run.Name, task.Id); - break; - } - - try - { - // Run the step on the pinned worker. Only a PostExecution run needs the output - // captured and stored; otherwise the seam returns empty and nothing is stored. - var output = await RunSequentialStepAsync(run, task, taskPath, worker); - if (!string.IsNullOrEmpty(run.PostExecFunctionName) - && !string.IsNullOrEmpty(output) - && !_writer.TryQueueResult(run.Name, task.Id, output)) - { - await _store.StoreResultAsync(run.Name, task.Id, output); - } - - lock (_lock) - { - task.Status = "Completed"; - task.CompletedUtc = DateTime.UtcNow; - task.Parameters = null!; - CheckRunCompletion(run); - } - PersistTaskAndRunAsync(run, task); - _logger.LogDebug("[Scheduler] Sequential task completed: {TaskId}", task.Id); - } - catch (OperationCanceledException) when (jobCt.IsCancellationRequested) - { - // App shutting down mid-step. Leave this step Running (its durable marker is written) - // for resume on next startup, and let the exception abort the driver. - _logger.LogInformation( - "[Scheduler] Sequential run {Run} interrupted by shutdown at {TaskId}", run.Name, task.Id); - throw; - } - catch (Exception ex) - { - lock (_lock) - { - task.Status = "Failed"; - task.LastError = ex.Message; - task.CompletedUtc = DateTime.UtcNow; - task.Parameters = null!; - CheckRunCompletion(run); - } - PersistTaskAndRunAsync(run, task); - _logger.LogError(ex, - "[Scheduler] Sequential task failed: {TaskId} — continuing with the next step", task.Id); - // best-effort: fall through to the next Pending step - } - } - } - catch (OperationCanceledException) when (jobCt.IsCancellationRequested) - { - // Shutdown (checkout or a step was cancelled). The pipeline may have been Stop()'d, so treat - // the worker as faulted on reclaim; rethrow so the JobManager marks the entry job Cancelled. - workerFaulted = true; - throw; - } - catch (Exception ex) - { - // An unexpected driver-level failure — not a per-step task error, which is handled inline. - workerFaulted = true; - _logger.LogError(ex, "[Scheduler] Sequential driver for {Run} failed", run.Name); - throw; - } - finally - { - ReclaimSequentialWorker(worker, workerFaulted); - _activeSequentialDrivers.TryRemove(run.Name, out _); - } - }; - } - - // ── sequential-driver seams ─────────────────────────────────────────────────────────────────────── - // The driver's loop logic (ordering, best-effort, cancellation, marker recovery) is exercised by unit - // tests through a subclass that overrides these three methods, so the tests need no PowerShell worker - // pool. Production runs the real pool: one worker checked out for the whole run, each step invoked on - // it, reclaimed once at the end. Keep the checkout/run/reclaim split — the whole point is one checkout - // and one reclaim around many step invocations. - - /// Check out the single worker a sequential run is pinned to. Virtual for tests. - internal virtual PowerShellWorker? CheckoutSequentialWorker(CancellationToken ct) - => _psRunner.CheckoutBackgroundWorker(ct); - - /// Return the pinned worker once the run is done. No-op for a null worker (checkout failed). - /// Virtual for tests. - internal virtual void ReclaimSequentialWorker(PowerShellWorker? worker, bool faulted) - { - if (worker != null) _psRunner.ReclaimBackgroundWorker(worker, faulted: faulted); - } - - /// Run one sequential step on the pinned worker and return its captured output (empty when the - /// run has no PostExecution and so needs no result). Virtual for tests. InvokeAsync resets the runspace - /// afterwards, so successive steps stay isolated on the shared worker. - internal virtual async Task RunSequentialStepAsync( - OrchestratorRun run, OrchestratorTaskItem task, string taskPath, PowerShellWorker? worker) + private async Task CreateAsync(string name, List tasks, int priority, string taskScriptName, + string? postExecFunctionName, string? postExecParametersJson, string? reference, string? parentRunKey, + string? childKey, bool sequential, CancellationToken ct) { - var parameters = new Dictionary + var started = DateTime.UtcNow; + var header = new RunHeader { - { "TaskJson", JsonSerializer.Serialize(task.Parameters, s_jsonOptions) } + RunKey = WorkStore.RunKeyFor(name, started), + Name = name, + Priority = Math.Clamp(priority, 0, 99), + StartedUtc = started, + TaskScriptName = taskScriptName, + PostExecFunctionName = postExecFunctionName, + PostExecParametersJson = postExecParametersJson, + Reference = reference, + ParentRunKey = parentRunKey, + ParentChildKey = childKey, + Sequential = sequential, }; - if (!string.IsNullOrEmpty(run.PostExecFunctionName)) - return await _psRunner.ExecuteScriptWithOutput(taskPath, parameters, pinnedWorker: worker); - await _psRunner.ExecuteScript(taskPath, parameters, pinnedWorker: worker); - return string.Empty; + await _store.CreateRunAsync(header, tasks.Select(t => new WorkStore.NewTask(t.Id, t.Parameters)).ToList(), ct); + _logger.LogInformation("[Orchestrator] Run {Name} created with {Count} tasks at P{Priority}{PostExec}{Sequential}", + name, tasks.Count, header.Priority, + postExecFunctionName != null ? $" (PostExec: Push-{postExecFunctionName})" : "", + sequential ? " (sequential)" : ""); } - /// - /// Mark every still-Pending or Running step of a cancelled sequential run Cancelled, in one lock, and - /// persist them. Called by the driver when it notices the run was cancelled between steps. - /// - private void CancelRemainingSequentialTasks(OrchestratorRun run) - { - List remaining; - lock (_lock) - { - remaining = run.Tasks.Where(t => t.Status is "Pending" or "Running").ToList(); - foreach (var t in remaining) - { - t.Status = "Cancelled"; - t.LastError = "Cancelled by user"; - t.CompletedUtc = DateTime.UtcNow; - t.Parameters = null!; - } - CheckRunCompletion(run); - } - foreach (var t in remaining) _writer.QueueTask(run.Name, t); - _writer.QueueRun(run); - } + // ── child runs ── /// - /// Put tasks of one run back on the durable queue, in one batch. Fire-and-forget because every caller - /// is on a lock or a timer callback, and a failure is recoverable: the tasks are still Pending in - /// storage, so the next re-drive finds them again. + /// Make wait for a child that is about to be queued. Called at enqueue time, + /// inside the parent's task, so the parent cannot finish first. Returns the parent's run key and the + /// placeholder key the child completes, or null when the parent is not running tasks (a run queued from an + /// aggregation is not a child) or is the child itself. /// - private void RequeueToTable(OrchestratorRun run, IReadOnlyList tasks) + internal (string ParentRunKey, string ChildKey)? RegisterPendingChild(string parentRunName, string childRunName) { - _ = Task.Run(async () => + if (parentRunName == childRunName) return null; + return Task.Run(async () => { - try - { - await _queue.EnqueueBatchAsync(run.Name, - tasks.Select(t => (t.Id, t.Priority ?? run.Priority)).ToList(), run.StartedUtc); - foreach (var task in tasks) _requeueFailures.TryRemove(DeferralKey(run.Name, task.Id), out _); - } - catch (Exception ex) - { - // A row storage rejects is rejected identically forever (illegal key remnant, - // oversized property), and the re-drive resets the deferral counter on every pass — - // without this cap the retry loop is infinite and the run it belongs to can never - // finalize. Consecutive failures only: a success above clears the count. - foreach (var task in tasks) - { - var key = DeferralKey(run.Name, task.Id); - var failures = _requeueFailures.AddOrUpdate(key, 1, (_, c) => c + 1); - if (failures < MaxRequeueFailures) continue; - _requeueFailures.TryRemove(key, out _); - FailTaskTerminally(run, task, - $"Could not re-queue after {failures} consecutive attempts: {ex.Message}"); - } - _logger.LogWarning(ex, - "[Scheduler] Could not re-queue {Count} task(s) in {Run} — the re-drive will retry", - tasks.Count, run.Name); - } - }); + var parent = await _store.GetRunByNameAsync(parentRunName); + if (parent is not { Phase: RunPhase.Tasks }) return ((string, string)?)null; + var childKey = $"{childRunName}|{Guid.NewGuid():N}"; + if (!await _store.AddChildAsync(parent.RunKey, childKey)) return null; + _logger.LogInformation("[Orchestrator] Registered child run {Child} under parent {Parent}", childRunName, parentRunName); + return (parent.RunKey, childKey); + }).GetAwaiter().GetResult(); } - /// - /// Move a task that can never run to Failed and let its run finish without it. The terminal write - /// flows through the status writer like any other completion, so the remaining counter decrements - /// and finalize proceeds — the alternative is a Pending task retried for the process lifetime, - /// pinning the whole run graph with it. - /// - private void FailTaskTerminally(OrchestratorRun run, OrchestratorTaskItem task, string reason) + /// A registered child that was never created: stop the parent waiting for it. + internal async Task AbandonPendingChildAsync(string parentRunKey, string childKey) { - lock (_lock) - { - if (task.Status is "Completed" or "Failed" or "Cancelled") return; - task.Status = "Failed"; - task.LastError = reason; - task.CompletedUtc = DateTime.UtcNow; - task.Parameters = null!; - CheckRunCompletion(run); - } - PersistTaskAndRunAsync(run, task); - _logger.LogError("[Scheduler] Task {TaskId} in {Run} permanently failed: {Reason}", - task.Id, run.Name, reason); + await _store.FinishAsync(parentRunKey, [new WorkStore.Finish(0, "Completed", ChildKey: childKey)], null); } - /// - /// Rebuild the work for a queued task. Registered on the JobManager at startup. - /// - /// Resolution order matters, for correctness before cost: - /// 1. the live run in , if present — this is the steady-state path and - /// costs ZERO storage reads; - /// 2. table storage, if the run is not in memory (crash recovery, or a run finalized and evicted - /// while its tasks were still queued). - /// - /// Step 1 is not an optimization, it is a requirement. decides - /// finalization by inspecting run.Tasks, and the task work mutates task.Status in - /// place. Handing out a freshly-deserialized task object would mutate a copy the live run graph - /// never sees, and no run would ever finalize. Object identity — not the field values — is the one - /// piece of state here that is genuinely not rehydratable. - /// - private async Task?> ResolveTaskWorkAsync(JobDescriptor descriptor, CancellationToken ct) - { - OrchestratorRun? run; - OrchestratorTaskItem? task; - - if (_activeRuns.TryGetValue(descriptor.RunName, out var liveRun)) - { - run = liveRun; - lock (_lock) - { - task = liveRun.Tasks.FirstOrDefault(t => t.Id == descriptor.TaskId); - } - } - else - { - // Run is no longer in memory — rehydrate it. This is the same read the crash-recovery path - // already performs (ResumeInterruptedRunsAsync), which resumed 421 runs in seconds. - run = await _store.GetRunAsync(descriptor.RunName); - - // A run absent from _activeRuns because it FINISHED must not be resurrected. Rehydrating it - // put a finalized run back into the live graph, where the next completion re-finalized it - // and re-dispatched its post-execution; and because terminal task writes are coalesced, the - // rehydrated task could still read Pending and be run a second time. Storage is authoritative - // here — FinalizeRunAsync flushes the run's terminal status before it removes the run from - // _activeRuns, so a terminal Status seen here is not a race. - if (run != null && run.Status is "Completed" or "CompletedWithErrors") - { - _logger.LogDebug( - "[Orchestrator] Descriptor {Run}/{Task} belongs to a run that already finished ({Status}) — dropping", - descriptor.RunName, descriptor.TaskId, run.Status); - return null; - } - - task = run?.Tasks.FirstOrDefault(t => t.Id == descriptor.TaskId); - if (run != null && task != null) - { - // Re-establish the live graph so sibling tasks share this instance and completion - // tracking works, exactly as it does on the resume path. - run = _activeRuns.GetOrAdd(run.Name, run); - task = run.Tasks.FirstOrDefault(t => t.Id == descriptor.TaskId) ?? task; - } - } - - if (run == null || task == null) - { - _logger.LogDebug("[Orchestrator] Stale descriptor {Run}/{Task} — no longer in storage", - descriptor.RunName, descriptor.TaskId); - return null; - } - - // Resolve the task script (a PowerShell function name) for this run. The steady-state path finds - // it in _taskScriptPaths, cached by DispatchPendingTasksAsync when the run was dispatched. A MISS - // is NOT a reason to drop the task. The pump is a BackgroundService that begins claiming persisted - // queue rows at host start, whereas the only writer of _taskScriptPaths does not run for a resumed - // run until ResumeInterruptedRunsAsync reaches it — and that waits on the worker pool first — and - // the rehydration branch above re-adds a run to _activeRuns without a cached path either. In both - // windows the run record still carries TaskScriptName, so rebuild the path from it exactly as the - // resume path does (ScriptRepository is fully loaded before the pump's first claim) and cache it, - // so sibling tasks of the same run cost nothing. Only an empty TaskScriptName with no - // naming-convention match is genuinely unrunnable. Returning null on the miss instead lets the - // JobManager mark the job Skipped and the pump delete its queue row, permanently dropping a task - // that is still Pending in the run — with no queue row left, nothing re-dispatches it. - if (!_taskScriptPaths.TryGetValue(run.Name, out var taskPath)) - { - taskPath = !string.IsNullOrEmpty(run.TaskScriptName) - ? _psRunner.FindScript(run.TaskScriptName) - : FindTaskScript(run.Name); - - if (string.IsNullOrEmpty(taskPath)) - { - _logger.LogWarning( - "[Orchestrator] No task script for run {Run} (TaskScriptName={Script}) — dropping {Task}", - run.Name, run.TaskScriptName, descriptor.TaskId); - return null; - } - - _taskScriptPaths[run.Name] = taskPath; - } - - // Already terminal (e.g. cancelled, or completed by a previous attempt while queued), or already - // executing. - // - // "Running" belongs here as well as the terminal states. A duplicate queue row claimed while its - // task is mid-flight used to pass this check and start a SECOND copy — seen with a five-minute - // Intune collection that was re-claimed four minutes in and ran twice. This does not block crash - // recovery: ResumeInterruptedRunsAsync flips interrupted tasks from Running back to Pending - // before re-dispatching them, so a task that genuinely needs re-running never reaches here as - // Running. - // - // Running only means "a worker HERE has it" when this process wrote it (OwnedHere). A Running read - // from storage is another process's pre-invoke marker, and it reaches the live graph whenever a run - // is rehydrated — by this resolver, or by the pump winning the _activeRuns race with recovery at - // startup, or from a container that outlived this one's recovery. Dropping those left the task - // Running forever: nothing re-drives Running, so the run never finalized. Reaching here means this - // process holds the task's queue claim (only the pump enqueues descriptors, only for rows it - // claimed), so the previous owner is gone: count the interrupted attempt exactly as recovery does, - // and run it. - var poisoned = false; - lock (_lock) - { - if (task.Status is "Completed" or "Failed" or "Cancelled") - return null; - if (task.Status == "Running") - { - if (task.OwnedHere) return null; - poisoned = ++task.AttemptCount >= 3; - if (!poisoned) task.Status = "Pending"; - } - } - if (poisoned) - { - FailTaskTerminally(run, task, $"Cancelled {task.AttemptCount} times by host interruption"); - return null; - } + // ── what happens when tasks finish ── - // Sequential runs are dispatched as a single entry row; that one claim drives the WHOLE run on one - // pinned worker (BuildSequentialRunWork ignores which step this descriptor named and works through - // every Pending step in order). The "Running" guard above already stops a duplicate row from - // starting a second driver once the entry step is marked Running, and the driver's own TryAdd closes - // the remaining pre-mark race. - if (run.Sequential) - return BuildSequentialRunWork(run, taskPath); - - return BuildTaskWork(run, task, taskPath); - } - - private Func BuildTaskWork(OrchestratorRun run, OrchestratorTaskItem task, string taskPath) + private async Task AfterFinishAsync(WorkStore.FinishOutcome outcome) { - return - async (jobCt) => - { - // Rehydrate the Parameters payload shed while this task waited in the backlog. Done BEFORE - // any status write — MarkRunningAsync snapshots Parameters and every task-row write is - // Replace, so a null payload here would overwrite the stored one. One point read, only for a - // task actually being dispatched. The ??= keeps a value another dispatch attempt already set. - if (_shedParameters && task.Parameters == null) - { - // GetTaskParametersAsync returns null only when the Tasks row itself is GONE (a - // present-but-empty payload comes back as an empty dictionary). A missing row means this - // task's durable state was lost out from under a live run — the shed dropped the in-memory - // copy on the promise that storage still had it. Running now would invoke the task with NO - // parameters (its FunctionName and inputs both live in the payload), which for a real task - // is worse than not running it. Fail closed instead of executing blank. Before shedding - // the payload was resident, so a deleted row could not affect an in-flight dispatch; this - // guard restores that safety for the one case shedding introduced. - var rehydrated = await _store.GetTaskParametersAsync(run.Name, task.Id, jobCt); - if (rehydrated == null) - { - FailTaskTerminally(run, task, - "Parameters could not be rehydrated at dispatch — the Tasks-table row is missing. " + - "The task's payload was shed from memory and storage no longer has it."); - return; - } - lock (_lock) { task.Parameters ??= rehydrated; } - } - - // Check if run was cancelled while this job was queued - if (_cancelledRuns.ContainsKey(run.Name)) - { - lock (_lock) - { - if (task.Status != "Cancelled") - { - task.Status = "Cancelled"; - task.LastError = "Cancelled by user"; - task.CompletedUtc = DateTime.UtcNow; - task.Parameters = null!; - CheckRunCompletion(run); - } - } - PersistTaskAndRunAsync(run, task); - return; - } - - lock (_lock) - { - task.Status = "Running"; - task.OwnedHere = true; - } - // Pre-script "Running" write is awaited — the durability marker for crash recovery. Batched - // across concurrently-starting tasks by the status writer, but still durable before the invoke. - try - { - await _writer.MarkRunningAsync(run.Name, task, jobCt); - } - catch (MarkerNotPersistedException ex) - { - // DEFERRAL, not failure. The marker never landed, so storage still has this task - // Pending — running it now would break the poison-task bound that the marker exists - // to provide. Put the in-memory copy back to Pending so it agrees with storage, give - // the slot up, and retry. Nothing is lost: even if this process dies first, recovery - // re-queues it from the Pending row. - lock (_lock) { task.Status = "Pending"; } - DeferTask(run, task, ex); - return; - } - _deferrals.TryRemove(DeferralKey(run.Name, task.Id), out _); - - try - { - var parameters = new Dictionary - { - { "TaskJson", JsonSerializer.Serialize(task.Parameters, s_jsonOptions) } - }; - - if (!string.IsNullOrEmpty(run.PostExecFunctionName)) - { - // Capture output so C# can store results for PostExecution - var output = await _psRunner.ExecuteScriptWithOutput(taskPath, parameters); - - // Sizing the result payload. This string is the whole task result held in one - // piece, and BgPoolSize of them can be live at once — each at roughly two bytes - // per char since it is UTF-16. - // - // Measured on a 16-tenant instance across 92 real task results (mailbox and - // calendar permission batches, the widest fan-out CIPP has): median 5.9K chars, - // p95 42.5K, max 43.3K — 0.08 MB in memory for the largest. Eight of those - // concurrently is under 1 MB against a 2398 MB heap cap, so this site is not - // where the memory goes; the aggregate built from these at post-execution is - // (see DispatchPostExecution). Reported in KB because MB rounds every real - // result to 0.0 and hides exactly that conclusion. - _logger.LogDebug( - "[Scheduler] Task {TaskId} in {Run} returned {Chars} chars (~{ApproxKB:F1} KB in memory)", - task.Id, run.Name, output?.Length ?? 0, (output?.Length ?? 0) * 2 / 1024.0); - - if (!string.IsNullOrEmpty(output) && !_writer.TryQueueResult(run.Name, task.Id, output)) - { - // Large result, or result-batching off: keep the directly-awaited chunked path. - // Either way this awaits BEFORE the task is marked Completed below, so the result - // is durable before the run can finalize. Small results instead ride the status - // writer (TryQueueResult == true), which writes them before this task's terminal - // marker in the same flush — same guarantee, off the slot-held critical path. - await _store.StoreResultAsync(run.Name, task.Id, output); - } - } - else - { - await _psRunner.ExecuteScript(taskPath, parameters); - } - - lock (_lock) - { - task.Status = "Completed"; - task.CompletedUtc = DateTime.UtcNow; - // Release parameters — they are persisted in Table Storage and no longer - // needed in-memory. For 738-task runs this frees significant Gen2 memory. - task.Parameters = null!; - CheckRunCompletion(run); - } - // Post-script writes are fire-and-forget so the JobManager slot releases - // immediately and the dispatch loop can hand the worker to the next task. - // Crash recovery still works: the next startup re-reads task state from the - // table and re-runs anything not marked Completed (idempotent). - PersistTaskAndRunAsync(run, task); - - _logger.LogDebug("[Scheduler] Task completed: {TaskId}", task.Id); - } - catch (OperationCanceledException) when (jobCt.IsCancellationRequested) - { - // App shutting down — leave task as Running for resume on next startup - _logger.LogInformation("[Scheduler] Task {TaskId} interrupted by shutdown", task.Id); - throw; // Let JobManager mark as Cancelled - } - catch (Exception ex) - { - lock (_lock) - { - task.Status = "Failed"; - task.LastError = ex.Message; - task.CompletedUtc = DateTime.UtcNow; - // Release parameters on failure too — Table Storage has the full state - task.Parameters = null!; - CheckRunCompletion(run); - } - PersistTaskAndRunAsync(run, task); - - _logger.LogError(ex, "[Scheduler] Task failed: {TaskId}", task.Id); - throw; // Let JobManager also track the failure - } - }; - } - - /// - /// Fire-and-forget persistence of task + run state. Callers do not await this — it lets the - /// JobManager slot release immediately so the dispatch loop can hand the worker to the next - /// task. Errors are logged; on host crash, ResumeInterruptedRunsAsync re-derives state from - /// whatever made it to the table (writes are idempotent). - /// - private void PersistTaskAndRunAsync(OrchestratorRun run, OrchestratorTaskItem task) - { - // Non-blocking: the status writer coalesces these terminal task/run writes and flushes them in batches - // (guaranteed flushed before the run finalizes). Previously two individual fire-and-forget writes. - _writer.QueueTask(run.Name, task); - _writer.QueueRun(run); - } - - /// How many times a task has been deferred, and when it last was. The timestamp is what lets - /// the re-drive tell an exhausted task that has been sitting for minutes from one that deferred a - /// moment ago and is still legitimately retrying. - private sealed record DeferralState(int Count, DateTime LastUtc); - - /// Deferrals per task while storage is unable to accept the durable marker. In-memory and - /// intentionally so — it bounds retries within one process life, nothing more. - private readonly ConcurrentDictionary _deferrals = new(); - - /// Cap on in-process retries before a task is left for the next recovery pass to pick up. - private const int MaxDeferrals = 3; - - /// Consecutive finalize checks where storage still reported outstanding work for a run whose - /// in-memory tasks are all terminal. At the counter is recounted - /// from the task rows — a lost decrement otherwise defers finalize forever. - private readonly ConcurrentDictionary _finalizeDeferrals = new(); - private const int ReconcileAfterDeferrals = 3; - - /// Consecutive re-queue failures per task. Storage rejecting the same entity is not - /// transient — the write fails identically forever (see ) — so past - /// the task is failed terminally instead of re-driven again. - private readonly ConcurrentDictionary _requeueFailures = new(); - private const int MaxRequeueFailures = 5; - - /// - /// Re-queue a task whose durable marker could not be written, so it retries once storage recovers - /// instead of waiting for a restart. Bounded: after the task is simply - /// left Pending, which is already the durable state — recovery re-queues it on the next startup. - /// - private static string DeferralKey(string runName, string taskId) => $"{runName}{taskId}"; - - private void DeferTask(OrchestratorRun run, OrchestratorTaskItem task, Exception cause) - { - var state = _deferrals.AddOrUpdate(DeferralKey(run.Name, task.Id), - _ => new DeferralState(1, DateTime.UtcNow), - (_, s) => new DeferralState(s.Count + 1, DateTime.UtcNow)); - var count = state.Count; - - if (count > MaxDeferrals) - { - // One exhausted deferral cycle counts as one attempt on the task, mirroring startup - // recovery's 3-attempts rule. The re-drive resets the deferral counter when it re-queues, - // so without this the marker-fail → re-queue → marker-fail cycle repeats for the process - // lifetime and the run never finalizes. Exactly-once per cycle: only the call that - // crosses the cap increments (a duplicate queue row can push count past it again). - if (count == MaxDeferrals + 1 && ++task.AttemptCount >= 3) - { - FailTaskTerminally(run, task, - $"Durable Running marker rejected across {task.AttemptCount} deferral cycles: {cause.Message}"); - return; - } + var h = outcome.Header; + if (!outcome.Completed) return; - // Left Pending on purpose — storage already says Pending, so nothing is lost. It is no longer - // terminal though: RedrivePendingTasks picks it up once it has aged, so recovery is not - // gated on a restart the way it used to be. - _logger.LogError(cause, - "[Scheduler] Task {TaskId} in {Run} deferred {Count} times — left Pending for the re-drive", - task.Id, run.Name, count); - return; - } + _lastStatusLog.TryRemove(h.RunKey, out _); + var wall = (h.CompletedUtc ?? DateTime.UtcNow) - h.StartedUtc; + _logger.LogInformation("[Scheduler] Run {Name} finalized: {Status} ({Completed}/{Failed}/{Cancelled}/{Total}) wall={Wall} {Memory}", + h.Name, h.Status, h.Done - h.Failed - h.Cancelled, h.Failed, h.Cancelled, h.Total, + wall.TotalSeconds < 60 ? $"{wall.TotalSeconds:F1}s" : $"{wall.TotalMinutes:F1}min", + BackgroundTaskLimiter.GetMemorySnapshot()); - _logger.LogWarning( - "[Scheduler] Task {TaskId} in {Run} could not be marked Running (attempt {Count}/{Max}) — re-queued, slot released", - task.Id, run.Name, count, MaxDeferrals); + try { await _results.DeleteRunAsync(h.RunKey); } + catch (Exception ex) { _logger.LogDebug(ex, "[Orchestrator] Could not drop results of {Run}", h.Name); } - // Back to the QUEUE, not to memory. The pump drops a claimed row once the JobManager is done with - // the job, so an in-memory re-queue here would leave the retry with no durable row behind it — and - // nothing to pick it up again if this instance went away. - RequeueToTable(run, [task]); + if (h.ParentRunKey is { } parentKey && h.ParentChildKey is { } childKey) + { + await _store.FinishAsync(parentKey, + [new WorkStore.Finish(0, h.Status == "Completed" ? "Completed" : "Failed", ChildKey: childKey)], null); + } } - /// - /// How long a task must have sat Pending-and-unowned before the re-drive claims it. Long enough that a - /// task mid-deferral (each attempt can take up to the barrier timeout) is not stolen out from under the - /// attempt already in progress. - /// - private static readonly TimeSpan RedriveAge = TimeSpan.FromMinutes(5); + // ── executing a claimed task ── - /// - /// Re-drive backoff bounds. The first verification for a run runs at the status-timer cadence; each time - /// it confirms nothing orphaned the interval doubles up to , so a run stuck for - /// hours costs a handful of index reads rather than one per minute. The cap bounds how long a genuinely - /// orphaned task can wait to be caught (worst case ~RedriveMax), which the watchdog trades for the cost. - /// - private static readonly TimeSpan RedriveMax = TimeSpan.FromMinutes(15); + private string? Script(string? name) => + string.IsNullOrEmpty(name) ? null : _scripts.GetOrAdd(name, n => FindScript(n)); - /// - /// Re-queue tasks that are Pending in memory but that nothing owns — no queued job, no running job. - /// - /// This is the safety net for the state a deferral leaves behind. A task whose durable "Running" marker - /// could not be written is rolled back to Pending and retried, but only - /// times; after that it used to sit Pending with nothing to pick it up, because the only other retry - /// path was startup recovery. A healthy instance never restarts, so in production that meant 658 tasks - /// pending and 0 running for three days, with "Run X already active, skipping" preventing a fresh run - /// from doing the work instead. - /// - /// Ownership is decided by rather than by a timestamp on the - /// task, so a task queued normally is never double-dispatched. The age gate only applies to tasks with - /// a deferral history — anything Pending and unowned with no deferral record was lost some other way - /// and there is nothing to wait for. - /// - private void RedrivePendingTasks(OrchestratorRun run) => _ = RedrivePendingTasksAsync(run); + internal virtual string? FindScript(string name) => _psRunner.FindScript(name); - /// - /// Re-queue tasks whose durable queue row went missing — a task Pending forever with nothing left to - /// dispatch it. Runs off the 60s status timer. - /// - /// "Orphaned" has to mean "storage has no row for it". It used to mean "the JobManager does not have - /// it queued or running", which was true in the world where dispatch enqueued every task into the - /// JobManager immediately. Under the pump that is simply what a BACKLOG looks like: the pump holds a - /// worker-pool-sized buffer and leaves the rest in storage, so a 124-task run against eight workers - /// has most of its tasks Pending and absent from the JobManager for minutes. - /// - /// Treating that as orphaned re-queued the entire un-started backlog every 60 seconds. - /// - private async Task RedrivePendingTasksAsync(OrchestratorRun run) + /// Run a script on the pool, or on when pinned; returns its output when captured. + internal virtual async Task RunScriptAsync(string path, Dictionary parameters, bool captureOutput, + PowerShellWorker? worker = null) { - var now = DateTime.UtcNow; + if (captureOutput) return await _psRunner.ExecuteScriptWithOutput(path, parameters, pinnedWorker: worker); + await _psRunner.ExecuteScript(path, parameters, pinnedWorker: worker); + return string.Empty; + } - // Backoff gate. Once the verification below has confirmed a run has nothing orphaned, it need not run - // again for a while: orphaning is caused by specific rare events (a removed/expired queue row, a - // crash/migration), not something that spontaneously arises every 60s. Skipping here avoids the whole - // tick body — the candidates scan AND the storage read — for a run that verified clean, which for a - // large stuck backlog is nearly every run on nearly every tick. - if (_redriveBackoffEnabled && _redriveBackoff.TryGetValue(run.Name, out var st) && now < st.NextUtc) return; + /// Turn a claimed task into work. Registered on the JobManager; runs on the dispatched job. + private async Task?> ResolveTaskWorkAsync(JobDescriptor descriptor, CancellationToken ct) + { + if (descriptor.RunKey is not { } runKey) return null; + var header = await _store.GetRunAsync(runKey, ct); + if (header == null || header.IsFinished) return null; - List candidates; + if (descriptor.Seq == WorkStore.AggregateSeq) return jobCt => RunPostExecutionAsync(header, descriptor, jobCt); - lock (_lock) + var taskPath = Script(header.TaskScriptName); + if (taskPath == null) { - candidates = run.Tasks - .Where(t => t.Status == "Pending") - .Where(t => !_jobManager.IsQueuedOrRunning($"{run.Name}-{t.Id}")) - .Where(t => !_deferrals.TryGetValue(DeferralKey(run.Name, t.Id), out var s) - || now - s.LastUtc >= RedriveAge) - .ToList(); - - // Sequential run: one pinned driver runs every step, so the ONLY queue row that ever exists is - // the entry row that started the driver — the not-yet-reached steps deliberately have none. - // Applying the generic "Pending with no queue row = orphaned" rule to them would re-drive them - // all and spawn a second driver. The run is progressing whenever a driver is registered for it - // OR its entry job is still queued/running (that job's identity is one of this run's tasks) — - // re-drive nothing in either case. The driver registration closes the gap the entry-job check - // alone leaves open between steps (no task queued, none marked Running for an instant). Only when - // neither holds is the driver truly gone (never started, or died with the process): re-enqueue - // ONLY the current step (lowest Sequence still Pending) so a fresh driver resumes the run. - if (run.Sequential) - { - var driverActive = _activeSequentialDrivers.ContainsKey(run.Name) - || run.Tasks.Any(t => _jobManager.IsQueuedOrRunning($"{run.Name}-{t.Id}")); - if (driverActive) - { - candidates.Clear(); - } - else - { - var next = candidates.OrderBy(t => t.Sequence).FirstOrDefault(); - candidates = next != null ? [next] : []; - } - } + await FinishAsync(header, descriptor.Seq, "Failed", $"Task script {header.TaskScriptName} not found"); + return null; } - if (candidates.Count == 0) + return header.Sequential + ? BuildSequentialRunWork(header, descriptor, taskPath) + : jobCt => RunTaskAsync(header, descriptor.Seq, descriptor.TaskId, taskPath, jobCt); + } + + private Task FinishAsync(RunHeader h, int seq, string status, string? error = null) => + _finisher.FinishAsync(h.RunKey, new WorkStore.Finish(seq, status, error, Owner)); + + private async Task RunTaskAsync(RunHeader header, int seq, string taskId, string taskPath, CancellationToken ct) + { + if (header.CancelRequested) { - // No candidates to verify (all Pending tasks are queued/running, or none are Pending). Don't grow - // the backoff — a run mid-drain legitimately produces no candidates and should stay responsive — - // just clear any prior backoff so the next real candidate is checked promptly. - _redriveBackoff.TryRemove(run.Name, out _); + await FinishAsync(header, seq, "Cancelled", "Cancelled by user"); return; } - // Storage decides — the queue TABLE, not the index, which can outlive the rows it points at (see - // GetDispatchableTaskIdsAsync). Anything the pump cannot still claim is a ghost to re-enqueue. - // The sweep starts this for every live run each tick without awaiting it, so unguarded a slow - // verification overlapped the next tick's for the same run and the reads piled up in-process. - if (!_redriveInFlight.TryAdd(run.Name, 0)) return; + var parameters = await _store.GetPayloadAsync(header.RunKey, seq, ct); + if (parameters == null) + { + await FinishAsync(header, seq, "Failed", "The task's payload row is missing"); + return; + } - HashSet dispatchable; - var holdsSlot = false; try { - await _redriveSlots.WaitAsync(); - holdsSlot = true; - Interlocked.Increment(ref _redriveStorageReads); - dispatchable = await _queue.GetDispatchableTaskIdsAsync( - run.Name, candidates.Select(t => t.Id).ToList()); + var output = await RunScriptAsync(taskPath, TaskInvocation(parameters), header.HasPostExec); + if (!string.IsNullOrEmpty(output)) await _results.StoreResultAsync(header.RunKey, taskId, output); + } + catch (OperationCanceledException) when (ct.IsCancellationRequested) + { + await _store.ReleaseAsync(header.RunKey, seq, Owner, refundAttempt: true, CancellationToken.None); + throw; } catch (Exception ex) { - // Without this answer every candidate looks orphaned, which is the failure being fixed. - // Skip this tick; the timer comes back in 60s. - _logger.LogWarning(ex, "[Scheduler] Could not read queued tasks for {Run} — skipping re-drive", run.Name); - return; + await FinishAsync(header, seq, "Failed", ex.Message); + _logger.LogError(ex, "[Scheduler] Task failed: {TaskId}", taskId); + throw; } - finally + + await FinishAsync(header, seq, "Completed"); + _logger.LogDebug("[Scheduler] Task completed: {TaskId}", taskId); + } + + private async Task RunPostExecutionAsync(RunHeader header, JobDescriptor descriptor, CancellationToken ct) + { + var postExecScript = Script(_settings.Orchestrator.PostExecFunction); + if (postExecScript == null) { - if (holdsSlot) _redriveSlots.Release(); - _redriveInFlight.TryRemove(run.Name, out _); + _logger.LogError("[Orchestrator] PostExec function '{Func}' not found, cannot run PostExecution for {Name}", + _settings.Orchestrator.PostExecFunction, header.Name); + await FinishAsync(header, WorkStore.AggregateSeq, "Failed", "PostExec function not found"); + return; } - // Re-check under the lock: a candidate can be claimed, run and finish while waiting for a slot or the - // read, and its row is then gone for the right reason. - List orphaned; - lock (_lock) + _logger.LogInformation("[Orchestrator] Dispatching PostExecution Push-{Function} for run {Name} {Memory}", + header.PostExecFunctionName, header.Name, BackgroundTaskLimiter.GetMemorySnapshot()); + var tempFile = Path.Combine(Path.GetTempPath(), $"craft-postexec-{Guid.NewGuid():N}.jsonl"); + try + { + var count = await _results.StreamResultsToJsonLinesAsync(header.RunKey, tempFile, ct); + _logger.LogInformation("[Orchestrator] PostExec results for {Name}: {Count} results, {SizeMB:F1}MB streamed to temp file", + header.Name, count, new FileInfo(tempFile).Length / (1024.0 * 1024.0)); + + var parameters = new Dictionary + { + ["FunctionName"] = header.PostExecFunctionName!, + ["ResultsPath"] = tempFile, + }; + if (await _store.GetPostExecParametersAsync(header.RunKey, ct) is { Length: > 0 } postParameters) + parameters["ParametersJson"] = postParameters; + + await RunScriptAsync(postExecScript, parameters, captureOutput: false); + await OrchestratorBridge.DrainPendingAsync(); + _logger.LogInformation("[Orchestrator] PostExecution Push-{Function} completed for run {Name}", + header.PostExecFunctionName, header.Name); + await FinishAsync(header, WorkStore.AggregateSeq, "Completed"); + } + catch (OperationCanceledException) when (ct.IsCancellationRequested) { - orphaned = candidates - .Where(t => !dispatchable.Contains(t.Id)) - .Where(t => t.Status == "Pending" && !_jobManager.IsQueuedOrRunning($"{run.Name}-{t.Id}")) - .ToList(); + await _store.ReleaseAsync(header.RunKey, WorkStore.AggregateSeq, Owner, refundAttempt: true, CancellationToken.None); + throw; } - if (orphaned.Count == 0) + catch (Exception ex) { - // Verified clean: grow the interval (double, capped) so this run's next storage read is further - // out. A run stuck for hours thus costs O(log) reads, not one per minute. - if (_redriveBackoffEnabled) + _logger.LogError(ex, "[Orchestrator] PostExecution Push-{Function} failed for run {Name} (attempt {Attempt})", + header.PostExecFunctionName, header.Name, descriptor.Attempt); + if (descriptor.Attempt < Math.Max(1, _settings.Orchestrator.MaxRetries)) + await _store.ReleaseAsync(header.RunKey, WorkStore.AggregateSeq, Owner, refundAttempt: false, CancellationToken.None); + else { - var next = _redriveBackoff.TryGetValue(run.Name, out var cur) - ? TimeSpan.FromTicks(Math.Min(cur.Interval.Ticks * 2, RedriveMax.Ticks)) - : _redriveBase; - _redriveBackoff[run.Name] = (now + next, next); + _logger.LogError("[Scheduler] PostExecution for {Name} failed {Count} times — giving up and cleaning up results", + header.Name, descriptor.Attempt); + await FinishAsync(header, WorkStore.AggregateSeq, "Failed", ex.Message); } - return; + throw; + } + finally + { + try { if (File.Exists(tempFile)) File.Delete(tempFile); } + catch (Exception ex) { _logger.LogDebug(ex, "[Orchestrator] Failed to delete temp file {Path}", tempFile); } } - - // Found orphans — something is wrong with this run's queue rows, so snap back to close watch and - // re-drive them. - if (_redriveBackoffEnabled) - _redriveBackoff[run.Name] = (now + _redriveBase, _redriveBase); - // Clear the exhausted counters, or DeferTask would abandon them again on their first attempt. - foreach (var task in orphaned) _deferrals.TryRemove(DeferralKey(run.Name, task.Id), out _); - RequeueToTable(run, orphaned); - - _logger.LogWarning( - "[Scheduler] Re-drove {Count} orphaned Pending task(s) in {Run} — no runnable queue row and not queued or running", - orphaned.Count, run.Name); } - private void LogRunStatus(OrchestratorRun run) - { - // Nothing consumes this Info line at a higher level, and the flood of them is itself a measured - // cost, so do no work at all when Info is disabled. - if (!_logger.IsEnabled(LogLevel.Information)) return; + // ── sequential runs ── - // One pass, not four Count(predicate) calls. Enumerable.Count over the List boxes an enumerator per - // call, and this runs on every run's 60s timer — four boxed enumerators × M runs per minute. - int completed = 0, failed = 0, running = 0, pending = 0; - lock (_lock) + /// + /// A sequential run: one driver checks out one worker and runs every step on it in payload order, claiming + /// each step itself under the run's driver lease so no other worker can take a step meanwhile. A failed + /// step is recorded and the next one runs. Cancelling the run stops it at the next step. + /// + private Func BuildSequentialRunWork(RunHeader header, JobDescriptor first, string taskPath) => + async jobCt => { - foreach (var t in run.Tasks) + PowerShellWorker? worker = null; + var faulted = false; + var step = new WorkStore.ClaimedTask(header.RunKey, first.Seq, first.TaskId, first.Attempt); + try { - switch (t.Status) + worker = CheckoutSequentialWorker(jobCt); + while (true) { - case "Completed": completed++; break; - case "Failed": failed++; break; - case "Running": running++; break; - case "Pending": pending++; break; + var current = await _store.GetRunAsync(header.RunKey, jobCt); + if (current == null || current.IsFinished) break; + if (current.CancelRequested) + { + await FinishAsync(header, step.Seq, "Cancelled", "Cancelled by user"); + await _store.CancelPendingAsync(header.RunKey, jobCt); + break; + } + if (step.Seq == WorkStore.AggregateSeq) + { + var aggregate = new JobDescriptor(header.Name, step.TaskId, header.Priority) { RunKey = header.RunKey, Seq = step.Seq, Attempt = step.Attempt }; + await RunPostExecutionAsync(current, aggregate, jobCt); + break; + } + + var parameters = await _store.GetPayloadAsync(header.RunKey, step.Seq, jobCt); + try + { + if (parameters == null) throw new InvalidOperationException("The task's payload row is missing"); + var output = await RunScriptAsync(taskPath, TaskInvocation(parameters), current.HasPostExec, worker); + if (current.HasPostExec && !string.IsNullOrEmpty(output)) + await _results.StoreResultAsync(header.RunKey, step.TaskId, output); + await FinishAsync(header, step.Seq, "Completed"); + } + catch (OperationCanceledException) when (jobCt.IsCancellationRequested) + { + throw; + } + catch (Exception ex) + { + _logger.LogError(ex, "[Scheduler] Sequential task failed: {TaskId} — continuing with the next step", step.TaskId); + await FinishAsync(header, step.Seq, "Failed", ex.Message); + } + + if (await _store.ClaimSequentialAsync(header.RunKey, Owner, Lease, continuing: true, jobCt) is not { } next) break; + step = next; } } - } + catch (OperationCanceledException) when (jobCt.IsCancellationRequested) + { + faulted = true; + await _store.ReleaseAsync(header.RunKey, step.Seq, Owner, refundAttempt: true, CancellationToken.None); + throw; + } + catch (Exception ex) + { + faulted = true; + _logger.LogError(ex, "[Scheduler] Sequential driver for {Run} failed", header.Name); + throw; + } + finally + { + ReclaimSequentialWorker(worker, faulted); + await _store.ReleaseDriverAsync(header.RunKey, Owner, CancellationToken.None); + } + }; - // Skip the line — and the string format, the nine boxed args, and the memory snapshot it needs — - // when nothing has changed since the last tick. A run parked at "0 running / N pending" for hours - // re-emitted the identical line every 60s (M of them per minute at scale). Log on a real transition, - // plus a slow heartbeat so a long-lived run still shows it is alive. - var now = DateTime.UtcNow; - if (_lastStatusLog.TryGetValue(run.Name, out var prev) - && prev.C == completed && prev.F == failed && prev.R == running && prev.P == pending - && now - prev.LoggedUtc < StatusHeartbeat) - return; - _lastStatusLog[run.Name] = (completed, failed, running, pending, now); + internal virtual PowerShellWorker? CheckoutSequentialWorker(CancellationToken ct) => _psRunner.CheckoutBackgroundWorker(ct); - var elapsed = now - run.StartedUtc; - _logger.LogInformation( - "[Scheduler] Run {Name} T+{Elapsed:F1}min: {Completed}/{Total} done {Running} running {Pending} pending {Failed} failed jobs={Active}a/{Queued}q {Memory}", - run.Name, elapsed.TotalMinutes, completed, run.Tasks.Count, running, pending, failed, - _jobManager.ActiveCount, _jobManager.QueuedCount, - BackgroundTaskLimiter.GetMemorySnapshot()); + internal virtual void ReclaimSequentialWorker(PowerShellWorker? worker, bool faulted) + { + if (worker != null) _psRunner.ReclaimBackgroundWorker(worker, faulted: faulted); } - private void CheckRunCompletion(OrchestratorRun run) + private static Dictionary TaskInvocation(Dictionary parameters) => + new() { ["TaskJson"] = JsonSerializer.Serialize(parameters, s_jsonOptions) }; + + // ── operator actions ── + + /// Cancel a run's pending tasks; running ones finish. Returns whether the run exists and how many were cancelled. + public async Task<(bool found, int cancelledCount)> CancelRunAsync(string name) { - // Already locked by caller - if (run.Tasks.All(t => t.Status is "Completed" or "Failed" or "Cancelled")) - { - // Don't finalize if child runs (sub-orchestrators spawned by tasks) are still active - if (!AllChildRunsComplete(run.Name)) - { - _logger.LogInformation( - "[Scheduler] Run {Name} tasks complete but waiting for child runs to finish", - run.Name); - return; - } + var header = await _store.GetRunByNameAsync(TableKeys.Sanitize(name)); + if (header == null || header.IsFinished) return (false, 0); - // Cannot await inside lock — schedule finalization - _ = Task.Run(async () => - { - try - { - // Flush before the counter read below. The batched status writer decrements the counter - // only when a terminal write flushes, so an unflushed read sees the pre-decrement value - // and defers a finalize that is due — a single GET beats the drain+decrement every time, - // stalling every run until the 60s timer. Same flush FinalizeRunCoreAsync relies on, just - // ahead of the veto read; bounded by the barrier timeout, so it cannot hang. - await _writer.FlushAsync(); - - // The in-memory graph proposes, storage disposes. Finalizing is irreversible - it - // writes the aggregate and cleans the run up - so it must not run while storage still - // shows work outstanding, which is exactly the case when terminal writes have not yet - // flushed. A null count means the run predates the counter and cannot veto anything. - var remaining = await _store.GetRemainingAsync(run.Name); - if (remaining is > 0) - { - // A counter that keeps contradicting a fully-terminal graph is drifted, not - // busy — a decrement that exhausted its retries is never re-applied, and - // without a recount this deferral repeats on every 60s tick for the process - // lifetime, pinning the run graph with it. Give in-flight terminal writes a - // few checks to land, then recount the partition the counter summarizes. - var misses = _finalizeDeferrals.AddOrUpdate(run.Name, 1, (_, c) => c + 1); - if (misses >= ReconcileAfterDeferrals) - { - _finalizeDeferrals.TryRemove(run.Name, out _); - if (await _store.ReconcileRemainingAsync(run.Name) is 0) - { - await FinalizeRunAsync(run); - return; - } - } - _logger.LogInformation( - "[Scheduler] Run {Name} complete in memory but storage shows {Remaining} outstanding - deferring finalize", - run.Name, remaining); - return; - } + await _store.RequestCancelAsync(header.RunKey); + var (cancelled, _) = await _store.CancelPendingAsync(header.RunKey); + _logger.LogInformation("[Scheduler] Run {Name} cancelled: {Cancelled} pending tasks cancelled", header.Name, cancelled); + return (true, cancelled); + } - _finalizeDeferrals.TryRemove(run.Name, out _); - await FinalizeRunAsync(run); - } - catch (Exception ex) { _logger.LogError(ex, "[Scheduler] FinalizeRun failed for {Name}", run.Name); } - }); - } + /// Cancel one task that is still pending in storage. False when it is not pending (or not found). + public async Task TryCancelQueuedTaskAsync(string runName, string taskId) + { + var header = await _store.GetRunByNameAsync(runName); + if (header == null || header.IsFinished) return false; + var task = (await _store.GetTasksAsync(header.RunKey, 'P')).FirstOrDefault(t => t.TaskId == taskId); + if (task == null) return false; + var outcome = await _store.FinishAsync(header.RunKey, [new WorkStore.Finish(task.Seq, "Cancelled", "Cancelled by user")], 'P'); + return outcome?.Applied > 0; } - /// - /// Register a child run under a parent at ENQUEUE time — while the parent task's script is - /// still executing, so the parent cannot pass its completion check before the link exists. The - /// parent will not finalize while any child is pending dispatch, active, or recovering. Only - /// registers if the parent is still active; returns whether a pending gate was taken (the - /// bridge releases exactly what was taken via ). - /// - internal bool TryRegisterPendingChildRun(string parentRunName, string childRunName) + /// Move a run to another priority band. Applies to the whole run: its tasks share one queue position. + public async Task ReprioritizeRunAsync(string runName, int priority) { - // A run re-queued from inside its own context arrives with itself as parent — the - // recurring-run pattern, or a duplicate enqueue of an already-active run. Linking it would - // deadlock finalization: the run stays in _activeRuns until it finalizes, so - // AllChildRunsComplete would wait on the run itself forever. Observed live as runs stuck - // "Running" for days with every task terminal and Remaining=0. - if (parentRunName == childRunName) - return false; - - if (!_activeRuns.ContainsKey(parentRunName)) - return false; // Parent no longer active (e.g. queued from PostExec context) - - // Gate before link: the moment the link is visible to AllChildRunsComplete the pending - // mark must already hold, or a completion check could slip between the two writes. - _pendingChildRuns.AddOrUpdate(childRunName, 1, (_, n) => n + 1); - _childRuns.GetOrAdd(parentRunName, _ => new ConcurrentBag()).Add(childRunName); - _logger.LogInformation("[Orchestrator] Registered child run {Child} under parent {Parent}", - childRunName, parentRunName); - return true; + var header = await _store.GetRunByNameAsync(runName); + return header is { IsFinished: false } && await _store.SetPriorityAsync(header.RunKey, Math.Clamp(priority, 0, 99)); } - /// - /// Lift the enqueue-time gate for ONE queued entry of this child. Called by the bridge after - /// the start attempt finishes, whatever the outcome: a started child is in _activeRuns by then - /// (which takes over blocking the parent), and one that failed to start must stop blocking — a - /// leaked gate would defer the parent's finalize for the process lifetime, re-checked every - /// 60s. Counted rather than boolean so two queued entries under the same child name cannot - /// release each other's gate. - /// - internal void ReleasePendingChildRun(string childRunName) + /// Cancel every pending task of every run — the whole backlog. Returns how many were cancelled. + public async Task ClearQueueAsync(CancellationToken ct = default) { - while (_pendingChildRuns.TryGetValue(childRunName, out var n)) + var total = 0; + var entries = new List(); + await foreach (var e in _store.ReadReadyAsync(200, ct)) entries.Add(e); + foreach (var e in entries) { - if (n <= 1) - { - if (_pendingChildRuns.TryRemove(new KeyValuePair(childRunName, n))) - return; - } - else if (_pendingChildRuns.TryUpdate(childRunName, n - 1, n)) - { - return; - } + await _store.RequestCancelAsync(e.RunKey, ct); + total += (await _store.CancelPendingAsync(e.RunKey, ct)).Cancelled; } + _logger.LogWarning("[JobQueue] Durable queue cleared — {Count} queued task(s) cancelled", total); + return total; } - private bool AllChildRunsComplete(string runName) + /// A claimed job reprioritized in the local buffer: it is already claimed, so nothing is stored. + public void PriorityChanged(JobDescriptor descriptor, int newPriority) { } + + /// A claimed job cancelled before it started: record it, so its claim does not lapse and run it. + public void Cancelled(JobDescriptor descriptor) { - if (!_childRuns.TryGetValue(runName, out var children)) - return true; - // A run is never its own blocker. TryRegisterPendingChildRun refuses self-links, but ones - // registered before that guard existed can still be sitting in the bag of a long-lived - // process. - return !children.Any(childName => - childName != runName && - (_pendingChildRuns.ContainsKey(childName) || - _activeRuns.ContainsKey(childName) || - _recoveringChildren.ContainsKey(childName))); + if (descriptor.RunKey is not { } runKey) return; + _ = _finisher.FinishAsync(runKey, new WorkStore.Finish(descriptor.Seq, "Cancelled", "Cancelled by user", Owner)); } - /// - /// Declare a run finished: write its terminal status, drop it from the live graph, and hand off to - /// post-execution. Runs at most once per run — see the claim below. - /// - private async Task FinalizeRunAsync(OrchestratorRun run) - { - // Finalize once. This is not an idempotent method: it re-arms PostExecStatus and dispatches the - // post-execution again, so entering it twice runs the run's aggregation twice. Observed live - // before this guard: one 13-task run finalized 7 times and dispatched Push-StoreMailboxRules 7 - // times. Idempotent consumers hid it; Push-ScheduledTaskPostExecution did not, because it - // advances a recurring task by one interval per invocation. - // - // The claim lives here rather than at the call sites because there are four of them — normal - // completion, two resume paths, and cancellation — and cancellation can race a completion. - // DispatchPendingTasksAsync releases it when a run becomes live again, so a recurring run name - // can finalize on its next outing. - if (!_finalizingRuns.TryAdd(run.Name, true)) - { - _logger.LogDebug("[Scheduler] Run {Name} has already been finalized — ignoring duplicate", run.Name); - return; - } + // ── lookups ── - try - { - await FinalizeRunCoreAsync(run); - } - catch - { - // Still needs finalizing, so it must stay claimable — the 60s status timer retriggers it. - _finalizingRuns.TryRemove(run.Name, out _); - throw; - } - } + public string? GetRunReference(string runName) => + Task.Run(() => _store.GetRunByNameAsync(runName)).GetAwaiter().GetResult()?.Reference; - private async Task FinalizeRunCoreAsync(OrchestratorRun run) + public string? FindRunByReference(string reference) => Task.Run(async () => { - var failed = run.Tasks.Count(t => t.Status == "Failed"); - var cancelled = run.Tasks.Count(t => t.Status == "Cancelled"); - var completed = run.Tasks.Count(t => t.Status == "Completed"); - - run.Status = (failed > 0 || cancelled > 0) ? "CompletedWithErrors" : "Completed"; - run.CompletedUtc = DateTime.UtcNow; - var wallClock = run.CompletedUtc.Value - run.StartedUtc; - - // Set PostExecStatus before persisting, so crash between here and DispatchPostExecution is recoverable - if (!string.IsNullOrEmpty(run.PostExecFunctionName)) - run.PostExecStatus = "Pending"; - - // Flush-before-finalize: queue the run's final status, then await a full drain so every task's terminal - // state + this run status are durable BEFORE we declare the run done and dispatch PostExecution. - _writer.QueueRun(run); - await _writer.FlushAsync(); - - // Removed from _activeRuns first, so the status sweep stops ticking it before its per-run maps go. - _activeRuns.TryRemove(run.Name, out _); - _cancelledRuns.TryRemove(run.Name, out _); - _taskScriptPaths.TryRemove(run.Name, out _); - _lastStatusLog.TryRemove(run.Name, out _); - _redriveBackoff.TryRemove(run.Name, out _); - _finalizeDeferrals.TryRemove(run.Name, out _); - // Deferral and re-queue tracking is keyed per task and nothing else removes entries for tasks - // that ended without passing through their happy-path cleanup — without this sweep the residue - // of every run that ever deferred outlives the run. - foreach (var t in run.Tasks) - { - var key = DeferralKey(run.Name, t.Id); - _deferrals.TryRemove(key, out _); - _requeueFailures.TryRemove(key, out _); - } + await foreach (var e in _store.ReadReadyAsync(200)) + if (string.Equals(e.Reference, reference, StringComparison.OrdinalIgnoreCase)) return e.Name; + return null; + }).GetAwaiter().GetResult(); - var wallDisplay = wallClock.TotalSeconds < 60 - ? $"{wallClock.TotalSeconds:F1}s" - : $"{wallClock.TotalMinutes:F1}min"; + // ── startup, retention, status lines ── - _logger.LogInformation( - "[Scheduler] Run {Name} finalized: {Status} ({Completed}/{Failed}/{Cancelled}/{Total}) wall={Wall} {Memory}", - run.Name, run.Status, completed, failed, cancelled, run.Tasks.Count, wallDisplay, - BackgroundTaskLimiter.GetMemorySnapshot()); + /// + /// Startup. Storage is the state, so there is nothing to recover: create the tables, drop the previous + /// design's tables once, and sweep expired runs. + /// + public async Task ResumeInterruptedRunsAsync(CancellationToken ct) + { + await _store.InitializeAsync(ct); + await _store.DropLegacyTablesAsync(ct); + try { await RunRetentionSweepAsync(ct); } + catch (Exception ex) { _logger.LogWarning(ex, "[Scheduler] Startup retention sweep failed"); } + } - // If this was a child run, re-check parent's completion — it may have been - // waiting for this child to finish before it can finalize and run PostExecution - if (!string.IsNullOrEmpty(run.ParentRunName) && - _activeRuns.TryGetValue(run.ParentRunName, out var parentRun)) - { - lock (_lock) { CheckRunCompletion(parentRun); } - } + public async Task RunRetentionSweepAsync(CancellationToken ct) + { + var removed = await _store.SweepFinishedAsync(TimeSpan.FromHours(Math.Max(1, _settings.Orchestrator.RetentionHours)), ct); + if (removed > 0) _logger.LogInformation("[OrchestratorStore] Retention sweep removed {Count} finished run(s)", removed); + return removed; + } - // Cleanup child run tracking for this run - _childRuns.TryRemove(run.Name, out _); - - // Anything of this run's still queued is now moot; leaving rows behind has the pump claim work - // for a run that is already finished. That applies to EVERY finalized run — this used to sit in - // the else below, so a run WITH post-execution kept its queue rows from finalize until - // post-execution succeeded, and the pump spent that window re-claiming them. Observed live: one - // 13-task run finalized 7 times, dispatched its Push-* aggregation 7 times, and re-ran - // individual tasks up to 4 times each. The durable queue only ever carries TASKS — the - // post-execution job is enqueued in-memory on the JobManager and, after a crash, is re-derived - // from PostExecStatus — so dropping these rows here cannot cost the post-execution its retry. - _ = _queue.RemoveRunAsync(run.Name, run.StartedUtc); - - // Dispatch PostExecution if configured - if (!string.IsNullOrEmpty(run.PostExecFunctionName)) - { - DispatchPostExecution(run); - } - else + public async Task RunRetentionLoopAsync(CancellationToken ct) + { + var hours = _settings.Orchestrator.CleanupIntervalHours; + if (hours <= 0) return; + using var timer = new PeriodicTimer(TimeSpan.FromHours(hours)); + try { - // No PostExec — cleanup results table (if any stray entries exist) - _ = _store.CleanupRunAsync(run.Name); + while (await timer.WaitForNextTickAsync(ct)) + { + try { await RunRetentionSweepAsync(ct); } + catch (Exception ex) when (ex is not OperationCanceledException) + { + _logger.LogWarning(ex, "[Scheduler] Retention sweep failed; next attempt in {Hours}h", hours); + } + } } + catch (OperationCanceledException) { } } - private void DispatchPostExecution(OrchestratorRun run) + /// + /// One status line per active run on a change, or every ten minutes when unchanged: + /// T+{min}min: {done}/{total} done {running} running {pending} pending {failed} failed. Health checks parse it to + /// spot runs that sit with work pending and nothing running. + /// + public async Task RunStatusSweepLoopAsync(CancellationToken ct) { - var postExecFunc = _settings.Orchestrator.PostExecFunction; - var postExecScript = !string.IsNullOrEmpty(postExecFunc) ? _psRunner.FindScript(postExecFunc) : null; - if (postExecScript == null) + using var timer = new PeriodicTimer(TimeSpan.FromSeconds(Math.Max(1, _settings.Orchestrator.StatusTimerIntervalSeconds))); + try { - _logger.LogError("[Orchestrator] PostExec function '{Func}' not found, cannot run PostExecution for {Name}", - postExecFunc, run.Name); - return; - } - - _logger.LogInformation( - "[Orchestrator] Dispatching PostExecution Push-{Function} for run {Name} {Memory}", - run.PostExecFunctionName, run.Name, BackgroundTaskLimiter.GetMemorySnapshot()); - - _jobManager.Enqueue( - name: $"{run.Name}-PostExec", - priority: run.Priority, - // Post-exec commonly starts follow-up runs (baseline → cache refresh); they should land - // at this run's priority, not the enqueue default. - inheritPriority: run.Priority, - runName: run.Name, - work: async (jobCt) => + while (await timer.WaitForNextTickAsync(ct)) { - // Mark PostExec as Running, and count the attempt. Incremented BEFORE the work so a - // crash mid-post-execution still burns an attempt — otherwise a post-execution that - // kills the host would be retried forever, which is precisely what the bound is for. - run.PostExecStatus = "Running"; - run.PostExecAttemptCount++; - await _store.UpsertRunAsync(run); - - // Stream results to a temp file rather than building the aggregate in memory. For large - // runs (738+ tasks) that aggregate is 50-150 MB, and StreamResultsToJsonLinesAsync holds - // one chunk at a time rather than the whole entity set (it used to buffer every result - // row into a dictionary before writing a byte, so "streaming" still peaked at the full - // payload in UTF-16 — roughly 2x the stored size — before this ran). - // - // The path — not the content — is what goes to PowerShell. This used to read the file - // back with File.ReadAllTextAsync and pass it as a ResultsJson string, which put the - // whole aggregate on the Large Object Heap and had PowerShell copy it a second time on - // ConvertFrom-Json. Handing over the path instead lets Invoke-CraftPostExecution walk - // the file one line at a time, so neither copy is ever made. The file is JSON Lines for - // exactly that reason; see StreamResultsToJsonLinesAsync. - // - // Consequence for the file's lifetime: it must now survive until PowerShell has read - // it, so it is deleted in the finally below rather than immediately after streaming. - var tempFile = Path.Combine(Path.GetTempPath(), $"craft-postexec-{Guid.NewGuid():N}.jsonl"); - try - { - var resultCount = await _store.StreamResultsToJsonLinesAsync(run.Name, tempFile, jobCt); - var fileSize = new FileInfo(tempFile).Length; - var fileSizeMB = fileSize / (1024.0 * 1024.0); - _logger.LogInformation( - "[Orchestrator] PostExec results for {Name}: {Count} results, {SizeMB:F1}MB streamed to temp file {Memory}", - run.Name, resultCount, fileSizeMB, BackgroundTaskLimiter.GetMemorySnapshot()); - - var parameters = new Dictionary - { - { "FunctionName", run.PostExecFunctionName! }, - { "ResultsPath", tempFile } - }; - if (!string.IsNullOrEmpty(run.PostExecParametersJson)) - parameters["ParametersJson"] = run.PostExecParametersJson; - - await _psRunner.ExecuteScript(postExecScript, parameters); - - // PostExecution functions may call Start-CIPPOrchestrator (Phase 2) - await OrchestratorBridge.DrainPendingAsync(); - - // Mark PostExec as Completed - run.PostExecStatus = "Completed"; - await _store.UpsertRunAsync(run); - - _logger.LogInformation("[Orchestrator] PostExecution Push-{Function} completed for run {Name}", - run.PostExecFunctionName, run.Name); - - // Cleanup after successful PostExec - await _store.CleanupRunAsync(run.Name); - await _queue.RemoveRunAsync(run.Name, run.StartedUtc, jobCt); - } - catch (Exception ex) - { - // Mark PostExec as Failed. ResumeInterruptedRunsAsync picks "Failed" back up on the - // next startup, up to MaxPostExecAttempts; the run's Results rows stay in storage - // until it either succeeds or is abandoned, because they are the retry's input. - run.PostExecStatus = "Failed"; - try { await _store.UpsertRunAsync(run); } catch { /* best effort */ } - _logger.LogError(ex, "[Orchestrator] PostExecution Push-{Function} failed for run {Name}", - run.PostExecFunctionName, run.Name); - throw; - } - finally + try { await LogRunStatusAsync(ct); } + catch (Exception ex) when (ex is not OperationCanceledException) { - // The only delete. PowerShell reads the file during ExecuteScript above, so it - // cannot be freed any earlier — and it must still be freed when that throws. - try { if (File.Exists(tempFile)) File.Delete(tempFile); } - catch (Exception ex) { _logger.LogDebug(ex, "[Orchestrator] Failed to delete temp file {Path}", tempFile); } + _logger.LogWarning(ex, "[Scheduler] Run status sweep failed"); } } - ); + } + catch (OperationCanceledException) { } + } + + private async Task LogRunStatusAsync(CancellationToken ct) + { + if (!_logger.IsEnabled(LogLevel.Information)) return; + var running = _jobManager.GetJobs(status: "Running").Where(j => j.RunName != null) + .GroupBy(j => j.RunName!).ToDictionary(g => g.Key, g => g.Count()); + var now = DateTime.UtcNow; + await foreach (var e in _store.ReadReadyAsync(200, ct)) + { + var r = running.GetValueOrDefault(e.Name); + var p = Math.Max(0, e.Total - e.Done - r); + if (_lastStatusLog.TryGetValue(e.RunKey, out var prev) && prev.C == e.Done && prev.R == r && prev.P == p + && now - prev.LoggedUtc < StatusHeartbeat) continue; + _lastStatusLog[e.RunKey] = (e.Done, r, p, now); + _logger.LogInformation( + "[Scheduler] Run {Name} T+{Elapsed:F1}min: {Completed}/{Total} done {Running} running {Pending} pending {Failed} failed jobs={Active}a/{Queued}q {Memory}", + e.Name, (now - e.StartedUtc).TotalMinutes, e.Done - e.Failed - e.Cancelled, e.Total, r, p, e.Failed, + _jobManager.ActiveCount, _jobManager.QueuedCount, BackgroundTaskLimiter.GetMemorySnapshot()); + } } + // ── parsing batches into tasks ── + private List ParseTasksFromJson(string json, string runName) { var tasks = new List(); @@ -2466,117 +882,4 @@ internal static void AddTaskFromElement(List tasks, HashSe Status = "Pending" }); } - - private string? FindTaskScript(string runName) - { - // Convention: strip "Start-" prefix → "Invoke-{rest}Task" - var baseName = runName.StartsWith("Start-", StringComparison.OrdinalIgnoreCase) - ? runName[6..] - : runName; - return _psRunner.FindScript($"Invoke-{baseName}Task"); - } - - /// - /// Empty the durable job queue (maintenance/reset). Delegates to - /// — see its remarks: in-flight work is unaffected and Pending tasks may be re-driven, so pair this - /// with when the intent is to STOP work rather than clear a wedged queue. - /// Returns the number of queue rows removed. - /// - public Task ClearQueueAsync(CancellationToken ct = default) => _queue.ClearAllAsync(ct); - - /// - /// Cancel a running orchestrator run. Pending tasks are marked Cancelled immediately. - /// Already-running tasks are allowed to finish (no force-kill). - /// Queued jobs in the JobManager will be skipped when they are dequeued. - /// - public async Task<(bool found, int cancelledCount)> CancelRunAsync(string name) - { - // The live graph when this node holds the run: cancelling a copy left the live tasks Pending, so - // the re-drive kept re-queueing them and the live run never finalized. - var run = _activeRuns.TryGetValue(name, out var live) ? live : await _store.GetRunAsync(name); - if (run == null) return (false, 0); - - // Mark this run as cancelled so dispatched-but-not-yet-started tasks skip execution - _cancelledRuns.TryAdd(name, true); - - int cancelled; - var tasksToUpdate = new List(); - lock (_lock) - { - var pendingTasks = run.Tasks.Where(t => t.Status == "Pending").ToList(); - cancelled = pendingTasks.Count; - foreach (var task in pendingTasks) - { - task.Status = "Cancelled"; - task.LastError = "Cancelled by user"; - task.CompletedUtc = DateTime.UtcNow; - tasksToUpdate.Add(task); - } - } - - // Persist cancelled task states through the status-guarded cancel, not a plain upsert. Two - // things ride on that. Cancelled is terminal, so the write must decrement the run counter — - // skip it and CheckRunCompletion's finalize veto reads "{cancelled count} still outstanding" - // forever once the running tasks drain. And the write must only land while storage still shows - // Pending: dispatch can move a task to Running between our read above and this write, and - // clobbering that would have the task's real completion decrement a second time. A task that - // moved on is un-cancelled in our copy and left to finish. - foreach (var t in tasksToUpdate) - { - var result = await _store.CancelPendingTaskAsync(run.Name, t); - if (result.Cancelled) continue; - - cancelled--; - lock (_lock) - { - t.Status = result.CurrentStatus ?? "Running"; - t.LastError = null; - t.CompletedUtc = null; - } - } - - await _queue.RemoveRunAsync(run.Name, run.StartedUtc); - - // Check if the run is now fully done (Running tasks will finalize themselves) - var remaining = run.Tasks.Count(t => t.Status is "Running"); - if (remaining == 0) - { - await FinalizeRunAsync(run); - } - else - { - await _store.UpsertRunAsync(run); - } - - _logger.LogInformation("[Scheduler] Run {Name} cancelled: {Cancelled} pending tasks cancelled, {Running} still running", - name, cancelled, run.Tasks.Count(t => t.Status == "Running")); - - return (true, cancelled); - } - - /// - /// Check whether a run has been cancelled (used by dispatch to skip queued tasks). - /// - public bool IsRunCancelled(string runName) => _cancelledRuns.ContainsKey(runName); - - /// - /// Get the current state of a run (or null if it doesn't exist). - /// Used by the API status endpoint. - /// - public async Task GetRunStatusAsync(string name) - { - await _store.InitializeAsync(); - await _queue.InitializeAsync(); - return await _store.GetRunAsync(name); - } - - /// - /// List all known run names from table storage. - /// - public async Task> ListRunsAsync() - { - await _store.InitializeAsync(); - await _queue.InitializeAsync(); - return await _store.ListRunsAsync(); - } } diff --git a/Services/Orchestration/OrchestratorStatusWriter.cs b/Services/Orchestration/OrchestratorStatusWriter.cs deleted file mode 100644 index 1208e8b..0000000 --- a/Services/Orchestration/OrchestratorStatusWriter.cs +++ /dev/null @@ -1,388 +0,0 @@ -using System.Text.Json; -using Craft.Configuration; -using Craft.Storage; - -namespace Craft.Orchestration; - -/// -/// Coalescing, batched, durable writer for orchestrator TASK and RUN status transitions. It removes the -/// per-task Azure Table write from the fan-out critical path (that write was the throughput ceiling — see -/// docs/orch-analysis.md) by coalescing many transitions and flushing them in ≤100-entity, byte-budgeted -/// transactions. -/// -/// Durability is preserved: -/// - the pre-invoke "Running" marker is written under a synchronous barrier (batched across concurrently -/// starting tasks, but still durable-BEFORE-invoke, so AttemptCount/MaxRetries still bound poison tasks); -/// - guarantees all pending terminal states are persisted before a run finalizes; -/// - a final drain runs on shutdown. -/// -/// RESULTS are deliberately NOT handled here — OrchestratorTableStore.StoreResultAsync keeps its -/// property-chunking / multi-row large-payload path completely untouched. -/// -public sealed class OrchestratorStatusWriter : IDisposable -{ - private readonly OrchestratorTableStore _store; - private readonly ILogger _logger; - private readonly bool _enabled; - private readonly bool _durableBarrier; - private readonly bool _batchResults; - private readonly int _flushIntervalMs; - - /// Results at or below this many chars fit a single Azure Table property and can be coalesced - /// here; larger ones need the chunked multi-row path and are written directly by the caller. Matches - /// OrchestratorTableStore's single-property fast-path bound. - private const int SmallResultMaxChars = 30_000; - private readonly TimeSpan _barrierTimeout; - private readonly TimeSpan _flushTimeout; - private readonly int _flushConcurrency; - - // Match the read path (OrchestratorTableStore serializes/deserializes ParametersJson camelCase). - private static readonly JsonSerializerOptions s_json = new() { PropertyNamingPolicy = JsonNamingPolicy.CamelCase }; - - private readonly object _lock = new(); - private Dictionary _pendingTasks = new(); // key: runName  taskId (last-wins coalesce) - private Dictionary _pendingRuns = new(); // key: runName - private Dictionary _pendingResults = new(); // key: run + task - private TaskCompletionSource _barrier = new(TaskCreationOptions.RunContinuationsAsynchronously); - private readonly SemaphoreSlim _signal = new(0, int.MaxValue); - private readonly CancellationTokenSource _cts = new(); - private readonly Task _drainLoop; - - /// Set first in so a status enqueue racing shutdown drops its wake - /// instead of throwing ObjectDisposedException into a job that already succeeded. - private volatile bool _disposed; - - public bool Enabled => _enabled; - - public OrchestratorStatusWriter(OrchestratorTableStore store, ILogger logger, CraftSettings settings) - { - _store = store; - _logger = logger; - _enabled = settings.Orchestrator.BatchStatusWrites; - _durableBarrier = settings.Orchestrator.DurableRunningBarrier; - _batchResults = settings.Orchestrator.BatchResultWrites; - _flushIntervalMs = Math.Max(5, settings.Orchestrator.StatusFlushIntervalMs); - _barrierTimeout = TimeSpan.FromSeconds(Math.Max(1, settings.Orchestrator.RunningBarrierTimeoutSeconds)); - _flushTimeout = TimeSpan.FromSeconds(Math.Max(1, settings.Orchestrator.StatusFlushTimeoutSeconds)); - _flushConcurrency = Math.Max(1, settings.Orchestrator.StatusFlushConcurrency); - _drainLoop = _enabled ? Task.Run(DrainLoopAsync) : Task.CompletedTask; - _logger.LogInformation( - "[Orchestrator] StatusWriter: enabled={E} durableBarrier={B} flushMs={F} barrierTimeout={BT}s flushTimeout={FT}s concurrency={C}", - _enabled, _durableBarrier, _flushIntervalMs, _barrierTimeout.TotalSeconds, _flushTimeout.TotalSeconds, - _flushConcurrency); - } - - private static string Key(string run, string task) => run + "" + task; - private static TaskStatusWrite Snap(string run, OrchestratorTaskItem t) => new( - run, t.Id, t.Status, JsonSerializer.Serialize(t.Parameters, s_json), t.AttemptCount, t.LastError, - t.CompletedUtc, t.Priority, t.Sequence); - - /// Persist the pre-invoke "Running" marker durably before the task runs. Under the barrier it is - /// batched with other concurrently-starting tasks (N tasks → ~1 transaction) yet still lands before the - /// invoke. Disabled → the original per-task awaited write. - /// - /// The marker did not persist within RunningBarrierTimeoutSeconds. The caller must treat this - /// as a DEFERRAL, not a failure: nothing was written, so storage still has the task Pending and it - /// is safe — and necessary — to retry it. - /// - public async Task MarkRunningAsync(string runName, OrchestratorTaskItem task, CancellationToken ct = default) - { - if (!_enabled) { await _store.UpsertTaskAsync(runName, task); return; } - if (!_durableBarrier) { QueueTask(runName, task); return; } // eventual mode (weaker poison guarantee) - - Task barrier; - lock (_lock) - { - _pendingTasks[Key(runName, task.Id)] = Snap(runName, task); - barrier = _barrier.Task; - } - Signal(); - - // Bounded. This wait sits between dispatch and worker checkout while holding a JobManager slot, - // so waiting forever converts a slow flush into a whole-host outage — observed in production as - // 8/8 slots held, every BG worker idle, 1,919 jobs queued and nothing moving until a restart. - try - { - if (await CompletesWithinAsync(barrier, _barrierTimeout, ct)) return; - } - catch (Exception ex) when (ex is not OperationCanceledException) - { - // The flush carrying this marker failed. It has been requeued, but THIS task must not start: - // the marker is not in storage, so its retry bound would not hold. - throw new MarkerNotPersistedException( - $"Durable 'Running' marker for {runName}/{task.Id} did not persist: {ex.Message}", ex); - } - - throw new MarkerNotPersistedException( - $"Durable 'Running' marker for {runName}/{task.Id} did not persist within " + - $"{_barrierTimeout.TotalSeconds:F0}s. The task was not started and remains Pending."); - } - - /// - /// Await with a ceiling. True if it finished (and any exception it carried - /// is rethrown), false if the ceiling was hit first. - /// - private static async Task CompletesWithinAsync(Task work, TimeSpan limit, CancellationToken ct) - { - if (work.IsCompleted) { await work; return true; } - - using var cts = CancellationTokenSource.CreateLinkedTokenSource(ct); - var timer = Task.Delay(limit, cts.Token); - var winner = await Task.WhenAny(work, timer); - cts.Cancel(); // stop the timer whichever way this went, so it cannot outlive the call - - if (winner != work) - { - ct.ThrowIfCancellationRequested(); - return false; - } - - await work; // observe a flush failure as an exception rather than a silent success - return true; - } - - /// Queue a task's (usually terminal) status — non-blocking, coalesced, flushed by the drain loop. - /// Disabled → the original fire-and-forget per-task write. - public void QueueTask(string runName, OrchestratorTaskItem task) - { - if (!_enabled) { _ = _store.UpsertTaskAsync(runName, task); return; } - lock (_lock) { _pendingTasks[Key(runName, task.Id)] = Snap(runName, task); } - Signal(); - } - - /// Queue a run's status — non-blocking, coalesced, flushed by the drain loop. - public void QueueRun(OrchestratorRun run) - { - if (!_enabled) { _ = _store.UpsertRunAsync(run); return; } - lock (_lock) { _pendingRuns[run.Name] = run; } - Signal(); - } - - /// - /// Try to coalesce a task result. Returns true if it was queued (written before this task's terminal - /// marker in the next flush, so it is durable before the task is counted done); false if the caller - /// must write it directly — because batching or result-batching is off, or the result is too large - /// for a single table property and needs the chunked path. - /// - public bool TryQueueResult(string runName, string taskId, string resultJson) - { - if (!_enabled || !_batchResults || resultJson.Length > SmallResultMaxChars) return false; - lock (_lock) { _pendingResults[Key(runName, taskId)] = new ResultWrite(runName, taskId, resultJson); } - Signal(); - return true; - } - - /// Flush all currently-pending writes and await their persistence. Call before finalizing a run - /// so terminal task states + run state are durable before post-execution reads results. - public async Task FlushAsync(CancellationToken ct = default) - { - if (!_enabled) return; - Task barrier; - lock (_lock) { barrier = _barrier.Task; } - Signal(); - - // Bounded for the same reason as the Running marker: this is awaited by FinalizeRunAsync, and an - // unbounded wait meant no run could finalize while a flush was stuck. Requeue-on-failure means a - // timeout here does not lose the writes — they persist on a later flush or on the shutdown drain, - // so finalization proceeds rather than blocking on transient storage trouble. - try - { - if (await CompletesWithinAsync(barrier, _barrierTimeout, ct)) return; - _logger.LogWarning("[Orchestrator] Flush barrier not met within {Sec}s — pending writes remain queued", - _barrierTimeout.TotalSeconds); - } - catch (Exception ex) when (ex is not OperationCanceledException) - { - _logger.LogWarning(ex, "[Orchestrator] Flush barrier reported a failed write — requeued for retry"); - } - } - - private async Task DrainLoopAsync() - { - while (!_cts.IsCancellationRequested) - { - try { await _signal.WaitAsync(_flushIntervalMs, _cts.Token); } - catch (OperationCanceledException) { break; } - - // FlushOnceAsync used to be called outside any try. Anything escaping it — an OOM while - // formatting the error log is enough — killed this loop silently, and a dead loop means no - // barrier is ever completed again, so every task hangs at its durable marker forever. - // This loop must not be able to die while the process lives. - try { await FlushOnceAsync(_flushTimeout); } - catch (Exception ex) { LogSafely(ex, "status flush"); } - } - - // Final drain on shutdown. Deliberately NOT bounded by _cts (which is already cancelled) and - // given a generous ceiling: this is the last chance for terminal task states to reach storage. - try { await FlushOnceAsync(TimeSpan.FromSeconds(30), ignoreShutdown: true); } - catch (Exception ex) { LogSafely(ex, "final status drain"); } - } - - /// Log without letting the logger itself take the drain loop down (it allocates, and this - /// path runs under exactly the memory pressure that makes allocation fail). - private void LogSafely(Exception ex, string what) - { - try { _logger.LogError(ex, "[Orchestrator] Unhandled error during {What} — drain loop continues", what); } - catch { /* nothing useful left to do; staying alive matters more than reporting */ } - } - - private async Task FlushOnceAsync(TimeSpan timeout, bool ignoreShutdown = false) - { - Dictionary tasks; - Dictionary runs; - Dictionary results; - TaskCompletionSource done; - lock (_lock) - { - done = _barrier; - _barrier = new(TaskCreationOptions.RunContinuationsAsynchronously); - if (_pendingTasks.Count == 0 && _pendingRuns.Count == 0 && _pendingResults.Count == 0) - { - // Nothing to flush — still release barrier waiters (e.g. FlushAsync on an already-drained run). - done.TrySetResult(); - return; - } - tasks = _pendingTasks; _pendingTasks = new(); - runs = _pendingRuns; _pendingRuns = new(); - results = _pendingResults; _pendingResults = new(); - } - - using var cts = ignoreShutdown - ? new CancellationTokenSource(timeout) - : CancellationTokenSource.CreateLinkedTokenSource(_cts.Token); - if (!ignoreShutdown) cts.CancelAfter(timeout); - - var unwritten = new List(); - Exception? failure = null; - - try - { - // Results FIRST, and a run whose result did not land withholds that run's terminal task and - // run-status writes THIS flush. A terminal marker must never persist ahead of its result: a - // crash in between would leave a task counted done with no result for post-execution to read. - var failedResultRuns = new HashSet(StringComparer.Ordinal); - if (results.Count > 0) - { - var failedResults = await _store.WriteResultBatchAsync(results.Values.ToList(), _flushConcurrency, cts.Token); - foreach (var run in failedResults) failedResultRuns.Add(run); - unwritten.AddRange(failedResults); - } - - if (tasks.Count > 0) - { - var toWrite = failedResultRuns.Count == 0 - ? tasks.Values.ToList() - : tasks.Values.Where(t => !failedResultRuns.Contains(t.RunName)).ToList(); - unwritten.AddRange(await _store.WriteTaskStatusBatchAsync(toWrite, _flushConcurrency, cts.Token)); - // Tasks held back because their run's result failed: requeue so they retry with the result. - if (failedResultRuns.Count > 0) - unwritten.AddRange(tasks.Values.Where(t => failedResultRuns.Contains(t.RunName)).Select(t => t.RunName)); - } - - // Batched, not one await per run. Run rows all share the "Run" partition key, so N of them - // cost ceil(N/100) transactions; the previous per-run loop cost N round-trips inside a flush - // bounded by _flushTimeout, which is how a busy instance (100+ live runs) blew the flush - // budget and then the durable barrier. The store isolates a poison row by retrying a failed - // chunk individually, so a single bad run no longer discards the other 99. - if (runs.Count > 0) - { - var toWrite = failedResultRuns.Count == 0 - ? runs.Values.ToList() - : runs.Values.Where(r => !failedResultRuns.Contains(r.Name)).ToList(); - unwritten.AddRange(await _store.WriteRunStatusBatchAsync(toWrite, cts.Token)); - if (failedResultRuns.Count > 0) - unwritten.AddRange(runs.Values.Where(r => failedResultRuns.Contains(r.Name)).Select(r => r.Name)); - } - } - catch (Exception ex) - { - // Timed out or the store threw wholesale — treat EVERYTHING in this snapshot as unwritten. - failure = ex; - unwritten.AddRange(results.Values.Select(r => r.RunName)); - unwritten.AddRange(tasks.Values.Select(t => t.RunName)); - unwritten.AddRange(runs.Keys); - _logger.LogError(ex, "[Orchestrator] Status flush failed ({Results} results, {Tasks} tasks, {Runs} runs) — requeued for retry", - results.Count, tasks.Count, runs.Count); - } - - // Durability: anything that did not reach storage goes back on the pending set. Dropping it - // (the previous behaviour on any exception) silently lost terminal task states — a completed - // task would look Pending forever and be re-run by the next recovery. - if (unwritten.Count > 0) Requeue(results, tasks, runs, unwritten); - - // The barrier may ONLY report success when this batch actually persisted. Waiters cannot tell - // which run in the batch was theirs, so any un-persisted write has to fail all of them — a - // waiter that returns successfully goes on to run its task, and running a task whose "Running" - // marker never landed is exactly the poison-retry hole the marker exists to close. Failing here - // is cheap: the writes are requeued, and the waiters defer and retry. - if (failure != null) - done.TrySetException(failure); - else if (unwritten.Count > 0) - done.TrySetException(new InvalidOperationException( - $"{unwritten.Count} status write(s) did not persist and were requeued")); - else - done.TrySetResult(); - } - - /// - /// Put un-persisted writes back. , not assignment: a - /// NEWER state for the same task may have arrived while the flush was in flight, and the retry of a - /// stale snapshot must never overwrite it. - /// - private void Requeue(Dictionary results, Dictionary tasks, - Dictionary runs, List unwrittenRuns) - { - var retry = new HashSet(unwrittenRuns, StringComparer.OrdinalIgnoreCase); - var restored = 0; - - lock (_lock) - { - // Results before tasks, mirroring the flush order: a result put back must be in the pending - // set before the terminal marker it guards is retried. - foreach (var (key, write) in results) - { - if (!retry.Contains(write.RunName)) continue; - if (_pendingResults.TryAdd(key, write)) restored++; - } - foreach (var (key, write) in tasks) - { - if (!retry.Contains(write.RunName)) continue; - if (_pendingTasks.TryAdd(key, write)) restored++; - } - foreach (var (name, run) in runs) - { - if (!retry.Contains(name)) continue; - if (_pendingRuns.TryAdd(name, run)) restored++; - } - } - - if (restored > 0) - { - Signal(); // make sure the next flush actually runs - _logger.LogWarning("[Orchestrator] Requeued {Count} un-persisted status writes across {Runs} runs", - restored, retry.Count); - } - } - - /// - /// Wake the drain loop. Guarded so a status enqueue that races — an in-flight - /// job finishing after shutdown began — drops the wake instead of throwing ObjectDisposedException. - /// By the time a task queues its terminal status its data is already persisted, so a lost coalesced - /// wake on the way down is harmless; a task falsely marked Failed because the release threw is not. - /// - private void Signal() - { - if (_disposed) return; - try { _signal.Release(); } - catch (ObjectDisposedException) { /* shutting down; the drain loop has already stopped */ } - } - - public void Dispose() - { - _disposed = true; - _cts.Cancel(); - try { _drainLoop.Wait(TimeSpan.FromSeconds(5)); } catch { /* best effort final drain */ } - _cts.Dispose(); - _signal.Dispose(); - } -} diff --git a/Services/Orchestration/OrchestratorTaskItem.cs b/Services/Orchestration/OrchestratorTaskItem.cs index 3df1994..59a1bf5 100644 --- a/Services/Orchestration/OrchestratorTaskItem.cs +++ b/Services/Orchestration/OrchestratorTaskItem.cs @@ -1,36 +1,9 @@ namespace Craft.Orchestration; +/// A task parsed from a batch or planner output, before it is stored. public class OrchestratorTaskItem { public string Id { get; set; } = string.Empty; public string Status { get; set; } = "Pending"; public Dictionary Parameters { get; set; } = []; - public int AttemptCount { get; set; } - public string? LastError { get; set; } - public DateTime? CompletedUtc { get; set; } - - /// - /// Dispatch priority override for this task alone. Null — the normal case — means "inherit the - /// run's priority", so existing rows and everything the planner emits behave exactly as before - /// with no backfill needed. - /// - /// Set only when an operator reprioritizes one queued job (JobManager.ChangePriority). - /// Persisted so the override survives a restart, instead of silently reverting to the run's - /// priority when ResumeInterruptedRunsAsync re-queues the task. - /// - public int? Priority { get; set; } - - /// - /// Position of this task in the batch as submitted (0-based). Only meaningful for a run marked - /// , where tasks are dispatched one at a time in ascending - /// Sequence order. Non-sequential runs leave it 0 and ignore it. - /// - public int Sequence { get; set; } - - /// - /// Set when a worker in THIS process marks the task Running. Never persisted, so a Running status - /// rehydrated from storage (another process's pre-invoke marker) reads false: it is not proof that - /// anything here is executing the task. See OrchestratorService.ResolveTaskWorkAsync. - /// - internal bool OwnedHere; } diff --git a/Services/Orchestration/WorkPump.cs b/Services/Orchestration/WorkPump.cs new file mode 100644 index 0000000..cf38b61 --- /dev/null +++ b/Services/Orchestration/WorkPump.cs @@ -0,0 +1,195 @@ +using Craft.Configuration; +using Craft.Storage; + +namespace Craft.Orchestration; + +/// +/// Keeps the JobManager's buffer topped up from storage. Each refill reads the Ready list (best band first, +/// oldest run first), claims from those runs' partitions until the batch is full, and hands the claims to the +/// JobManager as descriptors. A run with nothing claimable is skipped for a while rather than read every tick. +/// +/// The pump holds no state that matters after a crash: claims it held lapse and are claimed again, by this +/// process or any other. +/// +public class WorkPump : BackgroundService +{ + private readonly ILogger _logger; + private readonly WorkStore _store; + private readonly JobManager _jobs; + private readonly OrchestratorService? _orchestrator; + private readonly string _owner; + private readonly int _batchSize; + private readonly int _lowWater; + private readonly TimeSpan _lease; + private readonly TimeSpan _pollInterval; + private readonly TimeSpan _idlePollInterval; + private readonly Task? _claimGate; + + /// How long a run that had nothing claimable is skipped while its counts stand still, and how often a + /// run's lapsed leases are looked for. + private static readonly TimeSpan EmptyBackoff = TimeSpan.FromSeconds(30); + private static readonly TimeSpan ExpiredCheckInterval = TimeSpan.FromSeconds(60); + + /// Runs read from storage per refill. Skipped runs cost nothing, so a head of runs that are all waiting + /// (on children, or on their own running tasks) cannot hide the runs behind it. + private const int MaxRunsReadPerRefill = 32; + private const int ReadyPageSize = 100; + + private readonly Dictionary _emptyUntil = new(StringComparer.Ordinal); + private readonly Dictionary _expiredCheckedAt = new(StringComparer.Ordinal); + + /// Claims handed to the JobManager, by job id, with when their lease runs out. + private readonly Dictionary _inFlight = new(StringComparer.Ordinal); + + public WorkPump(ILogger logger, WorkStore store, JobManager jobs, IConfiguration configuration, + CraftSettings settings, OrchestratorService? orchestrator = null) + { + _logger = logger; + _store = store; + _jobs = jobs; + _orchestrator = orchestrator; + _claimGate = orchestrator?.RecoveryDone; + _owner = orchestrator?.Owner ?? Environment.GetEnvironmentVariable("HOSTNAME") ?? $"instance-{Environment.ProcessId}"; + _batchSize = Math.Max(1, configuration.GetValue("JobQueueBatchSize", Math.Max(1, settings.Worker.BgPoolSize))); + _lowWater = Math.Max(0, configuration.GetValue("JobQueueLowWaterMark", 2)); + _lease = orchestrator?.Lease ?? TimeSpan.FromSeconds(Math.Max(60, configuration.GetValue("JobQueueLeaseSeconds", 1800))); + _pollInterval = TimeSpan.FromMilliseconds(Math.Max(100, configuration.GetValue("JobQueuePollIntervalMs", 1000))); + _idlePollInterval = TimeSpan.FromMilliseconds(Math.Max(_pollInterval.TotalMilliseconds, + configuration.GetValue("JobQueueIdlePollIntervalMs", 10_000))); + } + + protected override async Task ExecuteAsync(CancellationToken stoppingToken) + { + _logger.LogInformation("[WorkPump] Started: owner={Owner} batch={Batch} lowWater={Low} lease={Lease}s", + _owner, _batchSize, _lowWater, _lease.TotalSeconds); + + if (_claimGate is { IsCompleted: false }) + { + try { await _claimGate.WaitAsync(stoppingToken); } + catch (OperationCanceledException) { return; } + } + + var idleTicks = 0; + while (!stoppingToken.IsCancellationRequested) + { + var claimed = 0; + try + { + Forget(); + claimed = await RefillAsync(stoppingToken); + await RenewAsync(stoppingToken); + } + catch (OperationCanceledException) when (stoppingToken.IsCancellationRequested) { break; } + catch (Exception ex) + { + try { _logger.LogError(ex, "[WorkPump] Cycle failed; continuing"); } catch { } + } + + idleTicks = claimed > 0 || _inFlight.Count > 0 ? 0 : idleTicks + 1; + var delay = idleTicks == 0 + ? _pollInterval + : TimeSpan.FromMilliseconds(Math.Min(_idlePollInterval.TotalMilliseconds, + _pollInterval.TotalMilliseconds * (1L << Math.Min(idleTicks, 20)))); + try { await Task.Delay(delay, stoppingToken); } + catch (OperationCanceledException) { break; } + } + } + + internal void ForgetBackoff() => _emptyUntil.Clear(); + + /// Stop tracking claims the JobManager is done with. Their finish (or release) was written by the job. + private void Forget() + { + foreach (var id in _inFlight.Keys.Where(id => !_jobs.IsQueuedOrRunning(id)).ToList()) + _inFlight.Remove(id); + } + + /// Claim until the buffer holds a batch. Returns how many tasks were claimed. + internal async Task RefillAsync(CancellationToken ct) + { + if (_jobs.QueuedCount > _lowWater) return 0; + var need = _batchSize - _jobs.QueuedCount; + var claimed = 0; + var now = DateTime.UtcNow; + var read = 0; + + await foreach (var entry in _store.ReadReadyAsync(ReadyPageSize, ct)) + { + if (need <= 0 || read >= MaxRunsReadPerRefill) break; + if (_emptyUntil.TryGetValue(entry.RunKey, out var skip) && skip.Until > now + && skip.Done == entry.Done && skip.Total == entry.Total) continue; + read++; + + var header = await _store.GetRunAsync(entry.RunKey, ct); + if (header == null || header.IsFinished) + { + await _store.DropReadyAsync(entry, ct); + continue; + } + + IReadOnlyList claims; + if (header.Sequential) + { + var step = await _store.ClaimSequentialAsync(header.RunKey, _owner, _lease, ct: ct); + claims = step == null ? [] : [step]; + } + else + { + var checkExpired = !_expiredCheckedAt.TryGetValue(header.RunKey, out var at) || now - at >= ExpiredCheckInterval; + if (checkExpired) _expiredCheckedAt[header.RunKey] = now; + claims = await _store.ClaimAsync(header.RunKey, need, _owner, _lease, checkExpired, ct); + } + + if (claims.Count == 0) + { + _emptyUntil[header.RunKey] = (now + EmptyBackoff, entry.Done, entry.Total); + continue; + } + _emptyUntil.Remove(header.RunKey); + + foreach (var c in claims) + { + var name = c.Seq == WorkStore.AggregateSeq ? $"{header.Name}-PostExec" : $"{header.Name}-{c.TaskId}"; + var descriptor = new JobDescriptor(header.Name, c.TaskId, header.Priority) { RunKey = c.RunKey, Seq = c.Seq, Attempt = c.Attempt }; + var jobId = _jobs.Enqueue(descriptor, name); + _inFlight[jobId] = (c, now + _lease); + } + need -= claims.Count; + claimed += claims.Count; + } + + if (_emptyUntil.Count > 10_000) + foreach (var key in _emptyUntil.Where(kv => kv.Value.Until <= now).Select(kv => kv.Key).ToList()) + { + _emptyUntil.Remove(key); + _expiredCheckedAt.Remove(key); + } + + return claimed; + } + + /// Renew claims in their last third, so a long buffer wait or a long task never loses its lease. + private async Task RenewAsync(CancellationToken ct) + { + var now = DateTime.UtcNow; + var due = _inFlight.Where(kv => kv.Value.LeaseUntil - now < _lease / 3).ToList(); + if (due.Count == 0) return; + + var lost = await _store.RenewAsync(due.Select(kv => kv.Value.Claim).ToList(), _owner, _lease, ct); + if (lost.Count > 0) + _logger.LogWarning("[WorkPump] {Count} claim(s) were taken back before renewal — their leases had lapsed", lost.Count); + foreach (var (id, v) in due) _inFlight[id] = (v.Claim, now + _lease); + } + + /// On shutdown, hand back claims that never started so another process can run them now. + public override async Task StopAsync(CancellationToken cancellationToken) + { + await base.StopAsync(cancellationToken); + foreach (var (id, v) in _inFlight.ToList()) + { + if (_jobs.GetJobs().FirstOrDefault(j => j.Id == id) is not { Status: "Queued" }) continue; + try { await _store.ReleaseAsync(v.Claim.RunKey, v.Claim.Seq, _owner, refundAttempt: true, cancellationToken); } + catch (Exception ex) { _logger.LogDebug(ex, "[WorkPump] Could not release {Job} on shutdown", id); } + } + } +} diff --git a/Services/Storage/AzureTableStore.cs b/Services/Storage/AzureTableStore.cs index 9072934..fca5a91 100644 --- a/Services/Storage/AzureTableStore.cs +++ b/Services/Storage/AzureTableStore.cs @@ -270,6 +270,45 @@ public async Task TryReplaceBatchAsync(string table, string partitionKey, return true; } + public async Task DeleteTableAsync(string table, CancellationToken ct = default) + { + try { await Service.DeleteTableAsync(table, ct); } + catch (RequestFailedException ex) when (ex.Status == 404) { } + } + + public async Task TrySubmitAsync(string table, string partitionKey, IReadOnlyList ops, + CancellationToken ct = default) + { + if (ops.Count == 0) return true; + if (ops.Count > MaxBatch) + throw new ArgumentException($"A transaction is limited to {MaxBatch} ops, got {ops.Count}.", nameof(ops)); + + var actions = ops.Select(op => op.Kind switch + { + StoreOpKind.Insert => new TableTransactionAction(TableTransactionActionType.Add, ToEntity(op.Row)), + StoreOpKind.Replace => new TableTransactionAction(TableTransactionActionType.UpdateReplace, ToEntity(op.Row), + new ETag(op.Row.ETag ?? throw new ArgumentException($"Replace of {op.Row.RowKey} needs an ETag.", nameof(ops)))), + StoreOpKind.Delete => new TableTransactionAction(TableTransactionActionType.Delete, + new TableEntity(op.Row.PartitionKey, op.Row.RowKey), op.Row.ETag is { } etag ? new ETag(etag) : ETag.All), + _ => new TableTransactionAction(TableTransactionActionType.UpsertReplace, ToEntity(op.Row)), + }).ToList(); + + try + { + await Client(table).SubmitTransactionAsync(actions, ct); + return true; + } + catch (RequestFailedException ex) when (IsTableNotFound(ex)) + { + await RecreateTableAsync(table, ct); + return false; + } + catch (RequestFailedException ex) when (ex.Status is 412 or 404 or 409) + { + return false; + } + } + private async Task SubmitAsync(string table, TableClient client, List batch, CancellationToken ct) { try diff --git a/Services/Storage/ICraftTableStore.cs b/Services/Storage/ICraftTableStore.cs index 436a03d..19096d1 100644 --- a/Services/Storage/ICraftTableStore.cs +++ b/Services/Storage/ICraftTableStore.cs @@ -103,6 +103,40 @@ async IAsyncEnumerable QueryRowKeyRangeAsync(string table, string part yield return row; } + /// + /// Apply to one partition as a single all-or-nothing transaction (at most 100 + /// ops, rows never split). Insert fails if the row exists; Replace and a Delete carrying an ETag fail if + /// the row changed or is gone. Returns false, with nothing written, when any guard fails. + /// + /// This default checks every guard and then applies the ops one by one, which is atomic only for a + /// single-threaded caller; submits a real transaction. + /// + async Task TrySubmitAsync(string table, string partitionKey, IReadOnlyList ops, + CancellationToken ct = default) + { + foreach (var op in ops) + { + var current = await GetAsync(table, op.Row.PartitionKey, op.Row.RowKey, ct); + var ok = op.Kind switch + { + StoreOpKind.Insert => current == null, + StoreOpKind.Replace => current != null && current.ETag == op.Row.ETag, + StoreOpKind.Delete => op.Row.ETag == null || (current != null && current.ETag == op.Row.ETag), + _ => true, + }; + if (!ok) return false; + } + foreach (var op in ops) + { + if (op.Kind == StoreOpKind.Delete) await DeleteAsync(table, op.Row.PartitionKey, op.Row.RowKey, ct); + else await UpsertAsync(table, op.Row, ct); + } + return true; + } + + /// Delete a whole table if it exists. The default does nothing. + Task DeleteTableAsync(string table, CancellationToken ct = default) => Task.CompletedTask; + /// Delete a single row. A missing row is not an error. Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default); @@ -124,3 +158,15 @@ async Task DeleteBatchAsync(string table, string partitionKey, IReadOnlyListDelete every row in a partition. Task DeletePartitionAsync(string table, string partitionKey, CancellationToken ct = default); } + +public enum StoreOpKind { Insert, Replace, Delete, Upsert } + +/// One write in a transaction. Replace and Delete are +/// guarded by (a Delete without one is unconditional). +public readonly record struct StoreOp(StoreOpKind Kind, StoreRow Row) +{ + public static StoreOp Insert(StoreRow row) => new(StoreOpKind.Insert, row); + public static StoreOp Replace(StoreRow row) => new(StoreOpKind.Replace, row); + public static StoreOp Upsert(StoreRow row) => new(StoreOpKind.Upsert, row); + public static StoreOp Delete(StoreRow row) => new(StoreOpKind.Delete, row); +} diff --git a/Services/Storage/JobQueueStore.cs b/Services/Storage/JobQueueStore.cs deleted file mode 100644 index de4c432..0000000 --- a/Services/Storage/JobQueueStore.cs +++ /dev/null @@ -1,972 +0,0 @@ -using System.Globalization; -using Craft.Configuration; - -namespace Craft.Storage; - -/// -/// The durable job queue: one row per queued task, claimed in batches under a lease. -/// -/// This exists so the dispatch side can hold a worker-pool-sized buffer instead of the whole backlog, -/// and so an edit to a queued task in storage is what actually runs — the in-memory copy is a buffer, -/// not the truth. -/// -/// Key design, and it is doing real work: -/// -/// PartitionKey "P04" — the priority bucket, zero-padded so it sorts numerically. -/// RowKey "{runEpochTicks:D19}|{run}|{task}" — the run's start time first, so a bucket drains oldest run -/// first, and one row per task because a run's start never -/// changes. A run's rows in a bucket are also contiguous, so -/// run-scoped queue reads are one RowKey range. -/// -/// Azure Table returns rows ordered by partition key then row key, so a single read yields the -/// highest-priority, oldest-run work FIRST, across every run, for one round-trip. -/// -/// A claim is one conditional transaction over rows sharing a bucket: one round-trip per BATCH, not per -/// task. Guarded by each row's ETag, so an external edit — or another instance claiming first — makes -/// the claim fail rather than silently overwrite, and the caller re-reads. -/// -/// THE RUN INDEX, and why the queue table alone is not enough: -/// -/// The key design above is right for the claim path and wrong for everything else. RunName is not part -/// of the key, so every run-scoped read — "which tasks does this run still have queued", "release this -/// run's claims", "drop this run's rows" — can only be answered by scanning every partition. Those run -/// on hot paths: once per finalize, once per resumed run, once per orphan re-drive. -/// -/// Measured on a production instance whose queue reached ~743,000 rows: 61,939 such scans in ten -/// minutes, p50 80 seconds, p95 100 seconds, then TaskCanceledException. The timeouts fell on the task -/// status writes, so tasks never reached a terminal state, so runs never finalized, so their rows were -/// never deleted — and the table that made the scans slow could only grow. 65% of the log file was the -/// resulting HTTP-SLOW warnings; 0.3% was actual task execution. -/// -/// A server-side $filter does not fix this and previously appeared to: filtering on a non-key property -/// narrows what crosses the wire, not what the backend reads. The scan is the cost. -/// -/// So run-scoped access gets its own index table, keyed the way those reads actually ask: -/// -/// PartitionKey the run name, escaped for the key charset -/// RowKey "{bucket}|{queue row key}" — enough to address the queue row directly -/// -/// Every run-scoped method below is now a single-partition read followed by point operations, and the -/// claim path is untouched. The index is maintained by the enqueue/remove paths, and built (and, from -/// schema v2, re-keyed) once for a pre-existing queue by . -/// -public sealed class JobQueueStore : IDisposable -{ - private readonly ILogger _logger; - private readonly ICraftTableStore _store; - private readonly string _queueTable; - private readonly string _indexTable; - private readonly string _runsTable; - private bool _initialized; - - /// - /// Wakes the the moment claimable rows appear, instead - /// of it discovering them only on its next poll tick. Bounded at one pending permit: many enqueues - /// between two pump cycles coalesce into a single wake, because one refill claims a whole batch anyway. - /// The pump keeps polling on its (idle-backing-off) interval as the backstop — this signal only - /// removes the wait, it does not replace the loop. Same-instance only, which is all that is needed: - /// the pump and the enqueue paths share this singleton, and a lease keeps cross-instance work safe. - /// - private readonly SemaphoreSlim _pumpWake = new(0, 1); - - /// Signal the pump that new claimable rows exist. Never throws and never exceeds one permit. - private void WakePump() - { - try { _pumpWake.Release(); } - catch (SemaphoreFullException) { /* a wake is already pending; the pump will claim the batch */ } - catch (ObjectDisposedException) { /* shutting down; the pump loop has already stopped */ } - } - - public void Dispose() => _pumpWake.Dispose(); - - /// - /// Block until the pump is woken by an enqueue or elapses, whichever - /// comes first. Returns true if woken (new work signalled), false on the poll timeout. - /// - public Task WaitForWorkAsync(TimeSpan pollInterval, CancellationToken ct = default) => - _pumpWake.WaitAsync(pollInterval, ct); - - /// Priorities above this share the lowest bucket. Callers use 0-6; the cap only bounds the key. - private const int MaxPriorityBucket = 99; - - /// Width of the zero-padded bucket key ("P04"), so the index row key splits at a fixed offset. - private const int BucketKeyLength = 3; - - /// - /// Where the schema marker lives. '$' is legal in a key and no run name starts with it — run names - /// are "{OrchestratorName}-{tenant}-{guid}" or "{OrchestratorName}_{...}". - /// - private const string SchemaPartition = "$schema"; - private const string SchemaRowKey = "queue-index"; - - /// - /// Current on-disk schema version, applied once per storage account by : - /// 1 — the run index exists (see the RUN INDEX note above). - /// 2 — queue RowKeys are deterministic per (run, task) — {run}|{task} — instead of - /// time-prefixed, so re-dispatching a task UPDATES its row instead of writing a second one - /// (the duplicate-execution class). Enqueue time moves to the QueuedUtc property. - /// 3 — the key gains the run's start time as a prefix — {epoch}|{run}|{task} — so a bucket - /// drains oldest run first instead of alphabetically by run name, which starved late-sorting runs. - /// A single forward migration takes any older account straight to this version; there is no - /// backward-compatible dual-read — after the migration only the new key scheme is used. - /// - private const int SchemaVersion = 3; - - /// - /// Rows buffered before the backfill flushes. Bounds peak memory on a very large queue — the - /// instance that motivated this was at 85% of a 2398MB heap cap before the backfill even started, - /// and buffering 743,000 rows to group them perfectly would have been the thing that OOMed it. - /// Rows for one run are written adjacently, so a window this size still groups them in practice. - /// - private const int BackfillFlushThreshold = 5_000; - - /// Partition writes in flight at once during the migration. - private const int MigrationConcurrency = 16; - - public JobQueueStore(ILogger logger, CraftSettings settings, ICraftTableStore store) - { - _logger = logger; - _store = store; - _queueTable = $"{settings.Orchestrator.TablePrefix}Queue"; - _indexTable = $"{settings.Orchestrator.TablePrefix}QueueIndex"; - _runsTable = $"{settings.Orchestrator.TablePrefix}Runs"; - } - - public async Task InitializeAsync(CancellationToken ct = default) - { - if (_initialized) return; - await _store.EnsureTableAsync(_queueTable, ct); - await _store.EnsureTableAsync(_indexTable, ct); - await MigrateSchemaAsync(ct); - _initialized = true; - } - - internal static string Bucket(int priority) => - "P" + Math.Clamp(priority, 0, MaxPriorityBucket).ToString("D2", CultureInfo.InvariantCulture); - - /// - /// The queue RowKey. The epoch is the run's start time, which every caller reads off the run, so the - /// key is the same on every enqueue of a task and a re-dispatch upserts its one row. - /// - internal static string BuildRowKey(DateTime runStartedUtc, string runName, string taskId) => - $"{RunKeyPrefix(runStartedUtc, runName)}{EscapeKeyComponent(taskId)}"; - - /// The shared start of every queue key of one run: {epoch}|{run}|. - private static string RunKeyPrefix(DateTime runStartedUtc, string runName) => - $"{EpochOf(runStartedUtc).Ticks.ToString("D19", CultureInfo.InvariantCulture)}|{EscapeKeyComponent(runName)}|"; - - /// The run prefix of a v3 queue key, or null for an older key. - private static string? RunKeyPrefixOf(string rowKey) - { - if (ParseEpoch(rowKey) == null) return null; - var end = rowKey.IndexOf('|', 20); - return end < 0 ? null : rowKey[..(end + 1)]; - } - - /// Everything that starts with : '|' is followed by '}' in ordinal order. - private static string PrefixUpperBound(string prefix) => prefix[..^1] + '}'; - - /// Whole seconds, so an epoch builds the same key before and after a storage round trip. - internal static DateTime EpochOf(DateTime utc) => - new(utc.Ticks - utc.Ticks % TimeSpan.TicksPerSecond, DateTimeKind.Utc); - - /// The run epoch a v3 key starts with, or null for an older key. - internal static DateTime? ParseEpoch(string rowKey) - { - if (rowKey.Length < 21 || rowKey[19] != '|') return null; - if (!long.TryParse(rowKey.AsSpan(0, 19), NumberStyles.None, CultureInfo.InvariantCulture, out var ticks)) - return null; - if (ticks <= 0 || ticks > DateTime.MaxValue.Ticks) return null; - return new DateTime(ticks, DateTimeKind.Utc); - } - - /// - /// Escape a run or task id for use inside a queue RowKey: the Azure-illegal key characters plus '|' - /// (the separator) and '%' (the escape marker itself), percent-encoded. Reversible and injective, so - /// distinct (run, task) pairs never collide, and ordinary names pass through untouched. - /// - private static string EscapeKeyComponent(string value) - { - var needsEscape = false; - foreach (var c in value) - { - if (c is '/' or '\\' or '#' or '?' or '%' or '|' || char.IsControl(c)) { needsEscape = true; break; } - } - if (!needsEscape) return value; - - var sb = new System.Text.StringBuilder(value.Length + 8); - foreach (var c in value) - { - if (c is '/' or '\\' or '#' or '?' or '%' or '|' || char.IsControl(c)) - sb.Append('%').Append(((int)c).ToString("X2", CultureInfo.InvariantCulture)); - else - sb.Append(c); - } - return sb.ToString(); - } - - /// The enqueue time embedded in a legacy (v1) row key — {ticks:D19}-{run}-{task} — or - /// null for a key that is not in that format. Used only by the one-time migration to carry the old - /// key's timestamp into the new row's QueuedUtc property. - internal static DateTime? ParseLegacyQueuedUtc(string rowKey) - { - if (rowKey.Length < 20 || rowKey[19] != '-') return null; - if (!long.TryParse(rowKey.AsSpan(0, 19), NumberStyles.None, CultureInfo.InvariantCulture, out var ticks)) - return null; - if (ticks <= 0 || ticks > DateTime.MaxValue.Ticks) return null; - return new DateTime(ticks, DateTimeKind.Utc); - } - - /// - /// A run name as an index partition key. Azure Tables rejects '/', '\', '#', '?' and control - /// characters in a key, and a run name carries a user-supplied scheduled-task name — "Alert on - /// Huntress Rogue Apps detected" is a real one, and nothing stops the next one containing a slash - /// or a question mark. Percent-escaping is reversible and leaves ordinary names untouched, so the - /// table stays readable in the portal, which is where anyone debugging this will be looking. - /// - internal static string IndexPartition(string runName) - { - var needsEscape = false; - foreach (var c in runName) - { - if (c is '/' or '\\' or '#' or '?' or '%' || char.IsControl(c)) { needsEscape = true; break; } - } - if (!needsEscape) return runName; - - var sb = new System.Text.StringBuilder(runName.Length + 8); - foreach (var c in runName) - { - // '%' first, or the escape sequences themselves would be ambiguous. - if (c is '/' or '\\' or '#' or '?' or '%' || char.IsControl(c)) - sb.Append('%').Append(((int)c).ToString("X2", CultureInfo.InvariantCulture)); - else - sb.Append(c); - } - return sb.ToString(); - } - - /// Index row key: the bucket (fixed width) plus the queue row key it points at. - internal static string IndexRowKey(string bucket, string queueRowKey) => $"{bucket}|{queueRowKey}"; - - /// The inverse of . Split at a fixed offset — the bucket is always - /// three characters, so a '|' inside the queue row key cannot confuse this. - internal static (string Bucket, string QueueRowKey)? SplitIndexRowKey(string indexRowKey) - { - if (indexRowKey.Length < BucketKeyLength + 2 || indexRowKey[BucketKeyLength] != '|') return null; - return (indexRowKey[..BucketKeyLength], indexRowKey[(BucketKeyLength + 1)..]); - } - - private static StoreRow IndexRow(string runName, string taskId, string bucket, string queueRowKey) => - new(IndexPartition(runName), IndexRowKey(bucket, queueRowKey)) - { - Properties = { ["TaskId"] = taskId, ["RunName"] = runName } - }; - - /// Add one task to the queue. Idempotent per (run, task). - public Task EnqueueAsync(string runName, string taskId, int priority, DateTime runStartedUtc, - CancellationToken ct = default) => - EnqueueBatchAsync(runName, [(taskId, priority)], runStartedUtc, ct); - - /// Queue tasks of one run, keyed by the run's start time and grouped into per-bucket batches. - /// - /// Queue rows first, index rows second. The queue row is what makes the task actually run; the index - /// only accelerates lookups. If the process dies between the two the task still executes, and the - /// missing index entry is repaired by the next enqueue of the same task, which rewrites both keys - /// unchanged. The other order would leave the index claiming a task is queued when no row exists — the - /// orphan re-drive trusts the index, would decline to re-queue, and the run would sit Pending with - /// nothing running. The index rows share the run's partition, so they cost one transaction. - /// - public async Task EnqueueBatchAsync(string runName, IReadOnlyList<(string TaskId, int Priority)> tasks, - DateTime runStartedUtc, CancellationToken ct = default) - { - if (tasks.Count == 0) return; - - var indexRows = new List(tasks.Count); - var queuedOffset = new DateTimeOffset(DateTime.SpecifyKind(runStartedUtc, DateTimeKind.Utc)); - - foreach (var byBucket in tasks.GroupBy(t => Bucket(t.Priority))) - { - var rows = byBucket.Select(t => new StoreRow(byBucket.Key, BuildRowKey(runStartedUtc, runName, t.TaskId)) - { - Properties = - { - ["RunName"] = runName, - ["TaskId"] = t.TaskId, - ["Priority"] = t.Priority, - ["Owner"] = "", - ["LeaseUntil"] = (DateTimeOffset?)null, - ["QueuedUtc"] = queuedOffset, - } - }).ToList(); - - await _store.UpsertBatchAsync(_queueTable, byBucket.Key, rows, ct); - - indexRows.AddRange(rows.Select(r => IndexRow(runName, r.GetString("TaskId")!, byBucket.Key, r.RowKey))); - } - - await _store.UpsertBatchAsync(_indexTable, IndexPartition(runName), indexRows, ct); - - WakePump(); - } - - /// A queued task this worker now owns, with the row key needed to release it. - public sealed record ClaimedJob(string RunName, string TaskId, int Priority, string Bucket, string RowKey); - - /// - /// Claim up to of the highest-priority, oldest queued tasks for - /// , for . - /// - /// One read plus one conditional transaction. The read stops as soon as it has a batch, so it costs - /// a single page however deep the queue is; the transaction covers one bucket, because that is the - /// unit a backend transaction can span. - /// - /// Returns empty when there is nothing claimable, and ALSO when another worker won the race — the - /// caller simply tries again rather than forcing the write, which is what stops two workers running - /// the same task. - /// - public async Task> ClaimBatchAsync(string owner, int max, TimeSpan leaseFor, - CancellationToken ct = default) - { - if (max <= 0) return []; - - var now = DateTimeOffset.UtcNow; - var candidates = new List(max); - string? bucket = null; - - // Ordered partition-then-row, so this walks highest priority first, oldest first within it. - // - // The filter is the same predicate as IsClaimable, pushed to the service so a backlog is not - // paged to the client on every pump tick just to find the few free rows at its head. It is an - // optimisation ONLY — a store that ignores it still returns everything — so IsClaimable below - // stays as the authority. Nothing here may assume the filter was applied. - await foreach (var row in _store.QueryTableAsync(_queueTable, ClaimableFilter(now), max, ct)) - { - if (!IsClaimable(row, now)) continue; - - // A transaction cannot span partitions, so the batch is whatever the top bucket offers. - bucket ??= row.PartitionKey; - if (row.PartitionKey != bucket) break; - - candidates.Add(row); - if (candidates.Count == max) break; - } - - if (candidates.Count == 0) return []; - - var leaseUntil = now.Add(leaseFor); - foreach (var row in candidates) - { - row["Owner"] = owner; - row["LeaseUntil"] = leaseUntil; - } - - if (!await _store.TryReplaceBatchAsync(_queueTable, bucket!, candidates, ct)) - { - // Someone else got there first, or a row changed underneath us. Not an error: the caller - // retries and takes whatever is genuinely free. - _logger.LogDebug("[JobQueue] Claim of {Count} from {Bucket} lost the race", candidates.Count, bucket); - return []; - } - - return candidates.Select(r => new ClaimedJob( - r.GetString("RunName") ?? "", - r.GetString("TaskId") ?? "", - r.GetInt32("Priority") ?? 0, - r.PartitionKey, - r.RowKey)).ToList(); - } - - /// - /// Claimable means unowned, or owned under a lease that has expired. - /// - /// Lease expiry is what replaces the age-based re-drive: a worker that dies holding a claim gives the - /// task back on its own, without anything having to notice the worker is gone. - /// - private static bool IsClaimable(StoreRow row, DateTimeOffset now) - { - if (string.IsNullOrEmpty(row.GetString("Owner"))) return true; - - var lease = row.GetDateTimeOffset("LeaseUntil"); - return lease == null || lease <= now; - } - - /// - /// The server-side half of : free rows, plus rows whose lease has run out. - /// - /// Enqueue writes Owner as an empty string and LeaseUntil as null, and a null property is simply - /// absent from an Azure Tables entity — so a free row is matched by the Owner clause rather than by - /// anything about LeaseUntil, which is why this does not try to express "LeaseUntil is null". - /// - /// One case is deliberately narrower than IsClaimable: a row with an Owner but NO LeaseUntil, which - /// IsClaimable treats as claimable, is not matched here. No write path produces one — Owner and - /// LeaseUntil are always set together, by the claim, the renewal and the release alike — so this - /// costs nothing in practice, and IsClaimable keeps the defensive reading for anything that - /// arrives through the unfiltered path. - /// - private static string ClaimableFilter(DateTimeOffset now) => - $"Owner eq '' or LeaseUntil lt datetime'{now.UtcDateTime:yyyy-MM-ddTHH:mm:ss.fffffffZ}'"; - - /// Remove a finished task from the queue. A missing row is not an error — it is the normal - /// result of a retry after the removal already landed. - /// - /// Index row first, mirroring the enqueue rationale from the other side. A crash between the two - /// leaves a queue row for a task that has finished; it gets claimed once more and the resolver drops - /// it as a stale descriptor, which is already a handled path. Deleting the queue row first would - /// instead leave the index advertising queued work that does not exist, which stalls the run. - /// - public async Task RemoveAsync(ClaimedJob job, CancellationToken ct = default) - { - await _store.DeleteAsync(_indexTable, IndexPartition(job.RunName), IndexRowKey(job.Bucket, job.RowKey), ct); - await _store.DeleteAsync(_queueTable, job.Bucket, job.RowKey, ct); - } - - /// - /// Remove many finished tasks at once. The pump releases a whole claimed batch per cycle, so this - /// turns what was 2 point deletes per task (index + queue, one each) into - /// one transaction per partition: index rows share a run's partition, queue rows share a bucket. - /// - /// Ordering matches at the batch level: ALL index rows first, then the - /// queue rows. A crash in between leaves queue rows whose tasks are finished — claimed once more and - /// dropped as stale descriptors, an already-handled path — whereas deleting the queue rows first - /// would leave the index advertising work that no longer exists and stall those runs. - /// - public async Task RemoveBatchAsync(IReadOnlyList jobs, CancellationToken ct = default) - { - if (jobs.Count == 0) return; - - foreach (var byRun in jobs.GroupBy(j => IndexPartition(j.RunName))) - await _store.DeleteBatchAsync(_indexTable, byRun.Key, - byRun.Select(j => IndexRowKey(j.Bucket, j.RowKey)).ToList(), ct); - - foreach (var byBucket in jobs.GroupBy(j => j.Bucket)) - await _store.DeleteBatchAsync(_queueTable, byBucket.Key, - byBucket.Select(j => j.RowKey).ToList(), ct); - } - - /// - /// This run's index rows. One single-partition read — the operation every run-scoped method below - /// used to perform as a full-table scan. - /// - private async Task> - ReadIndexAsync(string runName, CancellationToken ct) - { - var entries = new List<(string, string, string, string)>(); - - await foreach (var row in _store.QueryPartitionAsync(_indexTable, IndexPartition(runName), ct)) - { - var split = SplitIndexRowKey(row.RowKey); - if (split == null) continue; - - var taskId = row.GetString("TaskId"); - if (string.IsNullOrEmpty(taskId)) continue; - - entries.Add((taskId, split.Value.Bucket, split.Value.QueueRowKey, row.RowKey)); - } - - return entries; - } - - /// - /// Extend the lease on jobs still in flight. One transaction per bucket, so a full buffer costs one - /// round-trip rather than one per job. Returns false if any renewal was rejected, which means the - /// lease had already lapsed and the work may have been taken. - /// - public async Task RenewAsync(IReadOnlyList jobs, string owner, TimeSpan leaseFor, - CancellationToken ct = default) - { - if (jobs.Count == 0) return true; - - var leaseUntil = DateTimeOffset.UtcNow.Add(leaseFor); - var ok = true; - - foreach (var group in jobs.GroupBy(j => j.Bucket)) - { - var rows = new List(); - foreach (var job in group) - { - var row = await _store.GetAsync(_queueTable, job.Bucket, job.RowKey, ct); - // Gone means finished and removed; still ours means renewable. Anything else is not ours. - if (row == null) continue; - if (row.GetString("Owner") != owner) { ok = false; continue; } - - row["LeaseUntil"] = leaseUntil; - rows.Add(row); - } - - if (rows.Count > 0 && !await _store.TryReplaceBatchAsync(_queueTable, group.Key, rows, ct)) - ok = false; - } - - return ok; - } - - /// - /// Drop a run's queued rows. With , only that outing's rows: a recurring - /// run name shares one index partition across outings, and a late removal of the previous outing must - /// not take the next one's rows with it. - /// - public async Task RemoveRunAsync(string runName, DateTime? runStartedUtc = null, CancellationToken ct = default) - { - var entries = await ReadIndexAsync(runName, ct); - if (runStartedUtc is { } started) - { - var prefix = RunKeyPrefix(started, runName); - entries = entries.Where(e => e.QueueRowKey.StartsWith(prefix, StringComparison.Ordinal)).ToList(); - } - - foreach (var byBucket in entries.GroupBy(e => e.Bucket)) - await _store.DeleteBatchAsync(_queueTable, byBucket.Key, byBucket.Select(e => e.QueueRowKey).ToList(), ct); - - if (runStartedUtc == null) - await _store.DeletePartitionAsync(_indexTable, IndexPartition(runName), ct); - else if (entries.Count > 0) - await _store.DeleteBatchAsync(_indexTable, IndexPartition(runName), entries.Select(e => e.IndexRowKey).ToList(), ct); - } - - /// - /// Empty the durable queue: delete every queue row and every index row (keeping only the schema - /// marker). Returns the number of queue rows removed. - /// - /// A maintenance/reset primitive. It drops the BACKLOG, not in-flight work — a row a worker is already - /// running finishes, and the pump's later removal of it simply 404s. Tasks still Pending in the - /// orchestrator's own tables can be re-driven onto the queue by recovery, so pair this with cancelling - /// the runs when the intent is to STOP work rather than to clear a wedged or corrupted queue. - /// Deletes are streamed in bounded windows, so this holds a fixed amount of memory on any queue size. - /// - public async Task ClearAllAsync(CancellationToken ct = default) - { - await InitializeAsync(ct); - - var removed = await ClearTableAsync(_queueTable, keepSchema: false, ct); - await ClearTableAsync(_indexTable, keepSchema: true, ct); - - _logger.LogWarning("[JobQueue] Durable queue cleared — {Count} queue row(s) removed", removed); - return removed; - } - - /// Delete every row of one table (optionally sparing the schema marker), batched per - /// partition and flushed in bounded windows so a huge table never lands in memory at once. - private async Task ClearTableAsync(string table, bool keepSchema, CancellationToken ct) - { - var removed = 0; - var pending = new Dictionary>(StringComparer.Ordinal); - var buffered = 0; - - async Task FlushAsync() - { - foreach (var (partition, keys) in pending) - await _store.DeleteBatchAsync(table, partition, keys, ct); - removed += buffered; - pending.Clear(); - buffered = 0; - } - - await foreach (var row in _store.QueryTableAsync(table, ct)) - { - if (keepSchema && row.PartitionKey == SchemaPartition) continue; - if (!pending.TryGetValue(row.PartitionKey, out var keys)) pending[row.PartitionKey] = keys = []; - keys.Add(row.RowKey); - if (++buffered >= BackfillFlushThreshold) await FlushAsync(); - } - - await FlushAsync(); - return removed; - } - - /// - /// Hand back every claim on a run's rows, making them immediately claimable again. Returns how many - /// were released. - /// - /// For crash recovery only, where "this run was interrupted" already means the process that held - /// these claims is gone. Without it a crash strands the run for up to the full lease: the rows are - /// owned with a live LeaseUntil, so nothing can claim them, while re-dispatch correctly declines to - /// write duplicates for tasks that already have rows. Seen on a killed 140-task fanout — 12 tasks sat - /// Pending with 0 running for the remainder of a 30 minute lease. - /// - /// Rows are updated in place (same PartitionKey/RowKey), so this frees the existing row rather than - /// adding another one. - /// - public async Task ReleaseRunClaimsAsync(string runName, CancellationToken ct = default) - { - var released = 0; - - foreach (var (bucket, rows) in await ReadRunQueueRowsAsync(await ReadIndexAsync(runName, ct), null, ct)) - { - var owned = rows.Where(r => !string.IsNullOrEmpty(r.GetString("Owner"))).ToList(); - foreach (var row in owned) - { - row["Owner"] = ""; - row["LeaseUntil"] = (DateTimeOffset?)null; - } - if (owned.Count == 0) continue; - await _store.UpsertBatchAsync(_queueTable, bucket, owned, ct); - released += owned.Count; - } - - // Freed claims are claimable again — wake the pump to pick them up rather than waiting for the - // recovery-path re-drive on its own timer. - if (released > 0) WakePump(); - - return released; - } - - /// Above this many tasks a run's queue rows are read as one key range rather than one GET each. - private const int PointReadLimit = 32; - - /// - /// The queue rows behind , per bucket. A few are point reads; more are one - /// range read over the run's contiguous keys, so the cost follows the run's size, never the queue's. - /// - private async Task Rows)>> ReadRunQueueRowsAsync( - List<(string TaskId, string Bucket, string QueueRowKey, string IndexRowKey)> entries, - IReadOnlyList? properties, CancellationToken ct) - { - var result = new List<(string, List)>(); - foreach (var byBucket in entries.GroupBy(e => e.Bucket)) - { - var rows = new List(); - var keys = byBucket.Select(e => e.QueueRowKey).ToHashSet(StringComparer.Ordinal); - var prefixes = keys.Select(RunKeyPrefixOf).Distinct().ToList(); - - if (keys.Count <= PointReadLimit || prefixes.Contains(null)) - { - foreach (var key in keys) - if (await _store.GetAsync(_queueTable, byBucket.Key, key, ct) is { } row) rows.Add(row); - } - else - { - foreach (var prefix in prefixes) - await foreach (var row in _store.QueryRowKeyRangeAsync(_queueTable, byBucket.Key, prefix!, - PrefixUpperBound(prefix!), properties, ct)) - if (keys.Contains(row.RowKey)) rows.Add(row); - } - - result.Add((byBucket.Key, rows)); - } - return result; - } - - /// A queued row as the status APIs see it: identity, priority, age and claim state. - public sealed record QueuedRow(string RunName, string TaskId, int Priority, DateTime QueuedUtc, - bool Claimed, string Owner, string Bucket, string RowKey); - - /// - /// A row's enqueue time: the QueuedUtc property (schema v2), falling back to the timestamp a - /// legacy v1 key was built from, then to now. For age/status reporting only. - /// - private static DateTime QueuedUtcOf(StoreRow row) => - row.GetDateTimeOffset("QueuedUtc")?.UtcDateTime - ?? ParseLegacyQueuedUtc(row.RowKey) - ?? DateTime.UtcNow; - - /// A queue row's columns. Keys are named because a projection returns only what it lists. - private static readonly string[] s_queuedRowProperties = - ["PartitionKey", "RowKey", "RunName", "TaskId", "Priority", "Owner", "LeaseUntil", "QueuedUtc"]; - - /// - /// Every row currently in the queue, streamed in storage order (highest priority bucket first, oldest - /// run first within it) and projected to what the status APIs read. Claimed means owned under a live - /// lease. One scan, proportional to the backlog, so callers aggregate as it streams and cache the - /// result rather than holding the rows or calling this per poll. - /// - public async IAsyncEnumerable StreamQueuedAsync( - [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) - { - var now = DateTimeOffset.UtcNow; - await foreach (var row in _store.QueryTableAsync(_queueTable, null, s_queuedRowProperties, ct)) - { - yield return new QueuedRow( - row.GetString("RunName") ?? "", - row.GetString("TaskId") ?? "", - row.GetInt32("Priority") ?? 0, - QueuedUtcOf(row), - !IsClaimable(row, now), - row.GetString("Owner") ?? "", - row.PartitionKey, - row.RowKey); - } - } - - /// materialized, for small queues and tests. - public async Task> ListQueuedAsync(CancellationToken ct = default) - { - var rows = new List(); - await foreach (var row in StreamQueuedAsync(ct)) rows.Add(row); - return rows; - } - - /// - /// Remove every queue row for one task, regardless of claim state. Returns how many were removed. - /// Used by the durable cancel path — the caller must have already marked the task terminal in the - /// run graph, or the orphan re-drive sees a Pending task with no row and puts one straight back. - /// - public async Task RemoveTaskAsync(string runName, string taskId, CancellationToken ct = default) - { - var removed = 0; - var partition = IndexPartition(runName); - - foreach (var e in await ReadIndexAsync(runName, ct)) - { - if (e.TaskId != taskId) continue; - - await _store.DeleteAsync(_indexTable, partition, e.IndexRowKey, ct); - await _store.DeleteAsync(_queueTable, e.Bucket, e.QueueRowKey, ct); - removed++; - } - - return removed; - } - - /// - /// Move a task's queue rows to a new priority bucket, keeping the run's epoch so the task keeps its - /// place in line within the new priority. Returns how many rows moved. - /// - /// Delete-then-add, in that order: a crash in between loses the row, which the orphan re-drive - /// repairs by re-queueing the task. The other order leaves TWO claimable rows for one task, and a - /// duplicated row is executed once per copy — that is the failure mode this queue exists to prevent. - /// - public async Task ReprioritizeTaskAsync(string runName, string taskId, int newPriority, - CancellationToken ct = default) - { - var moved = 0; - var now = DateTimeOffset.UtcNow; - var partition = IndexPartition(runName); - var toMove = new List(); - - foreach (var e in await ReadIndexAsync(runName, ct)) - { - if (e.TaskId != taskId) continue; - if (e.Bucket == Bucket(newPriority)) continue; // already there - - var row = await _store.GetAsync(_queueTable, e.Bucket, e.QueueRowKey, ct); - if (row == null) continue; - - // A claimed row is already buffered on some instance and about to run — re-adding it - // unclaimed would create a second runnable copy of the task. Leave it be. - if (!IsClaimable(row, now)) continue; - - toMove.Add(row); - } - - foreach (var row in toMove) - { - await _store.DeleteAsync(_indexTable, partition, IndexRowKey(row.PartitionKey, row.RowKey), ct); - await _store.DeleteAsync(_queueTable, row.PartitionKey, row.RowKey, ct); - - await EnqueueAsync(runName, taskId, newPriority, ParseEpoch(row.RowKey) ?? QueuedUtcOf(row), ct); - moved++; - } - - return moved; - } - - /// - /// The task ids this run still has rows for, claimed or not. - /// - /// This is what tells a re-drive the difference between a task that is merely WAITING — Pending in - /// the run graph, sitting in this queue, not yet claimed by the pump — and one whose row is - /// genuinely gone. Under the pump, waiting is the normal state of a backlog: a 124-task run against - /// eight workers has most of its tasks Pending and absent from the JobManager for minutes at a time. - /// Treating that as orphaned re-queues the whole backlog on a timer. - /// - public async Task> GetQueuedTaskIdsAsync(string runName, CancellationToken ct = default) - { - var ids = new HashSet(StringComparer.Ordinal); - - // Index only — this never touches the queue table. It is the hottest of the run-scoped reads - // (once per run per re-drive) and was the single largest source of the scan volume. - await foreach (var row in _store.QueryPartitionAsync(_indexTable, IndexPartition(runName), ct)) - { - var taskId = row.GetString("TaskId"); - if (!string.IsNullOrEmpty(taskId)) ids.Add(taskId); - } - - return ids; - } - - /// - /// Of (all belonging to ), the ones the pump - /// can still dispatch: they have a queue row that EXISTS and that the claim filter will match — now, - /// because it is free, or later, because it holds a lease that will lapse. - /// - /// This is the queue-table counterpart to , which answers purely - /// from the index. The index is what makes "does this run still have queued work" a single-partition - /// read, but it can OUTLIVE the queue rows it points at, and then it lies: it reports a task queued - /// that no pump will ever run. That divergence is not hypothetical — - /// - /// a removal deletes the index row first, so a crash in between (or a - /// that deleted queue rows before its index partition) can leave the - /// opposite; - /// a run left Pending under a build that dispatched into memory rather than this - /// queue re-enters here with index rows and no queue rows; - /// a row owned with NO LeaseUntil is excluded by - /// forever, so it sits with an index entry advertising it. - /// - /// A task in any of those states is invisible to the pump AND reported "queued" by the index, so the - /// re-drive that trusts the index never re-enqueues it and the run stalls indefinitely with it, its - /// watchdog never firing. One index read plus , so the cost follows - /// the run's size. - /// - public async Task> GetDispatchableTaskIdsAsync( - string runName, IReadOnlyCollection taskIds, CancellationToken ct = default) - { - var result = new HashSet(StringComparer.Ordinal); - if (taskIds.Count == 0) return result; - - var wanted = taskIds as HashSet ?? new HashSet(taskIds, StringComparer.Ordinal); - var entries = (await ReadIndexAsync(runName, ct)).Where(e => wanted.Contains(e.TaskId)).ToList(); - - foreach (var (_, rows) in await ReadRunQueueRowsAsync(entries, s_queuedRowProperties, ct)) - { - foreach (var row in rows) - { - // Owned with no lease is what the server-side claim filter cannot match (it is neither - // Owner eq '' nor LeaseUntil lt now), so the pump would never dispatch it — a ghost as - // surely as a missing row. - if (!string.IsNullOrEmpty(row.GetString("Owner")) && row.GetDateTimeOffset("LeaseUntil") == null) - continue; - if (row.GetString("TaskId") is { } taskId) result.Add(taskId); - } - } - - return result; - } - - /// - /// Bring the queue tables up to , once per storage account. The marker row - /// written at the end is checked first, so every later start is a single point read. - /// - /// Re-keys every queue row to , taking each run's start time from the Runs - /// table, and rebuilds the index entries. One forward pass takes any older account straight to the - /// current version — there is no dual-read, and after the pass only the new key scheme is used. - /// - /// Rows are rewritten new-key-first, old-key-deleted-after, so a crash mid-pass leaves the marker - /// unset and the next start finishes the job, skipping rows already on their key. The pump awaits before it claims — and every - /// enqueue path calls it too — so no row is ever claimed or written while the migration is only - /// half-applied. - /// - /// Awaited by InitializeAsync rather than backgrounded: the run-scoped reads and the deterministic - /// keys are only correct once it has finished. - /// - private async Task MigrateSchemaAsync(CancellationToken ct) - { - var marker = await _store.GetAsync(_indexTable, SchemaPartition, SchemaRowKey, ct); - if ((marker?.GetInt32("Version") ?? 0) >= SchemaVersion) return; - - var started = DateTime.UtcNow; - _logger.LogInformation( - "[JobQueue] Migrating queue schema to v{Version} — one full pass over {Table}", SchemaVersion, _queueTable); - - var newQueue = new Dictionary>(StringComparer.Ordinal); // by bucket - var newIndex = new Dictionary>(StringComparer.Ordinal); // by run partition - var oldQueue = new Dictionary>(StringComparer.Ordinal); // bucket -> old row keys - var oldIndex = new Dictionary>(StringComparer.Ordinal); // run partition -> old index keys - var buffered = 0; - var migrated = 0; - var rekeyed = 0; - var skipped = 0; - var epochs = new Dictionary(StringComparer.Ordinal); - await foreach (var run in _store.QueryTableAsync(_runsTable, "PartitionKey eq 'Run'", ["PartitionKey", "RowKey", "StartedUtc"], ct)) - if (run.PartitionKey == "Run" && run.GetDateTimeOffset("StartedUtc") is { } runStarted) - epochs[run.RowKey] = runStarted.UtcDateTime; - - static void AddRow(Dictionary> map, string key, StoreRow row) - { - if (!map.TryGetValue(key, out var list)) map[key] = list = []; - list.Add(row); - } - static void AddKey(Dictionary> map, string key, string rowKey) - { - if (!map.TryGetValue(key, out var list)) map[key] = list = []; - list.Add(rowKey); - } - - // Chunked to a transaction's worth, so the one big bucket partition goes in parallel too. - Task EachAsync(Dictionary> map, Func, Task> write) => - Parallel.ForEachAsync(map.SelectMany(kv => kv.Value.Chunk(100).Select(c => (kv.Key, c))), - new ParallelOptions { MaxDegreeOfParallelism = MigrationConcurrency, CancellationToken = ct }, - async (part, _) => await write(part.Key, part.c)); - - async Task FlushAsync() - { - // New rows first, so a crash before the deletes leaves BOTH and the re-run converges — never - // the index advertising a queue row that no longer exists. Within a phase the partitions are - // independent, and the index has one per run, so they go in parallel: one at a time, a - // 16k-run backlog was ~32k sequential round trips while the pump waited to claim. - await EachAsync(newQueue, (bucket, rows) => _store.UpsertBatchAsync(_queueTable, bucket, rows, ct)); - await EachAsync(newIndex, (partition, rows) => _store.UpsertBatchAsync(_indexTable, partition, rows, ct)); - await EachAsync(oldQueue, (bucket, keys) => _store.DeleteBatchAsync(_queueTable, bucket, keys, ct)); - await EachAsync(oldIndex, (partition, keys) => _store.DeleteBatchAsync(_indexTable, partition, keys, ct)); - - migrated += buffered; - newQueue.Clear(); newIndex.Clear(); oldQueue.Clear(); oldIndex.Clear(); - buffered = 0; - } - - await foreach (var row in _store.QueryTableAsync(_queueTable, ct)) - { - var runName = row.GetString("RunName"); - var taskId = row.GetString("TaskId"); - if (string.IsNullOrEmpty(runName) || string.IsNullOrEmpty(taskId)) { skipped++; continue; } - - var bucket = row.PartitionKey; - var runPartition = IndexPartition(runName); - - // The run's start time, as every later enqueue will key it. A run with no Run row falls back to - // the first of its rows the scan meets, and keeps that for the rest of them. - if (!epochs.TryGetValue(runName, out var epoch)) - epochs[runName] = epoch = ParseEpoch(row.RowKey) ?? QueuedUtcOf(row); - var newKey = BuildRowKey(epoch, runName, taskId); - buffered++; - - // Already on its key (a re-run after a crash): only make sure the index points at it. - if (row.RowKey == newKey) - { - AddRow(newIndex, runPartition, IndexRow(runName, taskId, bucket, newKey)); - continue; - } - - // The new-scheme row: same bucket + claim state + priority, key deterministic, enqueue time as - // a property (from the row, or the legacy key, or now). - AddRow(newQueue, bucket, new StoreRow(bucket, newKey) - { - Properties = - { - ["RunName"] = runName, - ["TaskId"] = taskId, - ["Priority"] = row.GetInt32("Priority") ?? 0, - ["Owner"] = row.GetString("Owner") ?? "", - ["LeaseUntil"] = row.GetDateTimeOffset("LeaseUntil"), - ["QueuedUtc"] = new DateTimeOffset(QueuedUtcOf(row), TimeSpan.Zero), - } - }); - AddRow(newIndex, runPartition, IndexRow(runName, taskId, bucket, newKey)); - rekeyed++; - AddKey(oldQueue, bucket, row.RowKey); - AddKey(oldIndex, runPartition, IndexRowKey(bucket, row.RowKey)); - - if (buffered >= BackfillFlushThreshold) - { - await FlushAsync(); - _logger.LogInformation("[JobQueue] Schema migration: {Migrated:N0} rows so far", migrated); - } - } - - await FlushAsync(); - - await _store.UpsertAsync(_indexTable, new StoreRow(SchemaPartition, SchemaRowKey) - { - Properties = - { - ["Version"] = SchemaVersion, - ["BuiltUtc"] = new DateTimeOffset(started, TimeSpan.Zero), - ["RowsIndexed"] = migrated, - } - }, ct); - - _logger.LogInformation( - "[JobQueue] Queue schema at v{Version}: {Migrated:N0} row(s) processed, {Rekeyed:N0} re-keyed, in {Seconds:N0}s{Skipped} — this will not run again", - SchemaVersion, migrated, rekeyed, (DateTime.UtcNow - started).TotalSeconds, - skipped > 0 ? $", {skipped:N0} malformed row(s) skipped" : ""); - } -} diff --git a/Services/Storage/OrchestratorCleanupResult.cs b/Services/Storage/OrchestratorCleanupResult.cs deleted file mode 100644 index 959bb54..0000000 --- a/Services/Storage/OrchestratorCleanupResult.cs +++ /dev/null @@ -1,15 +0,0 @@ -namespace Craft.Storage; - -/// -/// What one pass removed, so the caller can -/// log it and take the follow-up that is not the store's business (an abandoned run's queue rows). -/// -/// Run rows scanned. -/// Runs that had finished and were past retention. -/// Runs that had not finished, were not active, and had not been written to within retention. -/// Tasks/Results partitions with no Run row whose newest row was past retention. -public sealed record OrchestratorCleanupResult( - int RunsExamined, - IReadOnlyList ExpiredRuns, - IReadOnlyList AbandonedRuns, - int OrphanPartitionsRemoved); diff --git a/Services/Storage/OrchestratorRunSummary.cs b/Services/Storage/OrchestratorRunSummary.cs deleted file mode 100644 index dd21983..0000000 --- a/Services/Storage/OrchestratorRunSummary.cs +++ /dev/null @@ -1,12 +0,0 @@ -namespace Craft.Storage; - -/// -/// A run's identity and parentage, read from the run row alone — no task rows. -/// -/// Exists so startup can rebuild parent/child run links from one partition scan instead of calling -/// GetRunAsync per run, which loads every task of every run. -/// -/// Run name (the run row's RowKey). -/// Running | Completed | CompletedWithErrors | Failed. -/// Parent run, when this run was spawned by a task of another run. -public record OrchestratorRunSummary(string Name, string Status, string? ParentRunName); diff --git a/Services/Storage/OrchestratorTableStore.cs b/Services/Storage/OrchestratorTableStore.cs deleted file mode 100644 index 1e207fc..0000000 --- a/Services/Storage/OrchestratorTableStore.cs +++ /dev/null @@ -1,997 +0,0 @@ -using System.Runtime.CompilerServices; -using System.Text; -using System.Text.Json; -using Craft.Configuration; -using Craft.Orchestration; - -namespace Craft.Storage; - -/// -/// Typed CRUD wrapper over for orchestrator persistence. Manages three -/// logical tables: {prefix}Runs, {prefix}Tasks, {prefix}Results. Persists through the -/// abstraction. -/// -public class OrchestratorTableStore -{ - private readonly ILogger _logger; - private readonly ICraftTableStore _store; - private readonly string _runsTable; - private readonly string _tasksTable; - private readonly string _resultsTable; - private bool _initialized; - - private static readonly JsonSerializerOptions s_jsonOptions = new() - { - WriteIndented = false, - PropertyNamingPolicy = JsonNamingPolicy.CamelCase - }; - - public OrchestratorTableStore(ILogger logger, CraftSettings settings, ICraftTableStore store) - { - _logger = logger; - _store = store; - var prefix = settings.Orchestrator.TablePrefix; - _runsTable = $"{prefix}Runs"; - _tasksTable = $"{prefix}Tasks"; - _resultsTable = $"{prefix}Results"; - } - - /// Create the three tables if they do not exist. Called once on startup. - public async Task InitializeAsync() - { - if (_initialized) return; - - await _store.EnsureTableAsync(_runsTable); - await _store.EnsureTableAsync(_tasksTable); - await _store.EnsureTableAsync(_resultsTable); - _initialized = true; - - _logger.LogInformation("[OrchestratorStore] Tables initialized"); - } - - /// Upsert run metadata (without tasks — tasks are separate rows). - public async Task UpsertRunAsync(OrchestratorRun run) - { - await _store.UpsertAsync(_runsTable, BuildRunRow(run)); - } - - /// - /// Persist many run rows in as few round-trips as possible. - /// - /// Every run row shares the constant "Run" partition key, so N of them cost ceil(N/100) - /// transactions rather than N. Writing them one at a time inside a flush bounded by - /// StatusFlushTimeoutSeconds is what pushed real flushes past 30s and then past the 90s durable - /// barrier: every task waiting on its "Running" marker deferred, and after MaxDeferrals was - /// abandoned as Pending with nothing left to retry it. - /// - /// Returns the names of runs that did not persist, so the caller can requeue exactly those. - /// - public async Task> WriteRunStatusBatchAsync(IReadOnlyList runs, - CancellationToken ct = default) - { - if (runs.Count == 0) return []; - - var failed = new List(); - - // 100 is the Azure Table transaction ceiling. - foreach (var chunk in runs.Chunk(100)) - { - try - { - await _store.UpsertBatchAsync(_runsTable, "Run", chunk.Select(BuildRunRow).ToList(), ct); - } - catch (Exception ex) - { - // A transaction fails atomically, so one rejected row would discard the other 99. Retry - // the chunk row by row: the poison row is isolated and the rest still land. - _logger.LogWarning(ex, - "[OrchestratorStore] Run status batch of {Count} failed — retrying individually", chunk.Length); - - foreach (var run in chunk) - { - try { await _store.UpsertAsync(_runsTable, BuildRunRow(run), ct); } - catch (Exception single) - { - failed.Add(run.Name); - _logger.LogWarning(single, - "[OrchestratorStore] Run status write failed for {Run} — will retry", run.Name); - } - } - } - } - - return failed; - } - - private static StoreRow BuildRunRow(OrchestratorRun run) => new("Run", run.Name) - { - Properties = - { - ["Status"] = run.Status, - ["Priority"] = run.Priority, - ["StartedUtc"] = run.StartedUtc, - ["CompletedUtc"] = run.CompletedUtc, - ["TaskScriptName"] = run.TaskScriptName, - ["PostExecFunctionName"] = run.PostExecFunctionName, - ["PostExecParametersJson"] = run.PostExecParametersJson, - ["PostExecStatus"] = run.PostExecStatus, - ["PostExecAttemptCount"] = run.PostExecAttemptCount, - ["Reference"] = run.Reference, - ["ParentRunName"] = run.ParentRunName, - ["TaskCount"] = run.Tasks.Count, - // Persisted so a resumed sequential run keeps advancing one task at a time (0/1 — StoreRow has - // no bool reader). Absent on older rows reads as 0 = the fan-out default. - ["Sequential"] = run.Sequential ? 1 : 0 - } - }; - - /// Load a run and all its tasks. Returns null if the run does not exist. - public async Task GetRunAsync(string name) - { - var runRow = await _store.GetAsync(_runsTable, "Run", name); - if (runRow == null) return null; - - var run = new OrchestratorRun - { - Name = name, - Status = runRow.GetString("Status") ?? "Pending", - Priority = runRow.GetInt32("Priority") ?? 4, - StartedUtc = runRow.GetDateTimeOffset("StartedUtc")?.UtcDateTime ?? DateTime.UtcNow, - CompletedUtc = runRow.GetDateTimeOffset("CompletedUtc")?.UtcDateTime, - TaskScriptName = runRow.GetString("TaskScriptName"), - PostExecFunctionName = runRow.GetString("PostExecFunctionName"), - PostExecParametersJson = runRow.GetString("PostExecParametersJson"), - PostExecStatus = runRow.GetString("PostExecStatus"), - // Absent on rows written before this existed — 0 is the right reading of "never retried". - PostExecAttemptCount = runRow.GetInt32("PostExecAttemptCount") ?? 0, - // Both were in-memory only until now: a resumed run came back with a null Reference - // (so FindRunByReference could not see it) and a null ParentRunName (so its finalize - // never re-checked the parent). Absent on rows written before this existed. - Reference = runRow.GetString("Reference"), - ParentRunName = runRow.GetString("ParentRunName"), - Sequential = (runRow.GetInt32("Sequential") ?? 0) == 1 - }; - - var tasks = new List(); - await foreach (var taskRow in _store.QueryPartitionAsync(_tasksTable, name)) - { - // The remaining-count row shares this partition so it can be updated in the same transaction - // as a task completion. It is bookkeeping, not work — materializing it would give every run a - // phantom task that never completes, and no run would ever finalize. - if (taskRow.RowKey == CounterRowKey) continue; - - var parametersJson = taskRow.GetString("ParametersJson"); - Dictionary parameters; - try - { - parameters = !string.IsNullOrEmpty(parametersJson) - ? JsonSerializer.Deserialize>(parametersJson, s_jsonOptions) ?? [] - : []; - } - catch - { - parameters = []; - } - - tasks.Add(new OrchestratorTaskItem - { - Id = taskRow.RowKey, - Status = taskRow.GetString("Status") ?? "Pending", - Parameters = parameters, - AttemptCount = taskRow.GetInt32("AttemptCount") ?? 0, - LastError = taskRow.GetString("LastError"), - // Absent on rows written before per-task priority existed — null means "inherit the run's". - Priority = taskRow.GetInt32("Priority"), - Sequence = taskRow.GetInt32("Sequence") ?? 0, - CompletedUtc = taskRow.GetDateTimeOffset("CompletedUtc")?.UtcDateTime - }); - } - - run.Tasks = tasks; - return run; - } - - /// - /// Read one task's Parameters from the Tasks table (a single point read + deserialize), matching the - /// deserialization uses. For the pending-Parameters shedding path: the live - /// graph keeps the task object but drops its Parameters payload while it waits, and this rehydrates them - /// at dispatch. Null if the row or its ParametersJson is missing. - /// - public async Task?> GetTaskParametersAsync( - string runName, string taskId, CancellationToken ct = default) - { - var row = await _store.GetAsync(_tasksTable, runName, taskId, ct); - var parametersJson = row?.GetString("ParametersJson"); - if (string.IsNullOrEmpty(parametersJson)) return null; - try - { - return JsonSerializer.Deserialize>(parametersJson, s_jsonOptions) ?? []; - } - catch - { - return []; - } - } - - /// List all known run names. - public async Task> ListRunsAsync() - { - var names = new List(); - await foreach (var row in _store.QueryPartitionAsync(_runsTable, "Run")) - names.Add(row.RowKey); - return names; - } - - /// - /// List every run's identity and parentage without loading its tasks. - /// - /// One scan of the single "Run" partition. would answer the same question - /// but does a task-partition query per run, so using it to rebuild parent/child links at startup - /// would load every task of every run — the opposite of what recovery should cost. - /// - public async Task> ListRunSummariesAsync() - { - var summaries = new List(); - await foreach (var row in _store.QueryPartitionAsync(_runsTable, "Run")) - { - summaries.Add(new OrchestratorRunSummary( - row.RowKey, - row.GetString("Status") ?? "Pending", - row.GetString("ParentRunName"))); - } - return summaries; - } - - // ── Remaining-task counter ──────────────────────────────────────────────── - // - // How many of a run's tasks are not yet terminal, kept in storage rather than derived from the - // in-memory run graph, so finalization stops depending on one process holding the whole graph. - // Scanning instead is not an option: Azure Table cannot count server-side, so "is this run done" - // would be a 7,000-row read per check. - // - // The row lives in the TASKS table, in the run's own partition, and that placement is deliberate: a - // task's terminal write and the counter decrement can then share ONE conditional transaction where - // exactly-once matters most — the cancel-a-run path (CancelPendingTaskAsync) uses exactly that, so a - // cancel racing a task's real completion cannot decrement the counter twice. - // - // The hot fan-out path decrements SEPARATELY (DecrementRemainingAsync), by design: the batched status - // writer coalesces terminal writes, and decrementing once per group after it lands keeps the write - // batched. Idempotency there rests on the writer never re-sending a group that landed, backstopped by - // ReconcileRemainingAsync (a full-partition recount) whenever a decrement is lost. - // - // The reserved row key cannot collide with a task id — task ids are caller-supplied names like - // "CIPPStandard_IntuneTemplate__", never a control character. - private const string CounterRowKey = "!!run-counter"; - private const int CounterAttempts = 8; - - /// The three statuses that mean a task will not run again, and so has been counted. - private static bool IsTerminal(string? status) => - status is "Completed" or "Failed" or "Cancelled"; - - /// Seed the counter when a run is created. Idempotent, so re-seeding a resumed run is safe. - public Task InitRemainingAsync(string runName, int total, CancellationToken ct = default) => - _store.UpsertAsync(_tasksTable, new StoreRow(runName, CounterRowKey) - { - Properties = { ["Remaining"] = total, ["Total"] = total } - }, ct); - - /// How many tasks the STORE believes are outstanding, or null if the run has no counter row. - public async Task GetRemainingAsync(string runName, CancellationToken ct = default) - => (await _store.GetAsync(_tasksTable, runName, CounterRowKey, ct))?.GetInt32("Remaining"); - - /// - /// The run's counter row as (Remaining, Total), or null if the run has no counter. This is what the - /// status APIs use for a run's true size and durable progress — the in-memory JobManager only ever - /// sees the slice of a run that has been claimed onto this instance. - /// - public async Task<(int Remaining, int Total)?> GetCounterAsync(string runName, CancellationToken ct = default) - { - var row = await _store.GetAsync(_tasksTable, runName, CounterRowKey, ct); - if (row == null) return null; - return (row.GetInt32("Remaining") ?? 0, row.GetInt32("Total") ?? 0); - } - - /// - /// Subtract from a run's outstanding count, for the batch path. - /// - /// Exactly-once here rests on the caller: the coalescing status writer holds one pending write per - /// task and re-queues only the runs whose batch did NOT land, so a group that applied is never - /// re-sent. Call this only after a group has been written successfully, counting the terminal rows - /// in that group. Retries on a lost race so a concurrent decrement cannot swallow this one. - /// - public async Task DecrementRemainingAsync(string runName, int by, CancellationToken ct = default) - { - if (by <= 0) return await GetRemainingAsync(runName, ct); - - for (var attempt = 0; attempt < CounterAttempts; attempt++) - { - var counter = await _store.GetAsync(_tasksTable, runName, CounterRowKey, ct); - if (counter == null) return null; - - counter["Remaining"] = Math.Max(0, (counter.GetInt32("Remaining") ?? 0) - by); - - if (await _store.TryReplaceBatchAsync(_tasksTable, runName, [counter], ct)) - return (int)counter["Remaining"]!; - } - - _logger.LogWarning("[OrchestratorStore] Could not decrement remaining for {Run} by {By} after {Attempts} attempts", - runName, by, CounterAttempts); - return null; - } - - /// - /// Recount Remaining from the task rows the counter summarizes, and repair the counter row - /// when they disagree. - /// - /// A decrement that exhausts its retries is never re-applied — the terminal task rows landed but - /// the counter kept its old value, and from then on it permanently overstates the outstanding work - /// and finalize defers forever. The scan is the whole-partition read the counter exists to avoid, - /// which is why this runs only when a caller has evidence of drift (a lost decrement, a finalize - /// deferred repeatedly), never on the hot path. - /// - /// The reconciled outstanding count, or null if the run has no counter row or a concurrent - /// writer moved the counter mid-recount — the caller's next pass re-reads either way. - public async Task ReconcileRemainingAsync(string runName, CancellationToken ct = default) - { - var counter = await _store.GetAsync(_tasksTable, runName, CounterRowKey, ct); - if (counter == null) return null; - - var outstanding = 0; - await foreach (var row in _store.QueryPartitionAsync(_tasksTable, runName, ct)) - { - if (row.RowKey == CounterRowKey) continue; - if (!IsTerminal(row.GetString("Status"))) outstanding++; - } - - var stored = counter.GetInt32("Remaining") ?? 0; - if (stored == outstanding) return outstanding; - - // ETag-guarded: a decrement landing between the read above and this write rejects the - // replace, so a recount can never overwrite fresher progress with a stale count. - counter["Remaining"] = outstanding; - if (!await _store.TryReplaceBatchAsync(_tasksTable, runName, [counter], ct)) - return null; - - _logger.LogWarning( - "[OrchestratorStore] Reconciled remaining for {Run}: counter said {Stored}, task rows say {Actual}", - runName, stored, outstanding); - return outstanding; - } - - /// What a status-guarded cancel actually did, so the caller can keep its view honest. - public sealed record CancelWriteResult(bool Cancelled, string? CurrentStatus); - - /// - /// Write a task as Cancelled and decrement the run counter, but ONLY while storage still shows the - /// task Pending. This is the cancel-a-run primitive, and the guard is the point: between the - /// caller's read and this write, dispatch can move a task to Running — an unguarded terminal write - /// would clobber that, and the task's real completion would then decrement the counter a SECOND - /// time, letting the run finalize while work is still outstanding. - /// - /// A run without a counter row (pre-counter) gets the same status guard with a task-ETag-only write. - /// - /// - /// Whether the cancel landed, plus the status storage showed when it did not — so the caller can - /// correct its in-memory copy rather than believing a cancel that never happened. - /// - public async Task CancelPendingTaskAsync(string runName, OrchestratorTaskItem task, - CancellationToken ct = default) - { - for (var attempt = 0; attempt < CounterAttempts; attempt++) - { - var existing = await _store.GetAsync(_tasksTable, runName, task.Id, ct); - var currentStatus = existing?.GetString("Status"); - if (existing == null || currentStatus != "Pending") - return new CancelWriteResult(false, currentStatus); - - var guarded = new StoreRow(runName, task.Id) - { - ETag = existing.ETag, - Properties = BuildTaskRow(runName, task).Properties, - }; - - var counter = await _store.GetAsync(_tasksTable, runName, CounterRowKey, ct); - if (counter == null) - { - if (await _store.TryReplaceBatchAsync(_tasksTable, runName, [guarded], ct)) - return new CancelWriteResult(true, null); - continue; - } - - counter["Remaining"] = Math.Max(0, (counter.GetInt32("Remaining") ?? 0) - 1); - if (await _store.TryReplaceBatchAsync(_tasksTable, runName, [guarded, counter], ct)) - return new CancelWriteResult(true, null); - } - - _logger.LogWarning("[OrchestratorStore] Could not cancel {Task} in {Run} after {Attempts} attempts", - task.Id, runName, CounterAttempts); - return new CancelWriteResult(false, null); - } - - /// Upsert a single task row. - public Task UpsertTaskAsync(string runName, OrchestratorTaskItem task) => - _store.UpsertAsync(_tasksTable, BuildTaskRow(runName, task)); - - /// Batch upsert all tasks for a run (used at run creation). - public Task UpsertTaskBatchAsync(string runName, List tasks) => - _store.UpsertBatchAsync(_tasksTable, runName, tasks.Select(t => BuildTaskRow(runName, t)).ToList()); - - /// - /// Write a set of coalesced task-status transitions. Rows are grouped by run (partition) and handed - /// to the store, which applies each group as atomically as the backend allows and chunks to any - /// per-request limits. Used by the batched status writer; the large-result path - /// () is untouched. - /// - /// - /// The run names whose writes did NOT persist. Callers must retry these rather than dropping them — - /// they are terminal task states, and losing one means a finished task looks Pending forever. - /// An empty list means everything landed. - /// - public async Task> WriteTaskStatusBatchAsync(IReadOnlyList writes, - int maxConcurrency = 8, CancellationToken ct = default) - { - // Grouped by run because a batch has to share a partition key. Run these with bounded - // concurrency instead of one after another: the workload that broke this was ~600 runs of a - // single task each, so "batching" degenerated into hundreds of sequential round-trips inside - // one flush — with the single drain loop, and therefore every task waiting on its durable - // marker, blocked for the whole duration. - var groups = writes.GroupBy(w => w.RunName).ToList(); - if (groups.Count == 0) return []; - - var failed = new System.Collections.Concurrent.ConcurrentBag(); - using var gate = new SemaphoreSlim(Math.Max(1, maxConcurrency)); - - var tasks = groups.Select(async group => - { - await gate.WaitAsync(ct); - try - { - await _store.UpsertBatchAsync(_tasksTable, group.Key, group.Select(BuildTaskRow).ToList(), ct); - - // The group landed, so the tasks it just moved to a terminal state are now durably - // finished. Decrement once for them here rather than per task: this is the only place - // that knows a terminal write actually applied, and the writer never re-sends a group - // that did, which is what keeps the count honest. - var terminal = group.Count(w => IsTerminal(w.Status)); - if (terminal > 0 && await DecrementRemainingAsync(group.Key, terminal, ct) == null) - { - // Retry exhaustion here loses the decrement for good — the terminal rows above - // landed, so the writer will never re-send this group. Recount now rather than - // letting the counter overstate the run's outstanding work forever. - await ReconcileRemainingAsync(group.Key, ct); - } - } - catch (Exception ex) - { - // One run's failure must not discard the other 599. Record it and carry on; the caller - // requeues just this run. - failed.Add(group.Key); - _logger.LogWarning(ex, "[OrchestratorStore] Task status write failed for run {Run} ({Count} tasks) — will retry", - group.Key, group.Count()); - } - finally - { - gate.Release(); - } - }); - - await Task.WhenAll(tasks); - return failed.ToList(); - } - - private static StoreRow BuildTaskRow(string runName, OrchestratorTaskItem task) => new(runName, task.Id) - { - Properties = - { - ["Status"] = task.Status, - ["ParametersJson"] = JsonSerializer.Serialize(task.Parameters, s_jsonOptions), - ["AttemptCount"] = task.AttemptCount, - ["LastError"] = task.LastError, - ["Priority"] = task.Priority, - ["Sequence"] = task.Sequence, - ["CompletedUtc"] = task.CompletedUtc.HasValue - ? new DateTimeOffset(task.CompletedUtc.Value, TimeSpan.Zero) - : (DateTimeOffset?)null - } - }; - - private static StoreRow BuildTaskRow(TaskStatusWrite w) => new(w.RunName, w.TaskId) - { - Properties = - { - ["Status"] = w.Status, - ["ParametersJson"] = w.ParametersJson, - ["AttemptCount"] = w.AttemptCount, - ["LastError"] = w.LastError, - ["Priority"] = w.Priority, - ["Sequence"] = w.Sequence, - ["CompletedUtc"] = w.CompletedUtc.HasValue - ? new DateTimeOffset(w.CompletedUtc.Value, TimeSpan.Zero) - : (DateTimeOffset?)null - } - }; - - // ─── Result storage ─── - // Results can be large (50–150 MB for big runs). We chunk a result across multiple properties and, - // if needed, multiple rows in the same partition. These bounds are sized for Azure Table Storage - // (64 KiB/property, 1 MiB/entity); on a backend without those limits the chunking is simply - // unnecessary but still correct, and it keeps per-row payloads small (good for e.g. SQL packet size). - private const int MaxPropertyChars = 30_000; - private const int MaxEntityChars = 450_000; - - /// - /// Write a set of coalesced SMALL task results (single-property rows) to the Results table, grouped - /// by run (partition) and chunked to the backend's transaction limits by the store. The batched - /// counterpart to for results that fit one property; larger results - /// still go through StoreResultAsync's multi-row chunking path. - /// - /// - /// The run names whose results did NOT persist. The caller must retry these AND withhold those runs' - /// terminal task markers this flush, or a task could be counted done while its result is lost. - /// An empty list means everything landed. - /// - public async Task> WriteResultBatchAsync(IReadOnlyList results, - int maxConcurrency = 8, CancellationToken ct = default) - { - var groups = results.GroupBy(r => r.RunName).ToList(); - if (groups.Count == 0) return []; - - var failed = new System.Collections.Concurrent.ConcurrentBag(); - using var gate = new SemaphoreSlim(Math.Max(1, maxConcurrency)); - - var tasks = groups.Select(async group => - { - await gate.WaitAsync(ct); - try - { - var rows = group.Select(r => new StoreRow(r.RunName, r.TaskId) - { - Properties = { ["ResultJson"] = r.ResultJson } - }).ToList(); - await _store.UpsertBatchAsync(_resultsTable, group.Key, rows, ct); - } - catch (Exception ex) - { - // One run's failure must not discard the others. Record it; the caller requeues just this - // run's results and holds its terminal markers until they land together. - failed.Add(group.Key); - _logger.LogWarning(ex, "[OrchestratorStore] Result write failed for run {Run} ({Count} results) — will retry", - group.Key, group.Count()); - } - finally - { - gate.Release(); - } - }); - - await Task.WhenAll(tasks); - return failed.ToList(); - } - - /// Store a single task result, chunking large JSON across properties/rows as needed. - public async Task StoreResultAsync(string runName, string taskId, string resultJson) - { - // Fast path: fits in a single property - if (resultJson.Length <= MaxPropertyChars) - { - var row = new StoreRow(runName, taskId) { Properties = { ["ResultJson"] = resultJson } }; - await _store.UpsertAsync(_resultsTable, row); - return; - } - - var chunks = ChunkString(resultJson, MaxPropertyChars); - - // Try to fit all chunks into a single row - if (EstimateTotalChars(chunks) <= MaxEntityChars) - { - var row = new StoreRow(runName, taskId); - for (int i = 0; i < chunks.Count; i++) - row[$"ResultJson_{i}"] = chunks[i]; - row["ResultChunkCount"] = chunks.Count; - - await _store.UpsertAsync(_resultsTable, row); - return; - } - - // Row too large — split across multiple rows - var rowIndex = 0; - var chunkIndex = 0; - - while (chunkIndex < chunks.Count) - { - var rowKey = rowIndex == 0 ? taskId : $"{taskId}-part{rowIndex}"; - var row = new StoreRow(runName, rowKey); - - if (rowIndex > 0) - { - row["OriginalEntityId"] = taskId; - row["PartIndex"] = rowIndex; - } - - var currentChars = runName.Length + rowKey.Length + 100; // overhead estimate - - while (chunkIndex < chunks.Count) - { - var chunkChars = chunks[chunkIndex].Length; - if (currentChars + chunkChars + 20 > MaxEntityChars) - break; - - row[$"ResultJson_{chunkIndex}"] = chunks[chunkIndex]; - currentChars += chunkChars + 20; - chunkIndex++; - } - - row["ResultChunkCount"] = chunks.Count; - await _store.UpsertAsync(_resultsTable, row); - rowIndex++; - } - } - - /// - /// Get all result JSON strings for a run, reassembling any chunked/multi-row results. - /// - /// Buffers every result by signature — prefer or - /// for run-sized payloads. - /// - public async Task GetResultsAsync(string runName, CancellationToken ct = default) - { - var results = new List(); - await foreach (var result in StreamResultsAsync(runName, ct)) - results.Add(result); - return results.ToArray(); - } - - /// - /// Stream each run result, reassembled, as it becomes available from the backing store. - /// - /// Nothing is buffered except spill groups still waiting for their remaining rows, so a run whose - /// results each fit in one row holds ONE row at a time regardless of how many there are. - /// - public async IAsyncEnumerable StreamResultsAsync(string runName, - [EnumeratorCancellation] CancellationToken ct = default) - { - await foreach (var chunks in StreamResultChunkGroupsAsync(runName, ct)) - yield return chunks.Count == 1 ? chunks[0] : string.Concat(chunks); - } - - /// - /// Stream all result JSON strings for a run directly to a file, reassembling any chunked/multi-row - /// results on the fly. Writes JSON Lines (NDJSON): one result per line, no enclosing array. - /// - /// Chunks are written individually, so the largest allocation this makes is one chunk - /// () — the reassembled result is never built as a string. - /// - /// JSON Lines rather than a JSON array because the consumer is PowerShell. A JSON array forces the - /// reader to hold the whole document to find where each element ends; one result per line lets - /// Invoke-CraftPostExecution walk the file with File.ReadLines and hold ONE result at a time. That - /// is the difference between a 50-150MB Large Object Heap allocation per post-execution and none. - /// It also isolates failure: a malformed result costs that result, not the entire aggregate. - /// - /// Returns the number of results written. - /// - public async Task StreamResultsToJsonLinesAsync(string runName, string filePath, - CancellationToken ct = default) - { - var count = 0; - - await using (var writer = new StreamWriter(filePath, append: false, Encoding.UTF8, bufferSize: 65536)) - { - await foreach (var chunks in StreamResultChunkGroupsAsync(runName, ct)) - { - foreach (var chunk in chunks) - await WriteSingleLineAsync(writer, chunk); - await writer.WriteAsync('\n'); - count++; - } - } - - _logger.LogInformation("[OrchestratorStore] Streamed {Count} results to {Path} for run {Name}", - count, filePath, runName); - - return count; - } - - /// - /// Write a chunk with any raw CR/LF removed, so one result stays on one line. - /// - /// Results are expected to be compact JSON on a single line — that is what Invoke-CraftTask's - /// `ConvertTo-Json -Compress` produces, and JSON escapes newlines inside strings as \n rather than - /// emitting them raw. A raw newline can therefore only appear as inter-token whitespace, which - /// carries no meaning, or in a result that was not valid JSON to begin with (a task script that - /// wrote several objects to the output stream — the runner joins those with "\n"). Dropping the - /// character is right in the first case and no worse than today's behaviour in the second, where - /// the malformed result currently takes the whole aggregate's parse down with it. - /// - /// The scan is the common-case fast path: no newline means the chunk is written untouched, with - /// no copy and no per-character work beyond the search itself. - /// - private static async Task WriteSingleLineAsync(StreamWriter writer, string chunk) - { - var start = 0; - int idx; - - while ((idx = chunk.AsSpan(start).IndexOfAny('\r', '\n')) >= 0) - { - var abs = start + idx; - if (abs > start) - await writer.WriteAsync(chunk.AsMemory(start, abs - start)); - start = abs + 1; - } - - if (start == 0) - await writer.WriteAsync(chunk); - else if (start < chunk.Length) - await writer.WriteAsync(chunk.AsMemory(start)); - } - - /// - /// The shared core: yields each logical result as its ordered chunk list, as soon as that result is - /// complete, and drops every row it has finished with. - /// - /// This used to be LoadResultGroupsAsync, which materialized EVERY result row for the run into a - /// dictionary before a single byte was written — so the callers named "stream" held the entire - /// payload (as UTF-16, ~2x the stored size) before they started. For a 738-task run whose aggregate - /// is 50-150MB that was a few hundred MB against a 2398MB heap cap, concurrently per post-execution. - /// - /// Rows are grouped by (OriginalEntityId ?? RowKey) and completed by chunk count rather than by - /// arrival order, so this makes no assumption about the order - /// returns rows in — the interface promises none. - /// - private async IAsyncEnumerable> StreamResultChunkGroupsAsync(string runName, - [EnumeratorCancellation] CancellationToken ct = default) - { - // Allocated only if this run actually has a result too large for a single row. - Dictionary? pending = null; - - await foreach (var row in _store.QueryPartitionAsync(_resultsTable, runName, ct)) - { - var totalChunks = row.GetInt32("ResultChunkCount") ?? 0; - - // Fast path: the whole result is one property on this row. Emit and release it. - if (totalChunks == 0) - { - var json = row.GetString("ResultJson"); - if (!string.IsNullOrEmpty(json)) yield return new[] { json }; - continue; - } - - var originalId = row.GetString("OriginalEntityId"); - var key = !string.IsNullOrEmpty(originalId) ? originalId : row.RowKey; - - pending ??= new Dictionary(StringComparer.OrdinalIgnoreCase); - if (!pending.TryGetValue(key, out var group)) - pending[key] = group = new PendingResult(totalChunks); - - group.Absorb(row); - - // Chunked but single-row results complete on their first (only) row. - if (group.IsComplete) - { - pending.Remove(key); - if (group.HasContent) yield return group.Chunks; - } - } - - // A spill row never arrived (partial write, or cleanup raced us). Emit what we have rather than - // silently dropping the result, and say so. - if (pending is { Count: > 0 }) - { - foreach (var (key, group) in pending) - { - _logger.LogWarning( - "[OrchestratorStore] Result {Key} in run {Run} is incomplete: {Have}/{Total} chunks present", - key, runName, group.PresentCount, group.TotalChunks); - if (group.HasContent) yield return group.Chunks; - } - } - } - - /// - /// A result being reassembled from chunks spread over one or more rows. Holds only this result's - /// chunks — never the s they came from. - /// - private sealed class PendingResult(int totalChunks) - { - private readonly string[] _chunks = new string[totalChunks]; - - public int TotalChunks => _chunks.Length; - public int PresentCount { get; private set; } - public bool IsComplete => PresentCount == _chunks.Length; - public bool HasContent => _chunks.Any(c => !string.IsNullOrEmpty(c)); - - /// Take any chunks this row carries that we do not already have. - public void Absorb(StoreRow row) - { - for (var i = 0; i < _chunks.Length; i++) - { - if (_chunks[i] != null) continue; - var chunk = row.GetString($"ResultJson_{i}"); - if (chunk == null) continue; - _chunks[i] = chunk; - PresentCount++; - } - } - - /// The chunks in index order. Missing chunks (incomplete result) are skipped. - public IReadOnlyList Chunks => - PresentCount == _chunks.Length ? _chunks : _chunks.Where(c => c != null).ToArray(); - } - - /// Delete all entities across the 3 tables for a completed run. - public async Task CleanupRunAsync(string runName) - { - try - { - await _store.DeleteAsync(_runsTable, "Run", runName); - await _store.DeletePartitionAsync(_tasksTable, runName); - await _store.DeletePartitionAsync(_resultsTable, runName); - - _logger.LogInformation("[OrchestratorStore] Cleaned up run: {Name}", runName); - } - catch (Exception ex) - { - _logger.LogError(ex, "[OrchestratorStore] Failed to cleanup run: {Name}", runName); - } - } - - /// The statuses a RUN ends in. Distinct from , which is about tasks. - private static bool IsTerminalRun(string? status) => - status is "Completed" or "CompletedWithErrors" or "Failed" or "Cancelled"; - - /// Keys and Timestamp only — what the orphan scan needs, and nothing a Results chunk carries. - private static readonly string[] s_keysAndTimestamp = ["PartitionKey", "RowKey", "Timestamp"]; - - /// - /// Retention sweep over the three tables. Everything removed is decided per run: - /// - /// A run in a terminal status (Completed, CompletedWithErrors, Failed, Cancelled) whose - /// CompletedUtc — or StartedUtc, for a row written before completion was stamped — is older than - /// loses its Run row and its Tasks and Results partitions. - /// A run in any other status is exempt while it is in (this - /// process is driving it). Otherwise it is abandoned once nothing about it has been written for - /// : task completion is written in one transaction with the - /// '!!run-counter' row, so that row's Timestamp is the heartbeat, and the Run row's own Timestamp - /// and StartedUtc count too. That covers runs recovery could not resume (task script gone), runs - /// queued but never dispatched, and — on a host that shares the tables — runs another process - /// stopped driving. - /// A Tasks or Results partition with no Run row at all is removed once its newest row is - /// older than . Those come from racing a - /// late status write, and from a Run row deleted while its partitions were still being written; - /// nothing else ever looked at them. - /// - /// This used to consider only Completed/CompletedWithErrors/Failed runs that carried a CompletedUtc, - /// and ran only from the startup recovery pass — so Cancelled runs, abandoned runs and orphaned - /// partitions lived forever, and on a host that was not restarted so did everything else. - /// - public async Task CleanupOldRunsAsync(TimeSpan retention, - IReadOnlySet? activeRuns = null, CancellationToken ct = default) - { - var cutoff = DateTimeOffset.UtcNow - retention; - var known = new HashSet(StringComparer.Ordinal); - var expired = new List(); - var abandoned = new List(); - - // Collect first, then delete — avoids mutating the "Run" partition while enumerating it. - var runs = new List(); - await foreach (var row in _store.QueryPartitionAsync(_runsTable, "Run", ct)) - runs.Add(row); - - foreach (var row in runs) - { - known.Add(row.RowKey); - - if (IsTerminalRun(row.GetString("Status"))) - { - var ended = row.GetDateTimeOffset("CompletedUtc") - ?? row.GetDateTimeOffset("StartedUtc") - ?? row.Timestamp; - if (ended < cutoff) expired.Add(row.RowKey); - continue; - } - - if (activeRuns != null && activeRuns.Contains(row.RowKey)) continue; - - var counter = await _store.GetAsync(_tasksTable, row.RowKey, CounterRowKey, ct); - var lastActivity = Newest(counter?.Timestamp, row.Timestamp, row.GetDateTimeOffset("StartedUtc")); - if (lastActivity < cutoff) abandoned.Add(row.RowKey); - } - - foreach (var name in expired) - await CleanupRunAsync(name); - foreach (var name in abandoned) - { - _logger.LogInformation( - "[OrchestratorStore] Run {Name} is not active and has not been written to for {Hours:F0}h — treating it as abandoned", - name, retention.TotalHours); - await CleanupRunAsync(name); - } - - var orphans = 0; - foreach (var table in new[] { _tasksTable, _resultsTable }) - orphans += await CleanupOrphanPartitionsAsync(table, known, cutoff, ct); - - if (expired.Count + abandoned.Count + orphans > 0) - _logger.LogInformation( - "[OrchestratorStore] Retention sweep removed {Expired} finished run(s), {Abandoned} abandoned run(s) and {Orphans} orphaned partition(s) older than {Hours:F0}h ({Examined} runs examined)", - expired.Count, abandoned.Count, orphans, retention.TotalHours, runs.Count); - - return new OrchestratorCleanupResult(runs.Count, expired, abandoned, orphans); - } - - private async Task CleanupOrphanPartitionsAsync(string table, HashSet knownRuns, - DateTimeOffset cutoff, CancellationToken ct) - { - // Newest row per partition that has no Run row. Only keys and Timestamp travel: a Results - // partition IS the run's payload, and reading that back every sweep would be the cost this - // sweep exists to avoid. A backend that ignores the projection still answers correctly, just - // expensively. - var newest = new Dictionary(StringComparer.Ordinal); - var unstamped = new HashSet(StringComparer.Ordinal); - await foreach (var row in _store.QueryTableAsync(table, null, s_keysAndTimestamp, ct)) - { - if (knownRuns.Contains(row.PartitionKey)) continue; - if (row.Timestamp is not { } stamped) - { - // No way to tell how old it is — never guess in the direction of deleting. - unstamped.Add(row.PartitionKey); - continue; - } - if (!newest.TryGetValue(row.PartitionKey, out var current) || stamped > current) - newest[row.PartitionKey] = stamped; - } - - var removed = 0; - foreach (var (partition, stamped) in newest) - { - if (stamped >= cutoff || unstamped.Contains(partition)) continue; - try - { - await _store.DeletePartitionAsync(table, partition, ct); - removed++; - _logger.LogInformation("[OrchestratorStore] Removed orphaned partition {Table}/{Partition}", table, partition); - } - catch (Exception ex) - { - _logger.LogWarning(ex, "[OrchestratorStore] Failed to remove orphaned partition {Table}/{Partition}", table, partition); - } - } - return removed; - } - - private static DateTimeOffset? Newest(params DateTimeOffset?[] candidates) - { - DateTimeOffset? newest = null; - foreach (var candidate in candidates) - if (candidate.HasValue && (!newest.HasValue || candidate.Value > newest.Value)) newest = candidate; - return newest; - } - - /// Split a string into chunks of at most maxChars characters, avoiding surrogate splits. - internal static List ChunkString(string value, int maxChars) - { - var chunks = new List(); - var start = 0; - - while (start < value.Length) - { - var remaining = value.Length - start; - var take = Math.Min(remaining, maxChars); - - if (take < remaining && char.IsHighSurrogate(value[start + take - 1])) - take--; - - chunks.Add(value.Substring(start, take)); - start += take; - } - - return chunks; - } - - private static int EstimateTotalChars(List chunks) - { - var total = 200; // overhead for keys + metadata properties - for (int i = 0; i < chunks.Count; i++) - total += chunks[i].Length + 20; // chunk + property name - return total; - } -} diff --git a/Services/Storage/ResultStore.cs b/Services/Storage/ResultStore.cs new file mode 100644 index 0000000..d2c2f38 --- /dev/null +++ b/Services/Storage/ResultStore.cs @@ -0,0 +1,315 @@ +using System.Runtime.CompilerServices; +using System.Text; +using Craft.Configuration; + +namespace Craft.Storage; + +/// +/// Task results for a run's aggregation, one partition per run (the run key). A result is written before +/// its task is counted done, so the aggregation never runs short of one that finished. +/// +public sealed class ResultStore +{ + private readonly ILogger _logger; + private readonly ICraftTableStore _store; + private readonly string _resultsTable; + + public ResultStore(ILogger logger, CraftSettings settings, ICraftTableStore store) + { + _logger = logger; + _store = store; + _resultsTable = $"{settings.Orchestrator.TablePrefix}TaskResults"; + } + + public Task InitializeAsync(CancellationToken ct = default) => _store.EnsureTableAsync(_resultsTable, ct); + + // ─── Result storage ─── + // Results can be large (50–150 MB for big runs). We chunk a result across multiple properties and, + // if needed, multiple rows in the same partition. These bounds are sized for Azure Table Storage + // (64 KiB/property, 1 MiB/entity); on a backend without those limits the chunking is simply + // unnecessary but still correct, and it keeps per-row payloads small (good for e.g. SQL packet size). + private const int MaxPropertyChars = 30_000; + private const int MaxEntityChars = 450_000; + + /// Store a single task result, chunking large JSON across properties/rows as needed. + public async Task StoreResultAsync(string runName, string taskId, string resultJson) + { + // Fast path: fits in a single property + if (resultJson.Length <= MaxPropertyChars) + { + var row = new StoreRow(runName, taskId) { Properties = { ["ResultJson"] = resultJson } }; + await _store.UpsertAsync(_resultsTable, row); + return; + } + + var chunks = ChunkString(resultJson, MaxPropertyChars); + + // Try to fit all chunks into a single row + if (EstimateTotalChars(chunks) <= MaxEntityChars) + { + var row = new StoreRow(runName, taskId); + for (int i = 0; i < chunks.Count; i++) + row[$"ResultJson_{i}"] = chunks[i]; + row["ResultChunkCount"] = chunks.Count; + + await _store.UpsertAsync(_resultsTable, row); + return; + } + + // Row too large — split across multiple rows + var rowIndex = 0; + var chunkIndex = 0; + + while (chunkIndex < chunks.Count) + { + var rowKey = rowIndex == 0 ? taskId : $"{taskId}-part{rowIndex}"; + var row = new StoreRow(runName, rowKey); + + if (rowIndex > 0) + { + row["OriginalEntityId"] = taskId; + row["PartIndex"] = rowIndex; + } + + var currentChars = runName.Length + rowKey.Length + 100; // overhead estimate + + while (chunkIndex < chunks.Count) + { + var chunkChars = chunks[chunkIndex].Length; + if (currentChars + chunkChars + 20 > MaxEntityChars) + break; + + row[$"ResultJson_{chunkIndex}"] = chunks[chunkIndex]; + currentChars += chunkChars + 20; + chunkIndex++; + } + + row["ResultChunkCount"] = chunks.Count; + await _store.UpsertAsync(_resultsTable, row); + rowIndex++; + } + } + + /// + /// Get all result JSON strings for a run, reassembling any chunked/multi-row results. + /// + /// Buffers every result by signature — prefer or + /// for run-sized payloads. + /// + public async Task GetResultsAsync(string runName, CancellationToken ct = default) + { + var results = new List(); + await foreach (var result in StreamResultsAsync(runName, ct)) + results.Add(result); + return results.ToArray(); + } + + /// + /// Stream each run result, reassembled, as it becomes available from the backing store. + /// + /// Nothing is buffered except spill groups still waiting for their remaining rows, so a run whose + /// results each fit in one row holds ONE row at a time regardless of how many there are. + /// + public async IAsyncEnumerable StreamResultsAsync(string runName, + [EnumeratorCancellation] CancellationToken ct = default) + { + await foreach (var chunks in StreamResultChunkGroupsAsync(runName, ct)) + yield return chunks.Count == 1 ? chunks[0] : string.Concat(chunks); + } + + /// + /// Stream all result JSON strings for a run directly to a file, reassembling any chunked/multi-row + /// results on the fly. Writes JSON Lines (NDJSON): one result per line, no enclosing array. + /// + /// Chunks are written individually, so the largest allocation this makes is one chunk + /// () — the reassembled result is never built as a string. + /// + /// JSON Lines rather than a JSON array because the consumer is PowerShell. A JSON array forces the + /// reader to hold the whole document to find where each element ends; one result per line lets + /// Invoke-CraftPostExecution walk the file with File.ReadLines and hold ONE result at a time. That + /// is the difference between a 50-150MB Large Object Heap allocation per post-execution and none. + /// It also isolates failure: a malformed result costs that result, not the entire aggregate. + /// + /// Returns the number of results written. + /// + public async Task StreamResultsToJsonLinesAsync(string runName, string filePath, + CancellationToken ct = default) + { + var count = 0; + + await using (var writer = new StreamWriter(filePath, append: false, Encoding.UTF8, bufferSize: 65536)) + { + await foreach (var chunks in StreamResultChunkGroupsAsync(runName, ct)) + { + foreach (var chunk in chunks) + await WriteSingleLineAsync(writer, chunk); + await writer.WriteAsync('\n'); + count++; + } + } + + _logger.LogInformation("[OrchestratorStore] Streamed {Count} results to {Path} for run {Name}", + count, filePath, runName); + + return count; + } + + /// + /// Write a chunk with any raw CR/LF removed, so one result stays on one line. + /// + /// Results are expected to be compact JSON on a single line — that is what Invoke-CraftTask's + /// `ConvertTo-Json -Compress` produces, and JSON escapes newlines inside strings as \n rather than + /// emitting them raw. A raw newline can therefore only appear as inter-token whitespace, which + /// carries no meaning, or in a result that was not valid JSON to begin with (a task script that + /// wrote several objects to the output stream — the runner joins those with "\n"). Dropping the + /// character is right in the first case and no worse than today's behaviour in the second, where + /// the malformed result currently takes the whole aggregate's parse down with it. + /// + /// The scan is the common-case fast path: no newline means the chunk is written untouched, with + /// no copy and no per-character work beyond the search itself. + /// + private static async Task WriteSingleLineAsync(StreamWriter writer, string chunk) + { + var start = 0; + int idx; + + while ((idx = chunk.AsSpan(start).IndexOfAny('\r', '\n')) >= 0) + { + var abs = start + idx; + if (abs > start) + await writer.WriteAsync(chunk.AsMemory(start, abs - start)); + start = abs + 1; + } + + if (start == 0) + await writer.WriteAsync(chunk); + else if (start < chunk.Length) + await writer.WriteAsync(chunk.AsMemory(start)); + } + + /// + /// The shared core: yields each logical result as its ordered chunk list, as soon as that result is + /// complete, and drops every row it has finished with. + /// + /// This used to be LoadResultGroupsAsync, which materialized EVERY result row for the run into a + /// dictionary before a single byte was written — so the callers named "stream" held the entire + /// payload (as UTF-16, ~2x the stored size) before they started. For a 738-task run whose aggregate + /// is 50-150MB that was a few hundred MB against a 2398MB heap cap, concurrently per post-execution. + /// + /// Rows are grouped by (OriginalEntityId ?? RowKey) and completed by chunk count rather than by + /// arrival order, so this makes no assumption about the order + /// returns rows in — the interface promises none. + /// + private async IAsyncEnumerable> StreamResultChunkGroupsAsync(string runName, + [EnumeratorCancellation] CancellationToken ct = default) + { + // Allocated only if this run actually has a result too large for a single row. + Dictionary? pending = null; + + await foreach (var row in _store.QueryPartitionAsync(_resultsTable, runName, ct)) + { + var totalChunks = row.GetInt32("ResultChunkCount") ?? 0; + + // Fast path: the whole result is one property on this row. Emit and release it. + if (totalChunks == 0) + { + var json = row.GetString("ResultJson"); + if (!string.IsNullOrEmpty(json)) yield return new[] { json }; + continue; + } + + var originalId = row.GetString("OriginalEntityId"); + var key = !string.IsNullOrEmpty(originalId) ? originalId : row.RowKey; + + pending ??= new Dictionary(StringComparer.OrdinalIgnoreCase); + if (!pending.TryGetValue(key, out var group)) + pending[key] = group = new PendingResult(totalChunks); + + group.Absorb(row); + + // Chunked but single-row results complete on their first (only) row. + if (group.IsComplete) + { + pending.Remove(key); + if (group.HasContent) yield return group.Chunks; + } + } + + // A spill row never arrived (partial write, or cleanup raced us). Emit what we have rather than + // silently dropping the result, and say so. + if (pending is { Count: > 0 }) + { + foreach (var (key, group) in pending) + { + _logger.LogWarning( + "[OrchestratorStore] Result {Key} in run {Run} is incomplete: {Have}/{Total} chunks present", + key, runName, group.PresentCount, group.TotalChunks); + if (group.HasContent) yield return group.Chunks; + } + } + } + + /// + /// A result being reassembled from chunks spread over one or more rows. Holds only this result's + /// chunks — never the s they came from. + /// + private sealed class PendingResult(int totalChunks) + { + private readonly string[] _chunks = new string[totalChunks]; + + public int TotalChunks => _chunks.Length; + public int PresentCount { get; private set; } + public bool IsComplete => PresentCount == _chunks.Length; + public bool HasContent => _chunks.Any(c => !string.IsNullOrEmpty(c)); + + /// Take any chunks this row carries that we do not already have. + public void Absorb(StoreRow row) + { + for (var i = 0; i < _chunks.Length; i++) + { + if (_chunks[i] != null) continue; + var chunk = row.GetString($"ResultJson_{i}"); + if (chunk == null) continue; + _chunks[i] = chunk; + PresentCount++; + } + } + + /// The chunks in index order. Missing chunks (incomplete result) are skipped. + public IReadOnlyList Chunks => + PresentCount == _chunks.Length ? _chunks : _chunks.Where(c => c != null).ToArray(); + } + + /// Drop a run's results once its aggregation has read them. + public Task DeleteRunAsync(string runKey, CancellationToken ct = default) => + _store.DeletePartitionAsync(_resultsTable, runKey, ct); + + /// Split a string into chunks of at most maxChars characters, avoiding surrogate splits. + internal static List ChunkString(string value, int maxChars) + { + var chunks = new List(); + var start = 0; + + while (start < value.Length) + { + var remaining = value.Length - start; + var take = Math.Min(remaining, maxChars); + + if (take < remaining && char.IsHighSurrogate(value[start + take - 1])) + take--; + + chunks.Add(value.Substring(start, take)); + start += take; + } + + return chunks; + } + + private static int EstimateTotalChars(List chunks) + { + var total = 200; // overhead for keys + metadata properties + for (int i = 0; i < chunks.Count; i++) + total += chunks[i].Length + 20; // chunk + property name + return total; + } +} diff --git a/Services/Storage/ResultWrite.cs b/Services/Storage/ResultWrite.cs deleted file mode 100644 index dbc8e6a..0000000 --- a/Services/Storage/ResultWrite.cs +++ /dev/null @@ -1,10 +0,0 @@ -namespace Craft.Storage; - -/// -/// A small task result queued for the coalescing status writer. Only results that fit a single Azure -/// Table property travel this way; larger results keep the chunked, directly-awaited -/// path. Written to the Results table before the -/// task's terminal status marker in the same flush, so a result is always durable before its task is -/// counted done. -/// -public record ResultWrite(string RunName, string TaskId, string ResultJson); diff --git a/Services/Storage/TaskStatusWrite.cs b/Services/Storage/TaskStatusWrite.cs deleted file mode 100644 index 6c22012..0000000 --- a/Services/Storage/TaskStatusWrite.cs +++ /dev/null @@ -1,12 +0,0 @@ -namespace Craft.Storage; - -/// -/// An immutable snapshot of a task's status for the coalescing batched writer. -/// -/// Every column the task row carries must appear here. The store upserts with -/// TableUpdateMode.Replace, so any column missing from this snapshot is ERASED by the next -/// status transition — which is how a per-task Priority override would silently vanish the first -/// time the task moved to Running. -/// -public record TaskStatusWrite(string RunName, string TaskId, string Status, string? ParametersJson, - int AttemptCount, string? LastError, DateTime? CompletedUtc, int? Priority, int Sequence = 0); diff --git a/Services/Storage/WorkStore.cs b/Services/Storage/WorkStore.cs new file mode 100644 index 0000000..acf068f --- /dev/null +++ b/Services/Storage/WorkStore.cs @@ -0,0 +1,782 @@ +using System.Collections.Concurrent; +using System.Globalization; +using System.Text.Json; +using Craft.Configuration; + +namespace Craft.Storage; + +/// +/// Durable orchestration state. Each run is one partition of the Work table, and every task's whole state +/// is one row in it whose RowKey prefix is its state: +/// +/// $run header: identity, options and the Total/Done counts +/// T|{seq} payload (immutable) +/// P|{seq} pending R|{seq} running (Owner, LeaseUntil) D|{seq} done +/// C|{child} a child run the parent waits for +/// +/// So every state change is one partition transaction, the counts can never drift from the rows, and +/// nothing has to be reconciled: a task is exactly where its row says. A worker that dies leaves an R row +/// whose lease lapses, and the next claim takes it back. +/// +/// Three small tables sit beside it: Ready (one row per run with work, ordered by band then start time, +/// read by the scheduler), Names (latest run per name) and Finished (completion order, for retention). +/// All three are hints derived from the Work rows: a stale one costs a read, never a wrong answer. +/// +public sealed class WorkStore +{ + public const string HeaderKey = "$run"; + + /// The PostExecution parameters, kept off the header: the header is rewritten in every finish + /// transaction, where rows cannot be split, and the parameters can outgrow a property. + private const string PostExecKey = "$post"; + public const int AggregateSeq = 99_999_999; + + /// Tasks per claim or completion transaction: two ops each plus the header. + public const int MaxPerTransaction = 49; + + private readonly int MaxAttempts; + private const int ConflictRetries = 16; + + private readonly ICraftTableStore _store; + private readonly ILogger _logger; + private readonly PartitionRateLimiter _rate; + private readonly string _work, _ready, _names, _finished, _results; + private readonly string[] _legacyTables; + private volatile bool _initialized; + + public WorkStore(ILogger logger, CraftSettings settings, ICraftTableStore store, + PartitionRateLimiter? rate = null) + { + _logger = logger; + _store = store; + _rate = rate ?? new PartitionRateLimiter(); + MaxAttempts = Math.Max(1, settings.Orchestrator.MaxRetries); + var p = settings.Orchestrator.TablePrefix; + _work = $"{p}Work"; + _ready = $"{p}Ready"; + _names = $"{p}Names"; + _finished = $"{p}Finished"; + _results = $"{p}TaskResults"; + _legacyTables = [$"{p}Queue", $"{p}QueueIndex", $"{p}Tasks", $"{p}Runs", $"{p}Results"]; + } + + public string ResultsTable => _results; + + /// Called after every finish transaction that reached the barrier or completed a run, whoever made it. + public Func? AfterFinish { get; set; } + + public async Task InitializeAsync(CancellationToken ct = default) + { + if (_initialized) return; + foreach (var t in new[] { _work, _ready, _names, _finished, _results }) + await _store.EnsureTableAsync(t, ct); + _initialized = true; + } + + /// The earlier orchestration tables, which this store replaces outright. Their in-flight work is + /// dropped; the timers that created it create it again. + public IReadOnlyList LegacyTables => _legacyTables; + + // ── keys ── + + public static string RunKeyFor(string name, DateTime startedUtc) => + $"{name}~{startedUtc.Ticks.ToString("x", CultureInfo.InvariantCulture)}"; + + private static string Seq(int seq) => seq.ToString("D8", CultureInfo.InvariantCulture); + private static string Key(char state, int seq) => $"{state}|{Seq(seq)}"; + private static int SeqOf(string rowKey) => int.Parse(rowKey.AsSpan(2), CultureInfo.InvariantCulture); + private static string ReadyPartition(int band) => "P" + Math.Clamp(band, 0, 99).ToString("D2", CultureInfo.InvariantCulture); + private static string ReadyKey(RunHeader h) => $"{h.StartedUtc.Ticks.ToString("D19", CultureInfo.InvariantCulture)}|{h.RunKey}"; + + /// A run's rows in one state, in seq order; 0 means all. + private async IAsyncEnumerable Range(string runKey, char state, int max = 0, + [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) + { + var count = 0; + await foreach (var row in _store.QueryRowKeyRangeAsync(_work, runKey, $"{state}|", $"{state}}}", null, ct)) + { + yield return row; + if (max > 0 && ++count >= max) yield break; + } + } + + // ── create ── + + public sealed record NewTask(string TaskId, Dictionary Parameters); + + /// + /// Persist a run. Payload and pending rows go first and the header last, so a crash part-way leaves rows + /// that no Ready entry points at, never a visible run with tasks missing. + /// + public async Task CreateRunAsync(RunHeader header, IReadOnlyList tasks, CancellationToken ct = default) + { + await InitializeAsync(ct); + var payload = new List(tasks.Count); + var pending = new List(tasks.Count); + for (var i = 0; i < tasks.Count; i++) + { + payload.Add(new StoreRow(header.RunKey, Key('T', i)) + { + Properties = { ["TaskId"] = tasks[i].TaskId, ["ParametersJson"] = JsonSerializer.Serialize(tasks[i].Parameters, RunHeader.Json) } + }); + pending.Add(PendingRow(header.RunKey, i, tasks[i].TaskId, 0)); + } + + await _rate.TakeAsync(header.RunKey, tasks.Count * 2 + 1, ct); + await _store.UpsertBatchAsync(_work, header.RunKey, payload, ct); + await _store.UpsertBatchAsync(_work, header.RunKey, pending, ct); + if (header.PostExecParametersJson is { } post) + await _store.UpsertAsync(_work, new StoreRow(header.RunKey, PostExecKey) { Properties = { ["Json"] = post } }, ct); + header.Total = tasks.Count; + await _store.UpsertAsync(_work, header.ToRow(), ct); + + await _store.UpsertAsync(_names, new StoreRow("N", header.Name) { Properties = { ["RunKey"] = header.RunKey } }, ct); + await PublishReadyAsync(header, ct); + return (await GetRunAsync(header.RunKey, ct))!; + } + + private static StoreRow PendingRow(string runKey, int seq, string taskId, int attempt) => new(runKey, Key('P', seq)) + { + Properties = { ["TaskId"] = taskId, ["Attempt"] = attempt } + }; + + public Task PublishReadyAsync(RunHeader h, CancellationToken ct = default) => + _store.UpsertAsync(_ready, new StoreRow(ReadyPartition(h.Priority), ReadyKey(h)) + { + Properties = + { + ["RunKey"] = h.RunKey, ["Name"] = h.Name, ["Total"] = h.Total, ["Done"] = h.Done, ["Failed"] = h.Failed, + ["Cancelled"] = h.Cancelled, ["Reference"] = h.Reference, + } + }, ct); + + // ── read ── + + public async Task GetRunAsync(string runKey, CancellationToken ct = default) + { + var row = await _store.GetAsync(_work, runKey, HeaderKey, ct); + return row == null ? null : RunHeader.FromRow(row); + } + + /// The latest run with this name, active or finished. + public async Task GetRunByNameAsync(string name, CancellationToken ct = default) + { + var row = await _store.GetAsync(_names, "N", name, ct); + return row?.GetString("RunKey") is { } key ? await GetRunAsync(key, ct) : null; + } + + public async Task GetPostExecParametersAsync(string runKey, CancellationToken ct = default) => + (await _store.GetAsync(_work, runKey, PostExecKey, ct))?.GetString("Json"); + + public async Task?> GetPayloadAsync(string runKey, int seq, CancellationToken ct = default) + { + var row = await _store.GetAsync(_work, runKey, Key('T', seq), ct); + var json = row?.GetString("ParametersJson"); + if (json == null) return null; + try { return JsonSerializer.Deserialize>(json, RunHeader.Json) ?? []; } + catch (JsonException) { return []; } + } + + public sealed record ReadyEntry(int Band, string RunKey, string Name, int Total, int Done, DateTime StartedUtc, + string? Reference = null, int Failed = 0, int Cancelled = 0); + + /// Runs with work, best band first and oldest first within it. + public async IAsyncEnumerable ReadReadyAsync(int pageSize = 32, + [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) + { + await foreach (var row in _store.QueryTableAsync(_ready, null, pageSize, ct)) + { + if (row.GetString("RunKey") is not { } key) continue; + var band = int.TryParse(row.PartitionKey.AsSpan(1), NumberStyles.None, CultureInfo.InvariantCulture, out var b) ? b : 99; + var ticks = long.TryParse(row.RowKey.AsSpan(0, Math.Min(19, row.RowKey.Length)), NumberStyles.None, CultureInfo.InvariantCulture, out var t) ? t : 0; + yield return new ReadyEntry(band, key, row.GetString("Name") ?? key, row.GetInt32("Total") ?? 0, + row.GetInt32("Done") ?? 0, new DateTime(ticks, DateTimeKind.Utc), row.GetString("Reference"), + row.GetInt32("Failed") ?? 0, row.GetInt32("Cancelled") ?? 0); + } + } + + public sealed record TaskRow(int Seq, string TaskId, char State, string? Status, int Attempt, string? Owner, + DateTimeOffset? LeaseUntil, string? LastError); + + private static TaskRow ToTask(StoreRow r) => new(SeqOf(r.RowKey), r.GetString("TaskId") ?? "", r.RowKey[0], + r.GetString("Status"), r.GetInt32("Attempt") ?? 0, r.GetString("Owner"), r.GetDateTimeOffset("LeaseUntil"), + r.GetString("LastError")); + + /// Every task row of a run (P, R and D), for status views and cancel lookups. + public async Task> GetTasksAsync(string runKey, char? state = null, CancellationToken ct = default) + { + var rows = new List(); + foreach (var s in state is { } one ? [one] : new[] { 'P', 'R', 'D' }) + await foreach (var r in Range(runKey, s, ct: ct)) rows.Add(ToTask(r)); + return rows; + } + + // ── claim ── + + public sealed record ClaimedTask(string RunKey, int Seq, string TaskId, int Attempt); + + /// + /// Move up to pending tasks to running under , plus, with + /// , running tasks whose lease lapsed. A task claimed for the + /// th time and lapsing again is failed instead of claimed. One transaction; a lost + /// race returns empty and the caller moves on. + /// + public async Task> ClaimAsync(string runKey, int max, string owner, TimeSpan lease, + bool reclaimExpired, CancellationToken ct = default) + { + max = Math.Min(max, MaxPerTransaction); + if (max <= 0) return []; + + var now = DateTimeOffset.UtcNow; + var expired = new List(); + if (reclaimExpired) + { + await foreach (var r in Range(runKey, 'R', ct: ct)) + if (r.GetDateTimeOffset("LeaseUntil") is not { } until || until <= now) + { + expired.Add(r); + if (expired.Count >= max) break; + } + } + var pending = new List(); + if (expired.Count < max) + await foreach (var r in Range(runKey, 'P', max - expired.Count, ct: ct)) pending.Add(r); + if (pending.Count + expired.Count == 0) return []; + + var leaseUntil = now.Add(lease); + var ops = new List(); + var claimed = new List(); + var poisoned = new List<(StoreRow Row, int Attempt)>(); + + foreach (var r in pending) + { + var seq = SeqOf(r.RowKey); + var attempt = (r.GetInt32("Attempt") ?? 0) + 1; + ops.Add(StoreOp.Delete(r)); + ops.Add(StoreOp.Insert(RunningRow(runKey, seq, r.GetString("TaskId")!, attempt, owner, leaseUntil))); + claimed.Add(new ClaimedTask(runKey, seq, r.GetString("TaskId")!, attempt)); + } + foreach (var r in expired) + { + var attempt = r.GetInt32("Attempt") ?? 0; + if (attempt >= MaxAttempts) { poisoned.Add((r, attempt)); continue; } + var seq = SeqOf(r.RowKey); + ops.Add(StoreOp.Replace(new StoreRow(runKey, r.RowKey) + { + ETag = r.ETag, + Properties = RunningRow(runKey, seq, r.GetString("TaskId")!, attempt + 1, owner, leaseUntil).Properties + })); + claimed.Add(new ClaimedTask(runKey, seq, r.GetString("TaskId")!, attempt + 1)); + } + + if (poisoned.Count > 0) + { + var finish = poisoned.Select(p => new Finish(SeqOf(p.Row.RowKey), "Failed", + $"Interrupted {p.Attempt} times without completing")).ToList(); + await FinishAsync(runKey, finish, null, ct); + } + if (ops.Count == 0) return []; + + await _rate.TakeAsync(runKey, ops.Count, ct); + return await _store.TrySubmitAsync(_work, runKey, ops, ct) ? claimed : []; + } + + /// + /// Claim the next step of a sequential run for , taking or keeping the run's driver + /// lease in the same transaction. Empty while another owner's driver lease is live, so a run never has two + /// drivers; a lapsed driver's step is reclaimed with it. + /// + /// True for the driver claiming its own next step; false (the pump) defers to any live driver. + public async Task ClaimSequentialAsync(string runKey, string owner, TimeSpan lease, bool continuing = false, + CancellationToken ct = default) + { + var headerRow = await _store.GetAsync(_work, runKey, HeaderKey, ct); + if (headerRow == null) return null; + var header = RunHeader.FromRow(headerRow); + var now = DateTimeOffset.UtcNow; + if (header.DriverOwner != null && header.DriverLease > now && !(continuing && header.DriverOwner == owner)) return null; + + StoreRow? row = null; + await foreach (var r in Range(runKey, 'R', ct: ct)) { row = r; break; } + var reclaiming = row != null; + if (row == null) await foreach (var r in Range(runKey, 'P', 1, ct)) { row = r; break; } + if (row == null) return null; + + var seq = SeqOf(row.RowKey); + var attempt = (row.GetInt32("Attempt") ?? 0) + 1; + if (reclaiming && attempt > MaxAttempts) + { + await FinishAsync(runKey, [new Finish(seq, "Failed", $"Interrupted {attempt - 1} times without completing")], 'R', ct); + return await ClaimSequentialAsync(runKey, owner, lease, continuing, ct); + } + + var until = now.Add(lease); + header.DriverOwner = owner; + header.DriverLease = until; + var running = RunningRow(runKey, seq, row.GetString("TaskId")!, attempt, owner, until); + var ops = new List { StoreOp.Replace(header.ToRow(headerRow.ETag)) }; + if (reclaiming) ops.Add(StoreOp.Replace(new StoreRow(runKey, row.RowKey) { ETag = row.ETag, Properties = running.Properties })); + else { ops.Add(StoreOp.Delete(row)); ops.Add(StoreOp.Insert(running)); } + + await _rate.TakeAsync(runKey, ops.Count, ct); + return await _store.TrySubmitAsync(_work, runKey, ops, ct) + ? new ClaimedTask(runKey, seq, row.GetString("TaskId")!, attempt) + : null; + } + + /// Give up a sequential run's driver lease so the next step can be claimed by anyone. + public async Task ReleaseDriverAsync(string runKey, string owner, CancellationToken ct = default) + { + for (var attempt = 0; attempt < ConflictRetries; attempt++) + { + var headerRow = await _store.GetAsync(_work, runKey, HeaderKey, ct); + if (headerRow == null) return; + var header = RunHeader.FromRow(headerRow); + if (header.DriverOwner != owner) return; + header.DriverOwner = null; + header.DriverLease = null; + if (await _store.TrySubmitAsync(_work, runKey, [StoreOp.Replace(header.ToRow(headerRow.ETag))], ct)) return; + } + } + + private static StoreRow RunningRow(string runKey, int seq, string taskId, int attempt, string owner, DateTimeOffset leaseUntil) => + new(runKey, Key('R', seq)) + { + Properties = { ["TaskId"] = taskId, ["Attempt"] = attempt, ["Owner"] = owner, ["LeaseUntil"] = leaseUntil } + }; + + /// + /// Hand a running task back to pending without counting the attempt as spent: a shutdown that interrupted + /// it, or an aggregation that failed and has attempts left. Only if still holds it. + /// + public async Task ReleaseAsync(string runKey, int seq, string owner, bool refundAttempt, CancellationToken ct = default) + { + var row = await _store.GetAsync(_work, runKey, Key('R', seq), ct); + if (row == null || row.GetString("Owner") != owner) return false; + var attempt = (row.GetInt32("Attempt") ?? 1) - (refundAttempt ? 1 : 0); + await _rate.TakeAsync(runKey, 2, ct); + return await _store.TrySubmitAsync(_work, runKey, + [StoreOp.Delete(row), StoreOp.Insert(PendingRow(runKey, seq, row.GetString("TaskId")!, attempt))], ct); + } + + /// Push the lease out on claims this owner still holds. Returns the claims it no longer holds. + public async Task> RenewAsync(IReadOnlyList claims, string owner, TimeSpan lease, + CancellationToken ct = default) + { + var lost = new List(); + var leaseUntil = DateTimeOffset.UtcNow.Add(lease); + foreach (var byRun in claims.GroupBy(c => c.RunKey)) + { + var ops = new List(); + foreach (var c in byRun) + { + var row = await _store.GetAsync(_work, c.RunKey, Key('R', c.Seq), ct); + if (row == null) continue; + if (row.GetString("Owner") != owner) { lost.Add(c); continue; } + row["LeaseUntil"] = leaseUntil; + ops.Add(StoreOp.Replace(row)); + } + foreach (var chunk in ops.Chunk(100)) + { + await _rate.TakeAsync(byRun.Key, chunk.Length, ct); + if (!await _store.TrySubmitAsync(_work, byRun.Key, chunk, ct)) + _logger.LogWarning("[WorkStore] Lease renewal for {Run} lost a race; the next renewal retries", byRun.Key); + } + } + return lost; + } + + // ── finish ── + + /// A task reaching a terminal status. set means "only if I still hold it". + public sealed record Finish(int Seq, string Status, string? Error = null, string? Owner = null, string? ChildKey = null); + + /// What a finish did to the run: the header after it, and whether this was the transaction that + /// completed the run's tasks (the barrier) or its aggregation. + public sealed record FinishOutcome(RunHeader Header, bool ReachedBarrier, bool Completed, int Applied); + + /// + /// Move tasks (R or, for a cancel, P) and child placeholders to done and update the counts, in one + /// transaction per chunk. The chunk that brings Done to Total is the barrier: it inserts the aggregation + /// task when the run has one, or completes the run when it does not. Finishing the aggregation task + /// completes the run. + /// + public async Task FinishAsync(string runKey, IReadOnlyList finishes, char? fromState = 'R', + CancellationToken ct = default) + { + FinishOutcome? outcome = null; + foreach (var chunk in finishes.Chunk(MaxPerTransaction)) + outcome = await FinishChunkAsync(runKey, chunk, fromState, ct) ?? outcome; + return outcome; + } + + private async Task FinishChunkAsync(string runKey, IReadOnlyList chunk, char? fromState, + CancellationToken ct) + { + for (var attempt = 0; attempt < ConflictRetries; attempt++) + { + var headerRow = await _store.GetAsync(_work, runKey, HeaderKey, ct); + if (headerRow == null) return null; + var header = RunHeader.FromRow(headerRow); + var ops = new List(); + var applied = 0; + var barrier = false; + var completed = false; + + foreach (var f in chunk) + { + if (f.ChildKey is { } child) + { + var placeholder = await _store.GetAsync(_work, runKey, $"C|{child}", ct); + if (placeholder == null) continue; + ops.Add(StoreOp.Delete(placeholder)); + header.Done++; + if (f.Status != "Completed") header.Failed++; + applied++; + continue; + } + + StoreRow? row = null; + foreach (var state in fromState is { } s ? [s] : new[] { 'R', 'P' }) + { + row = await _store.GetAsync(_work, runKey, Key(state, f.Seq), ct); + if (row != null) break; + } + if (row == null) continue; + if (f.Owner != null && row.RowKey[0] == 'R' && row.GetString("Owner") != f.Owner) continue; + + ops.Add(StoreOp.Delete(row)); + if (f.Seq == AggregateSeq) + { + header.PostExecStatus = f.Status == "Completed" ? "Completed" : "Failed"; + completed = true; + } + else + { + ops.Add(StoreOp.Insert(new StoreRow(runKey, Key('D', f.Seq)) + { + Properties = + { + ["TaskId"] = row.GetString("TaskId"), + ["Status"] = f.Status, + ["LastError"] = f.Error, + ["Attempt"] = row.GetInt32("Attempt") ?? 0, + ["CompletedUtc"] = DateTimeOffset.UtcNow, + } + })); + header.Done++; + if (f.Status == "Failed") header.Failed++; + else if (f.Status == "Cancelled") header.Cancelled++; + } + applied++; + } + + if (applied == 0) return new FinishOutcome(header, false, false, 0); + + if (!completed && header.Phase == RunPhase.Tasks && header.Done >= header.Total) + { + barrier = true; + if (header.HasPostExec) + { + header.Phase = RunPhase.Aggregate; + header.PostExecStatus = "Pending"; + ops.Add(StoreOp.Insert(PendingRow(runKey, AggregateSeq, "PostExecution", 0))); + } + else completed = true; + } + if (completed) + { + header.Phase = RunPhase.Done; + header.Status = header.Failed > 0 || header.Cancelled > 0 ? "CompletedWithErrors" : "Completed"; + header.CompletedUtc = DateTime.UtcNow; + } + ops.Add(StoreOp.Replace(header.ToRow(headerRow.ETag))); + + await _rate.TakeAsync(runKey, ops.Count, ct); + if (await _store.TrySubmitAsync(_work, runKey, ops, ct)) + { + var after = (await GetRunAsync(runKey, ct)) ?? header; + if (completed) await RetireAsync(after, ct); + else await PublishReadyAsync(after, ct); + var outcome = new FinishOutcome(after, barrier, completed, applied); + if ((barrier || completed) && AfterFinish is { } hook) + { + try { await hook(outcome); } + catch (Exception ex) { _logger.LogWarning(ex, "[WorkStore] After-finish handling for {Run} failed", runKey); } + } + return outcome; + } + } + + _logger.LogWarning("[WorkStore] Finishing {Count} task(s) of {Run} kept losing races; the next attempt retries", chunk.Count, runKey); + return null; + } + + /// Take a finished run off the Ready list and record it for retention. + private async Task RetireAsync(RunHeader h, CancellationToken ct) + { + _rate.Forget(h.RunKey); + await _store.DeleteAsync(_ready, ReadyPartition(h.Priority), ReadyKey(h), ct); + var done = (h.CompletedUtc ?? DateTime.UtcNow).Ticks.ToString("D19", CultureInfo.InvariantCulture); + await _store.UpsertAsync(_finished, new StoreRow("F", $"{done}|{h.RunKey}") { Properties = { ["RunKey"] = h.RunKey } }, ct); + } + + /// Remove a stale Ready entry (its run is gone or finished). + public Task DropReadyAsync(ReadyEntry e, CancellationToken ct = default) => + _store.DeleteAsync(_ready, ReadyPartition(e.Band), $"{e.StartedUtc.Ticks.ToString("D19", CultureInfo.InvariantCulture)}|{e.RunKey}", ct); + + // ── children ── + + /// + /// Make wait for a child run. Only while the parent is still running its tasks: + /// a run queued from an aggregation is not a child. + /// + public async Task AddChildAsync(string parentKey, string childKey, CancellationToken ct = default) + { + for (var attempt = 0; attempt < ConflictRetries; attempt++) + { + var headerRow = await _store.GetAsync(_work, parentKey, HeaderKey, ct); + if (headerRow == null) return false; + var header = RunHeader.FromRow(headerRow); + if (header.Phase != RunPhase.Tasks) return false; + header.Total++; + var ops = new List + { + StoreOp.Insert(new StoreRow(parentKey, $"C|{childKey}") { Properties = { ["Child"] = childKey } }), + StoreOp.Replace(header.ToRow(headerRow.ETag)), + }; + if (await _store.TrySubmitAsync(_work, parentKey, ops, ct)) return true; + } + return false; + } + + // ── cancel ── + + /// Cancel every pending task of a run. Running tasks finish; the barrier then fires as usual. + public async Task<(int Cancelled, FinishOutcome? Outcome)> CancelPendingAsync(string runKey, CancellationToken ct = default) + { + var cancelled = 0; + FinishOutcome? outcome = null; + while (true) + { + var page = new List(); + await foreach (var r in Range(runKey, 'P', MaxPerTransaction, ct: ct)) + if (SeqOf(r.RowKey) != AggregateSeq) page.Add(new Finish(SeqOf(r.RowKey), "Cancelled", "Cancelled by user")); + if (page.Count == 0) return (cancelled, outcome); + var result = await FinishAsync(runKey, page, 'P', ct); + if (result == null || result.Applied == 0) return (cancelled, outcome); + cancelled += result.Applied; + outcome = result; + } + } + + // ── retention ── + + /// Delete runs that finished before the retention cutoff: their Work and Results partitions and + /// index rows. Reads the Finished index oldest first and stops at the cutoff. + public async Task SweepFinishedAsync(TimeSpan retention, CancellationToken ct = default) + { + var cutoff = (DateTime.UtcNow - retention).Ticks.ToString("D19", CultureInfo.InvariantCulture); + var expired = new List(); + await foreach (var row in _store.QueryRowKeyRangeAsync(_finished, "F", "", cutoff, null, ct)) + expired.Add(row); + + foreach (var row in expired) + { + var runKey = row.GetString("RunKey") ?? row.RowKey[20..]; + await DeleteRunAsync(runKey, ct); + await _store.DeleteAsync(_finished, "F", row.RowKey, ct); + } + return expired.Count; + } + + /// Delete a run's partitions and its name entry if it still points at this run. + public async Task DeleteRunAsync(string runKey, CancellationToken ct = default) + { + var header = await GetRunAsync(runKey, ct); + await _store.DeletePartitionAsync(_work, runKey, ct); + await _store.DeletePartitionAsync(_results, runKey, ct); + if (header == null) return; + var name = await _store.GetAsync(_names, "N", header.Name, ct); + if (name?.GetString("RunKey") == runKey) await _store.DeleteAsync(_names, "N", header.Name, ct); + await _store.DeleteAsync(_ready, ReadyPartition(header.Priority), ReadyKey(header), ct); + } + + /// Delete the tables of the previous orchestration design. Their in-flight work is dropped; the timers + /// that created it create it again. + public async Task DropLegacyTablesAsync(CancellationToken ct = default) + { + foreach (var t in _legacyTables) + { + try { await _store.DeleteTableAsync(t, ct); } + catch (Exception ex) { _logger.LogWarning(ex, "[WorkStore] Could not drop legacy table {Table}", t); } + } + } + + /// Flag a run as cancelled, so its running tasks and a sequential driver stop at their next step. + public Task RequestCancelAsync(string runKey, CancellationToken ct = default) => + UpdateHeaderAsync(runKey, h => { h.CancelRequested = true; return true; }, ct); + + /// Move a run to another priority band (its Ready entry moves with it). + public async Task SetPriorityAsync(string runKey, int priority, CancellationToken ct = default) + { + RunHeader? before = null; + var ok = await UpdateHeaderAsync(runKey, h => + { + before ??= new RunHeader { RunKey = h.RunKey, Name = h.Name, Priority = h.Priority, StartedUtc = h.StartedUtc }; + if (h.IsFinished) return false; + h.Priority = priority; + return true; + }, ct); + if (!ok || before == null) return false; + await _store.DeleteAsync(_ready, ReadyPartition(before.Priority), ReadyKey(before), ct); + if (await GetRunAsync(runKey, ct) is { IsFinished: false } after) await PublishReadyAsync(after, ct); + return true; + } + + private async Task UpdateHeaderAsync(string runKey, Func change, CancellationToken ct) + { + for (var attempt = 0; attempt < ConflictRetries; attempt++) + { + var row = await _store.GetAsync(_work, runKey, HeaderKey, ct); + if (row == null) return false; + var header = RunHeader.FromRow(row); + if (!change(header)) return false; + if (await _store.TrySubmitAsync(_work, runKey, [StoreOp.Replace(header.ToRow(row.ETag))], ct)) return true; + } + return false; + } +} + +public enum RunPhase { Tasks, Aggregate, Done } + +/// A run's header row: identity, options and counts. +public sealed class RunHeader +{ + internal static readonly JsonSerializerOptions Json = new() { PropertyNamingPolicy = JsonNamingPolicy.CamelCase }; + + public string RunKey { get; init; } = ""; + public string Name { get; init; } = ""; + public string Status { get; set; } = "Running"; + public RunPhase Phase { get; set; } = RunPhase.Tasks; + public int Priority { get; set; } = 4; + public DateTime StartedUtc { get; init; } + public DateTime? CompletedUtc { get; set; } + public string? TaskScriptName { get; init; } + public string? PostExecFunctionName { get; init; } + /// Set on creation only; read back with . + public string? PostExecParametersJson { get; init; } + public string? PostExecStatus { get; set; } + public string? ParentRunKey { get; init; } + + /// The placeholder this run fills in its parent (C|{key}), completed when this run finishes. + public string? ParentChildKey { get; init; } + public string? Reference { get; init; } + + /// Sequential runs: the worker driving the run, and until when. One driver at a time. + public string? DriverOwner { get; set; } + public DateTimeOffset? DriverLease { get; set; } + public bool Sequential { get; init; } + public bool CancelRequested { get; set; } + public int Total { get; set; } + public int Done { get; set; } + public int Failed { get; set; } + public int Cancelled { get; set; } + + public bool HasPostExec => !string.IsNullOrEmpty(PostExecFunctionName); + public bool IsFinished => Phase == RunPhase.Done; + + public StoreRow ToRow(string? etag = null) => new(RunKey, WorkStore.HeaderKey) + { + ETag = etag, + Properties = + { + ["Name"] = Name, + ["Status"] = Status, + ["Phase"] = Phase.ToString(), + ["Priority"] = Priority, + ["StartedUtc"] = new DateTimeOffset(DateTime.SpecifyKind(StartedUtc, DateTimeKind.Utc)), + ["CompletedUtc"] = CompletedUtc is { } c ? new DateTimeOffset(DateTime.SpecifyKind(c, DateTimeKind.Utc)) : (DateTimeOffset?)null, + ["TaskScriptName"] = TaskScriptName, + ["PostExecFunctionName"] = PostExecFunctionName, + ["PostExecStatus"] = PostExecStatus, + ["ParentRunKey"] = ParentRunKey, + ["ParentChildKey"] = ParentChildKey, + ["Reference"] = Reference, + ["DriverOwner"] = DriverOwner, + ["DriverLease"] = DriverLease, + ["Sequential"] = Sequential ? 1 : 0, + ["CancelRequested"] = CancelRequested ? 1 : 0, + ["Total"] = Total, + ["Done"] = Done, + ["Failed"] = Failed, + ["Cancelled"] = Cancelled, + } + }; + + public static RunHeader FromRow(StoreRow r) => new() + { + RunKey = r.PartitionKey, + Name = r.GetString("Name") ?? r.PartitionKey, + Status = r.GetString("Status") ?? "Running", + Phase = Enum.TryParse(r.GetString("Phase"), out var phase) ? phase : RunPhase.Tasks, + Priority = r.GetInt32("Priority") ?? 4, + StartedUtc = r.GetDateTimeOffset("StartedUtc")?.UtcDateTime ?? DateTime.UnixEpoch, + CompletedUtc = r.GetDateTimeOffset("CompletedUtc")?.UtcDateTime, + TaskScriptName = r.GetString("TaskScriptName"), + PostExecFunctionName = r.GetString("PostExecFunctionName"), + PostExecStatus = r.GetString("PostExecStatus"), + ParentRunKey = r.GetString("ParentRunKey"), + ParentChildKey = r.GetString("ParentChildKey"), + Reference = r.GetString("Reference"), + DriverOwner = r.GetString("DriverOwner"), + DriverLease = r.GetDateTimeOffset("DriverLease"), + Sequential = r.GetInt32("Sequential") == 1, + CancelRequested = r.GetInt32("CancelRequested") == 1, + Total = r.GetInt32("Total") ?? 0, + Done = r.GetInt32("Done") ?? 0, + Failed = r.GetInt32("Failed") ?? 0, + Cancelled = r.GetInt32("Cancelled") ?? 0, + }; +} + +/// +/// Keeps each partition under Azure's ~2,000 entities/s target: a token bucket per partition at +/// , refilled continuously. Callers await their cost before touching the partition. +/// +public sealed class PartitionRateLimiter(int perSecond = 1_900) +{ + public int PerSecond { get; } = perSecond; + private readonly ConcurrentDictionary _buckets = new(StringComparer.Ordinal); + + public void Forget(string partition) => _buckets.TryRemove(partition, out _); + + public async Task TakeAsync(string partition, int cost, CancellationToken ct = default) + { + var bucket = _buckets.GetOrAdd(partition, _ => new Bucket(PerSecond)); + while (true) + { + var wait = bucket.TryTake(Math.Min(cost, PerSecond), PerSecond); + if (wait <= TimeSpan.Zero) return; + await Task.Delay(wait, ct); + } + } + + private sealed class Bucket(double tokens) + { + private double _tokens = tokens; + private long _stamp = Environment.TickCount64; + + public TimeSpan TryTake(int cost, int perSecond) + { + lock (this) + { + var now = Environment.TickCount64; + _tokens = Math.Min(perSecond, _tokens + (now - _stamp) * perSecond / 1000.0); + _stamp = now; + if (_tokens >= cost) { _tokens -= cost; return TimeSpan.Zero; } + return TimeSpan.FromMilliseconds(Math.Ceiling((cost - _tokens) * 1000.0 / perSecond)); + } + } + } +} diff --git a/docs/configuration.md b/docs/configuration.md index ee195f7..780ba70 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -370,27 +370,16 @@ By default it starts narrow and ramps slowly, to keep idle memory low; tune it f ### Orchestrator -Fan-out/fan-in task execution with crash recovery. +Fan-out/fan-in task execution. Every run's state lives in storage — one partition of `{Prefix}Work` per +run, where each task is one row and every state change is one partition transaction — so a restart has +nothing to recover: claims held by a stopped process lapse and are taken again. ```jsonc "Orchestrator": { - // Prefix for Azure Tables: {Prefix}Runs, {Prefix}Tasks, {Prefix}Results + // Prefix for Azure Tables: {Prefix}Work, {Prefix}Ready, {Prefix}Names, {Prefix}Finished, {Prefix}TaskResults. + // The previous design's {Prefix}Queue/QueueIndex/Tasks/Runs/Results tables are dropped at startup. "TablePrefix": "Orchestrator", - // Batch + coalesce per-task/run STATUS writes off the fan-out critical path, in ≤100-entity byte-budgeted - // Azure Table transactions. Default true. This is the throughput fix for large fan-outs — the per-task - // table write was the ceiling (see docs/orch-analysis.md). Results are NEVER batched (their chunking / - // multi-row large-payload path is untouched). Set false to fall back to per-task writes. - "BatchStatusWrites": true, - // Write the pre-invoke "Running" marker under a durable barrier (persisted BEFORE the task runs, batched - // with concurrently-starting tasks) so AttemptCount/MaxRetries still bounds poison tasks. Default true. - // False = eventual: the marker rides the periodic flush and the task doesn't wait — max throughput (100% - // pool utilization) at the cost of the strict poison-before-invoke guarantee. Terminal + run states stay - // durable in both modes (flushed before a run finalizes and on shutdown). - "DurableRunningBarrier": true, - // Status-writer flush interval / barrier latency ceiling (ms). Default 25. - "StatusFlushIntervalMs": 25, - // PS function that executes individual tasks. Receives TaskJson parameter. // Default: "Invoke-CraftTask" (provided in CraftRuntime/) "GenericTaskFunction": "Invoke-CraftTask", @@ -407,14 +396,13 @@ Fan-out/fan-in task execution with crash recovery. // Default: "Invoke-CraftPostExecution" (provided in CraftRuntime/) "PostExecFunction": "Invoke-CraftPostExecution", - // Max task interruptions (host crash/restart) before marking Failed. + // Max task interruptions (host crash/restart) before marking Failed; also PostExecution attempts. "MaxRetries": 3, - // Retention sweep over the three tables. A run that finished — or that nothing is driving and that - // last wrote to storage — longer ago than RetentionHours is removed together with its Tasks/Results - // partitions, as is any Tasks/Results partition whose Run row is already gone. Runs once at startup - // (after crash recovery) and then every CleanupIntervalHours; 0 keeps only the startup pass. Craft - // needs the rows only while a run is live — the retention is for operators reading recent history. + // Retention sweep. A run that finished longer ago than RetentionHours is removed with its Work and + // Results partitions. Runs once at startup and then every CleanupIntervalHours; 0 keeps only the + // startup pass. Craft needs the rows only while a run is live — the retention is for operators + // reading recent history. "RetentionHours": 48, "CleanupIntervalHours": 4 } diff --git a/perf-harness/docker-compose.bg.yml b/perf-harness/docker-compose.bg.yml index 4e429f6..cee201f 100644 --- a/perf-harness/docker-compose.bg.yml +++ b/perf-harness/docker-compose.bg.yml @@ -55,15 +55,10 @@ services: - BackgroundBurstToCeiling=${BG_BURST:-false} - BackgroundOverSubscribe=${BG_OVERSUB:-0} # Batched status writer (#3). run-orch.ps1 -NoBatch sets false to A/B the per-task-write "before". - - App__Orchestrator__BatchStatusWrites=${BATCH_WRITES:-true} - - App__Orchestrator__DurableRunningBarrier=${DURABLE_BARRIER:-true} - - App__Orchestrator__StatusFlushIntervalMs=${FLUSH_MS:-25} # Per-run status/re-drive tick cadence. run-manyruns.ps1 lowers it to compress the re-drive backoff. - App__Orchestrator__StatusTimerIntervalSeconds=${STATUS_INTERVAL:-60} # Re-drive backoff (② ). run-manyruns.ps1 flips it to A/B the backoff's effect at a fixed interval. - - App__Orchestrator__RedriveBackoff=${REDRIVE_BACKOFF:-true} # Pending-Parameters shedding (retained-memory fix). run-manyruns.ps1 flips it to A/B the memory effect. - - App__Orchestrator__ShedPendingParameters=${SHED_PARAMS:-true} # Default Warning keeps the other bg-harness runs quiet; run-manyruns.ps1 sets LOG_LEVEL=Information # to reproduce production's Info-level per-run status logging (the log flood is part of the cost under test). - CRAFT_LOG_LEVEL=${LOG_LEVEL:-Warning} diff --git a/tests/Craft.Tests/AzureTableStoreLargeEntityTests.cs b/tests/Craft.Tests/AzureTableStoreLargeEntityTests.cs index 1338d28..064221a 100644 --- a/tests/Craft.Tests/AzureTableStoreLargeEntityTests.cs +++ b/tests/Craft.Tests/AzureTableStoreLargeEntityTests.cs @@ -7,10 +7,8 @@ namespace Craft.Tests; /// -/// Tests here allocate multi-MB strings to force the storage size limits, and heap-delta measurement -/// tests (see ) cannot share a process with concurrent allocation. -/// Marking this collection non-parallel puts it in the same sequential phase as those, so the two never -/// run at once. See the note on for the failure mode this avoids. +/// Tests here allocate multi-MB strings to force the storage size limits, so they run alone rather than +/// alongside tests that are sensitive to heap pressure. /// [CollectionDefinition(LargeAllocationSerialTests.Name, DisableParallelization = true)] public class LargeAllocationSerialTests diff --git a/tests/Craft.Tests/JobDescriptorRehydrationTests.cs b/tests/Craft.Tests/JobDescriptorRehydrationTests.cs index 9f12c89..1550d7b 100644 --- a/tests/Craft.Tests/JobDescriptorRehydrationTests.cs +++ b/tests/Craft.Tests/JobDescriptorRehydrationTests.cs @@ -1,99 +1,18 @@ -using System.Runtime.CompilerServices; using Craft.Configuration; using Craft.Orchestration; using Craft.PowerShellHost; -using Craft.Storage; using Microsoft.Extensions.Configuration; using Microsoft.Extensions.Logging.Abstractions; namespace Craft.Tests; /// -/// Covers the descriptor queue's dispatch-time rehydration: what the resolver is handed, what it costs, -/// and what happens when the descriptor has gone stale underneath it. -/// -/// The design question these answer is "what does an extra storage read per dispatch cost at ~200 -/// dispatches/min?" — the answer being that the steady-state path performs ZERO reads, because the run -/// is already live in _activeRuns and object identity must be preserved anyway (see -/// OrchestratorService.ResolveTaskWorkAsync). Reads happen only on the recovery path. +/// Covers the descriptor queue's dispatch-time rehydration: what the resolver is handed, and what happens +/// when the descriptor has gone stale underneath it. /// public class JobDescriptorRehydrationTests { - /// An in-memory that counts point reads. - private sealed class CountingStore : ICraftTableStore - { - - // Claims are not exercised by this fake. Fail loudly rather than pretend the guard held — - // a silent 'true' here would look exactly like a successful claim. - public Task TryReplaceBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) => throw new NotSupportedException(); - private readonly Dictionary> _tables = new(); - public int PointReads; - public int PartitionScans; - - public Task PingAsync(CancellationToken ct = default) => Task.CompletedTask; - - public Task EnsureTableAsync(string table, CancellationToken ct = default) - { - if (!_tables.ContainsKey(table)) _tables[table] = new(); - return Task.CompletedTask; - } - - public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) - { - _tables[table][(row.PartitionKey, row.RowKey)] = row; - return Task.CompletedTask; - } - - public Task UpsertBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) - { - foreach (var r in rows) _tables[table][(r.PartitionKey, r.RowKey)] = r; - return Task.CompletedTask; - } - - public Task GetAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) - { - Interlocked.Increment(ref PointReads); - return Task.FromResult(_tables[table].TryGetValue((partitionKey, rowKey), out var r) ? r : null); - } - - public async IAsyncEnumerable QueryPartitionAsync(string table, string partitionKey, - [EnumeratorCancellation] CancellationToken ct = default) - { - Interlocked.Increment(ref PartitionScans); - foreach (var kv in _tables[table].Where(k => k.Key.Item1 == partitionKey).ToList()) - { - yield return kv.Value; - await Task.Yield(); - } - } - - public async IAsyncEnumerable QueryTableAsync(string table, - [EnumeratorCancellation] CancellationToken ct = default) - { - foreach (var kv in _tables[table].ToList()) - { - yield return kv.Value; - await Task.Yield(); - } - } - - public Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) - { - _tables[table].Remove((partitionKey, rowKey)); - return Task.CompletedTask; - } - - public Task DeletePartitionAsync(string table, string partitionKey, CancellationToken ct = default) - { - foreach (var k in _tables[table].Keys.Where(k => k.Item1 == partitionKey).ToList()) - _tables[table].Remove(k); - return Task.CompletedTask; - } - } - - private static (JobManager Jobs, CountingStore Store, OrchestratorTableStore Orch) NewHarness() + private static JobManager NewHarness() { var settings = new CraftSettings(); settings.Worker.BgPoolSize = 8; @@ -102,9 +21,7 @@ private static (JobManager Jobs, CountingStore Store, OrchestratorTableStore Orc var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); var jobs = new JobManager(NullLogger.Instance, settings, limiter); - var store = new CountingStore(); - var orch = new OrchestratorTableStore(NullLogger.Instance, settings, store); - return (jobs, store, orch); + return jobs; } private static Task Pump(JobManager jobs) => Task.Run(() => jobs.StartAsync(CancellationToken.None)); @@ -124,7 +41,7 @@ private static async Task WaitUntil(Func condition, int timeoutMs = [Fact] public async Task Dispatch_HandsTheDescriptorToTheResolver() { - var (jobs, _, _) = NewHarness(); + var jobs = NewHarness(); var seen = new List(); var done = 0; @@ -157,7 +74,7 @@ public async Task Dispatch_HandsTheDescriptorToTheResolver() [Fact] public async Task StaleDescriptor_IsSkipped_AndDispatchContinues() { - var (jobs, _, _) = NewHarness(); + var jobs = NewHarness(); var ran = 0; jobs.SetWorkResolver((d, _) => Task.FromResult?>( @@ -183,7 +100,7 @@ public async Task StaleDescriptor_IsSkipped_AndDispatchContinues() [Fact] public async Task DescriptorWithNoResolver_FailsTheJob_RatherThanDisappearing() { - var (jobs, _, _) = NewHarness(); + var jobs = NewHarness(); jobs.Enqueue(new JobDescriptor("run", "task", 0), "run-task"); _ = Pump(jobs); @@ -195,46 +112,6 @@ public async Task DescriptorWithNoResolver_FailsTheJob_RatherThanDisappearing() Assert.Contains("resolver", failed.LastError, StringComparison.OrdinalIgnoreCase); } - /// - /// The cost question. Rehydrating a task that is NOT already in memory costs exactly one partition - /// read of the run — the same read the crash-recovery path already performs. This pins the cost so a - /// future change that turns it into a per-task read shows up as a failure. - /// - [Fact] - public async Task Rehydration_FromStorage_CostsOneRunReadPerRun_NotPerTask() - { - var (_, counting, store) = NewHarness(); - await store.InitializeAsync(); - - var tasks = Enumerable.Range(0, 200).Select(i => new OrchestratorTaskItem - { - Id = $"Graph_tenant{i:D3}", - Status = "Pending", - Parameters = new Dictionary { ["TenantFilter"] = $"tenant{i:D3}.onmicrosoft.com" }, - }).ToList(); - - await store.UpsertRunAsync(new OrchestratorRun - { - Name = "CIPPDBCacheRun", - Status = "Running", - Priority = 5, - StartedUtc = DateTime.UtcNow, - Tasks = tasks, - TaskScriptName = "Invoke-CIPPDBCacheTask", - }); - await store.UpsertTaskBatchAsync("CIPPDBCacheRun", tasks); - - counting.PointReads = 0; - counting.PartitionScans = 0; - - var rehydrated = await store.GetRunAsync("CIPPDBCacheRun"); - - Assert.NotNull(rehydrated); - Assert.Equal(200, rehydrated!.Tasks.Count); - Assert.Equal(1, counting.PointReads); // the run row - Assert.Equal(1, counting.PartitionScans); // all 200 task rows in one partition query - } - // ── OWNERSHIP, as used by the Pending re-drive ─────────────────────────────────────────────────── /// @@ -245,7 +122,7 @@ await store.UpsertRunAsync(new OrchestratorRun [Fact] public void IsQueuedOrRunning_True_ForAQueuedJob() { - var (jobs, _, _) = NewHarness(); + var jobs = NewHarness(); jobs.Enqueue(new JobDescriptor("run-a", "task-1", 4), name: "run-a-task-1"); @@ -261,7 +138,7 @@ public void IsQueuedOrRunning_True_ForAQueuedJob() [Fact] public async Task IsQueuedOrRunning_False_OnceTheJobHasCompleted() { - var (jobs, _, _) = NewHarness(); + var jobs = NewHarness(); _ = Pump(jobs); var ran = new TaskCompletionSource(); @@ -277,7 +154,7 @@ public async Task IsQueuedOrRunning_False_OnceTheJobHasCompleted() [Fact] public void IsQueuedOrRunning_False_ForAnUnknownJob() { - var (jobs, _, _) = NewHarness(); + var jobs = NewHarness(); Assert.False(jobs.IsQueuedOrRunning("run-c-task-never-enqueued")); } diff --git a/tests/Craft.Tests/JobDurabilityTests.cs b/tests/Craft.Tests/JobDurabilityTests.cs index 4fbfd03..5644d7f 100644 --- a/tests/Craft.Tests/JobDurabilityTests.cs +++ b/tests/Craft.Tests/JobDurabilityTests.cs @@ -83,12 +83,6 @@ public Task DeletePartitionAsync(string table, string partitionKey, Cancellation } } - private static OrchestratorTableStore NewStore(out FakeStore backing) - { - backing = new FakeStore(); - return new OrchestratorTableStore(NullLogger.Instance, new CraftSettings(), backing); - } - private static JobManager NewJobManager() { var settings = new CraftSettings(); @@ -117,62 +111,6 @@ public void Cancelled(JobDescriptor descriptor) } } - // ── Per-task priority ─────────────────────────────────────────────────────────────────────────── - - [Fact] - public async Task TaskPriority_RoundTripsThroughStorage() - { - var store = NewStore(out _); - await store.InitializeAsync(); - - var tasks = new List - { - new() { Id = "inherits", Status = "Pending" }, // null ⇒ run priority - new() { Id = "overridden", Status = "Pending", Priority = 0 }, // escalated by an operator - }; - await store.UpsertRunAsync(new OrchestratorRun - { - Name = "run", - Status = "Running", - Priority = 5, - StartedUtc = DateTime.UtcNow, - Tasks = tasks, - }); - await store.UpsertTaskBatchAsync("run", tasks); - - var recovered = await store.GetRunAsync("run"); - - Assert.Equal(5, recovered!.Priority); - Assert.Null(recovered.Tasks.Single(t => t.Id == "inherits").Priority); - Assert.Equal(0, recovered.Tasks.Single(t => t.Id == "overridden").Priority); - } - - /// - /// The Replace-mode trap: the batched status writer rewrites the WHOLE row, so a column it does not - /// carry is erased. If Priority ever drops out of TaskStatusWrite, the override survives exactly - /// until the task moves to Running — this fails the moment that regresses. - /// - [Fact] - public async Task StatusWrite_PreservesTaskPriority_RatherThanErasingIt() - { - var store = NewStore(out _); - await store.InitializeAsync(); - - var task = new OrchestratorTaskItem { Id = "t1", Status = "Pending", Priority = 0 }; - await store.UpsertTaskBatchAsync("run", [task]); - await store.UpsertRunAsync(new OrchestratorRun { Name = "run", Status = "Running", Priority = 5 }); - - // A status transition through the coalescing writer's path. - await store.WriteTaskStatusBatchAsync( - [new TaskStatusWrite("run", "t1", "Running", "{}", 0, null, null, task.Priority)]); - - var recovered = await store.GetRunAsync("run"); - var reloaded = recovered!.Tasks.Single(); - - Assert.Equal("Running", reloaded.Status); - Assert.Equal(0, reloaded.Priority); - } - // ── Durable operator actions ──────────────────────────────────────────────────────────────────── [Fact] diff --git a/tests/Craft.Tests/JobManagerReEnqueueTests.cs b/tests/Craft.Tests/JobManagerReEnqueueTests.cs index 2c34ea1..4dadbb0 100644 --- a/tests/Craft.Tests/JobManagerReEnqueueTests.cs +++ b/tests/Craft.Tests/JobManagerReEnqueueTests.cs @@ -16,8 +16,9 @@ namespace Craft.Tests; /// fresh copy of that job is queued or running. IsQueuedOrRunning reads the frozen record and /// answers "no" for a job that is very much running. /// -/// That answer is load-bearing. JobQueuePump.ReleaseFinishedAsync treats it as "this job is done" and -/// DELETES the task's durable queue row. Observed live: the pump released 7-9 "finished" jobs every +/// That answer is load-bearing. WorkPump.Forget treats it as "this job is done" and stops renewing the +/// task's claim, so a long task loses its lease and runs again elsewhere. Observed live (when the pump +/// deleted the queue row instead): the pump released 7-9 "finished" jobs every /// second while only 8 could physically be running, on a run where individual tasks executed up to five /// times. /// @@ -74,10 +75,10 @@ public async Task ATaskEnqueuedAgainAfterItRan_IsReportedAsRunning_NotAsItsPrevi await WaitUntilAsync(() => Volatile.Read(ref ran) == 2, "second job never started"); - // The job IS running. Answering "no" here is what makes the pump delete a live job's queue row. + // The job IS running. Answering "no" here is what makes the pump stop renewing a live job's claim. Assert.True(jobs.IsQueuedOrRunning(id), "the manager reports a running job as finished — its record is frozen at the previous run's " + - "status, so JobQueuePump.ReleaseFinishedAsync will drop the durable row out from under it"); + "status, so WorkPump.Forget will stop renewing its claim"); release.Release(); await jobs.StopAsync(CancellationToken.None); diff --git a/tests/Craft.Tests/JobQueueAzuriteTests.cs b/tests/Craft.Tests/JobQueueAzuriteTests.cs deleted file mode 100644 index 9b34da8..0000000 --- a/tests/Craft.Tests/JobQueueAzuriteTests.cs +++ /dev/null @@ -1,305 +0,0 @@ -using Craft.Configuration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// Verifies the two properties that are the BACKEND's to provide, not ours, against a real Azure Tables -/// implementation (Azurite). The unit tests use a fake, and a fake is only ever as honest as whoever -/// wrote it — during development that fake got both of these wrong and the tests still passed. -/// -/// 1. Ordering is server-side and priority-first. Rows come back sorted by partition key then row key, -/// so "P00" precedes "P04" and, within a bucket, older precedes newer. Nothing sorts client-side. -/// 2. Exclusivity is real optimistic concurrency. Two claimers racing the same rows cannot both win, -/// because the second one's If-Match no longer matches. -/// -/// On leases: there is no Azure lease API here. Azure Tables does not have one — that is Blob. "Lease" -/// is only the name of an ordinary DateTimeOffset column, and the exclusivity comes entirely from ETag -/// conditional writes. -/// -/// Runs against the local emulator by default, or a real storage account when -/// CRAFT_TEST_TABLE_CONNECTION is set. Skipped, not failed, when neither is reachable, so this is safe -/// in CI — which does mean a skip looks like a pass; see TryConnectAsync for how to re-prove it. -/// -public class JobQueueAzuriteTests -{ - private static async Task TryConnectAsync() - { - var settings = new CraftSettings(); - - // Point at a real storage account by setting CRAFT_TEST_TABLE_CONNECTION; otherwise this runs - // against the local emulator. Same assertions either way — the properties under test are the - // backend's, and Azurite is a reimplementation of them, not the thing that ships. - var connection = Environment.GetEnvironmentVariable("CRAFT_TEST_TABLE_CONNECTION"); - if (!string.IsNullOrWhiteSpace(connection)) - settings.Auth.UserStorageConnection = connection; - else - settings.Storage.AllowDevelopmentStorage = true; - - // Unique per run so repeated runs cannot see each other's rows, and so a real account is never - // left holding fixtures under a name a later run would reuse. - settings.Orchestrator.TablePrefix = "azqt" + Guid.NewGuid().ToString("N")[..8]; - - var store = new AzureTableStore(settings); - try - { - using var cts = new CancellationTokenSource(TimeSpan.FromSeconds(3)); - await store.PingAsync(cts.Token); - } - catch - { - // Nothing to verify against. NOTE: an early return is indistinguishable from a pass — to - // re-prove these actually ran, swap the call sites' null check for Assert.NotNull(queue) - // and confirm they still pass. - return null; - } - - var queue = new JobQueueStore(NullLogger.Instance, settings, store); - await queue.InitializeAsync(); - return queue; - } - - private static DateTime At(int minute) => new(2026, 8, 9, 2, minute, 0, DateTimeKind.Utc); - - /// - /// The claim query narrows server-side with an OData $filter so a backlog is not paged to the client - /// on every pump tick. Every in-memory fake ignores that filter and returns everything, so this is - /// the only place the filter's semantics are actually exercised — and a filter that is subtly wrong - /// does not fail loudly, it silently hides claimable work and the queue stops draining. - /// - /// The three states that must survive it: never claimed, lease expired, lease live. - /// - [Fact] - public async Task ServerSideFilter_ReturnsFreeAndExpiredRows_AndHidesLiveOnes() - { - var queue = await TryConnectAsync(); - if (queue == null) return; - - await queue.EnqueueBatchAsync("run", [("free", 4), ("expired", 4), ("live", 4)], At(0)); - - // Give "live" a long lease and "expired" one that lapses almost immediately. - var first = await queue.ClaimBatchAsync("holder", 1, TimeSpan.FromMinutes(30)); - var second = await queue.ClaimBatchAsync("holder", 1, TimeSpan.FromMilliseconds(1)); - Assert.Single(first); - Assert.Single(second); - await Task.Delay(50); - - // Whatever is left free, plus the lapsed one — never the live one. - var claimable = await queue.ClaimBatchAsync("worker", 10, TimeSpan.FromMinutes(30)); - var ids = claimable.Select(c => c.TaskId).ToHashSet(StringComparer.Ordinal); - - Assert.Equal(2, ids.Count); - Assert.Contains(second[0].TaskId, ids); // lease lapsed -> reclaimable - Assert.DoesNotContain(first[0].TaskId, ids); // lease live -> hidden - } - - /// - /// The run-scoped reads (remove, release, queued-ids) narrow on RunName server-side. Every - /// in-memory fake ignores the filter and returns everything, so this is the only place the - /// predicate is actually evaluated — and getting it wrong is silent: a filter that matches nothing - /// makes cleanup delete nothing and de-duplication see nothing, both of which look like success. - /// - [Fact] - public async Task ServerSideRunFilter_ScopesToOneRun() - { - var queue = await TryConnectAsync(); - if (queue == null) return; - - await queue.EnqueueBatchAsync("run-a", [("a1", 4), ("a2", 4)], At(0)); - await queue.EnqueueBatchAsync("run-b", [("b1", 4)], At(0)); - - Assert.Equal(["a1", "a2"], - (await queue.GetQueuedTaskIdsAsync("run-a")).OrderBy(x => x, StringComparer.Ordinal)); - - // Removing one run must not touch the other. - await queue.RemoveRunAsync("run-a"); - - Assert.Empty(await queue.GetQueuedTaskIdsAsync("run-a")); - Assert.Equal(["b1"], await queue.GetQueuedTaskIdsAsync("run-b")); - } - - /// - /// A large run's queue rows are read as one RowKey range, built from the run's key prefix. The fakes - /// filter ranges in memory, so only here is the service's own range filter — and the page-sized claim - /// beside it — evaluated. A wrong bound is silent: the run's work reads as gone, or a neighbour's rows - /// leak in. - /// - [Fact] - public async Task RunKeyRange_ScopesToOneRun_AndAPageSizedClaimStillTakesTheHead() - { - var queue = await TryConnectAsync(); - if (queue == null) return; - - var ids = Enumerable.Range(0, 40).Select(i => $"t{i:D2}").ToList(); - await queue.EnqueueBatchAsync("Big", ids.Select(i => (i, 4)).ToList(), At(1)); - await queue.EnqueueBatchAsync("Bigger", [("x", 4)], At(2)); - await queue.EnqueueBatchAsync("Early", [("e", 4)], At(0)); - - Assert.Equal(ids, (await queue.GetDispatchableTaskIdsAsync("Big", ids)).Order()); - - var claimed = await queue.ClaimBatchAsync("dead", 3, TimeSpan.FromMinutes(30)); - Assert.Equal(["e", "t00", "t01"], claimed.Select(c => c.TaskId)); - - var head = await queue.ListQueuedAsync(); - Assert.Equal(42, head.Count); - Assert.All(head, r => Assert.StartsWith("P04", r.Bucket)); - Assert.Equal(3, head.Count(r => r.Claimed)); - - Assert.Equal(2, await queue.ReleaseRunClaimsAsync("Big")); - await queue.RemoveRunAsync("Big", At(1)); - Assert.Empty(await queue.GetQueuedTaskIdsAsync("Big")); - Assert.Equal(["x"], await queue.GetQueuedTaskIdsAsync("Bigger")); - } - - /// - /// The v3 migration keys each run by the StartedUtc on its Run row, read with a projection. A projection - /// returns only the columns it names, so a missing key column here reads every run as unknown and keys - /// it off its queue rows instead — silently, since the fakes return whole rows. - /// - [Fact] - public async Task Migration_KeysARunByItsRunRowsStartTime() - { - var settings = new CraftSettings { Storage = { AllowDevelopmentStorage = true } }; - var connection = Environment.GetEnvironmentVariable("CRAFT_TEST_TABLE_CONNECTION"); - if (!string.IsNullOrWhiteSpace(connection)) settings.Auth.UserStorageConnection = connection; - settings.Orchestrator.TablePrefix = "azqm" + Guid.NewGuid().ToString("N")[..8]; - var store = new AzureTableStore(settings); - try - { - using var cts = new CancellationTokenSource(TimeSpan.FromSeconds(3)); - await store.PingAsync(cts.Token); - } - catch { return; } - - var prefix = settings.Orchestrator.TablePrefix; - var started = At(0).AddDays(-3); - await store.EnsureTableAsync($"{prefix}Runs"); - await store.EnsureTableAsync($"{prefix}Queue"); - await store.EnsureTableAsync($"{prefix}QueueIndex"); - await store.UpsertAsync($"{prefix}Runs", new StoreRow("Run", "MailboxRules_t1") - { - Properties = { ["StartedUtc"] = new DateTimeOffset(started) } - }); - await store.UpsertAsync($"{prefix}Queue", new StoreRow("P04", "MailboxRules_t1|b1") - { - Properties = - { - ["RunName"] = "MailboxRules_t1", ["TaskId"] = "b1", ["Priority"] = 4, ["Owner"] = "", - ["QueuedUtc"] = new DateTimeOffset(At(0)), - } - }); - await store.UpsertAsync($"{prefix}QueueIndex", new StoreRow("$schema", "queue-index") { Properties = { ["Version"] = 2 } }); - - var queue = new JobQueueStore(NullLogger.Instance, settings, store); - await queue.InitializeAsync(); - - var row = Assert.Single(await queue.ListQueuedAsync()); - Assert.Equal(JobQueueStore.BuildRowKey(started, "MailboxRules_t1", "b1"), row.RowKey); - } - - /// A run name containing a quote must not break the filter or leak into it. - [Fact] - public async Task ServerSideRunFilter_HandlesAQuoteInTheRunName() - { - var queue = await TryConnectAsync(); - if (queue == null) return; - - const string Odd = "run-o'brien"; - await queue.EnqueueBatchAsync(Odd, [("t1", 4)], At(0)); - await queue.EnqueueBatchAsync("run-plain", [("t2", 4)], At(0)); - - Assert.Equal(["t1"], await queue.GetQueuedTaskIdsAsync(Odd)); - Assert.Equal(["t2"], await queue.GetQueuedTaskIdsAsync("run-plain")); - } - - [Fact] - public async Task BackendReturnsWorkHighestPriorityFirst() - { - var queue = await TryConnectAsync(); - if (queue == null) return; - - // Inserted in the least helpful order: bulk P4 work first, urgent P0 last, idle P6 early. If any - // of this came back in insertion order the priority scheme is broken. - await queue.EnqueueBatchAsync("StandardsApply", - [("std-a", 4), ("std-b", 4)], At(30)); - await queue.EnqueueAsync("StandardsApply", "std-c", 4, At(1)); - await queue.EnqueueAsync("DbCache", "cache-0", 6, At(0)); - await queue.EnqueueAsync("AuditLogIngest", "audit-0", 0, At(59)); - - // Claim one at a time so the order the backend hands work out is directly observable. - var order = new List(); - for (var i = 0; i < 5; i++) - { - var claimed = await queue.ClaimBatchAsync($"w{i}", 1, TimeSpan.FromMinutes(20)); - order.Add(Assert.Single(claimed).TaskId); - } - - // P0 first despite being queued last; P6 last despite being queued early. - Assert.Equal("audit-0", order[0]); - Assert.Equal("cache-0", order[4]); - - // The three P4 tasks fill the middle, ahead of P6 and behind P0. Schema v2 keys per (run, task), - // so ORDER within a priority is no longer oldest-first — only that they all rank between the two. - var middle = order.GetRange(1, 3); - Assert.Contains("std-a", middle); - Assert.Contains("std-b", middle); - Assert.Contains("std-c", middle); - } - - [Fact] - public async Task TwoClaimersRacingTheSameRowsCannotBothWin() - { - var queue = await TryConnectAsync(); - if (queue == null) return; - - await queue.EnqueueBatchAsync("run", - Enumerable.Range(0, 6).Select(i => ($"task-{i:D2}", 4)).ToList(), At(0)); - - // Both read the same rows before either writes, so both hold the same ETags. - var a = queue.ClaimBatchAsync("worker-a", 6, TimeSpan.FromMinutes(20)); - var b = queue.ClaimBatchAsync("worker-b", 6, TimeSpan.FromMinutes(20)); - var results = await Task.WhenAll(a, b); - - var winners = results.Where(r => r.Count > 0).ToList(); - Assert.Single(winners); - Assert.Equal(6, winners[0].Count); - - // And the rows really are owned now, not merely reported as claimed. - Assert.Empty(await queue.ClaimBatchAsync("worker-c", 6, TimeSpan.FromMinutes(20))); - } - - [Fact] - public async Task ExpiredLeaseIsReclaimableAndALiveOneIsNot() - { - var queue = await TryConnectAsync(); - if (queue == null) return; - - await queue.EnqueueAsync("run", "task-0", 4, At(0)); - - // Long lease for the "still held" assertions. They only need the lease to outlive a storage - // round-trip, and a short one makes that a race: this test used to claim for 2 seconds and - // then assert liveness through two more round-trips, so a loaded backend expired the lease - // before the assertion ran and the test failed claiming exclusivity was broken. Nothing here - // is waiting out this lease, so minutes cost nothing. - var held = await queue.ClaimBatchAsync("worker-a", 1, TimeSpan.FromMinutes(5)); - Assert.Single(held); - - // Live lease: nobody else may take it. - Assert.Empty(await queue.ClaimBatchAsync("worker-b", 1, TimeSpan.FromMinutes(20))); - - // Renewal keeps ownership rather than releasing it — the long-running-task path. - Assert.True(await queue.RenewAsync(held, "worker-a", TimeSpan.FromMinutes(5))); - Assert.Empty(await queue.ClaimBatchAsync("worker-b", 1, TimeSpan.FromMinutes(20))); - - // Renewal sets the deadline against the real clock, so renewing SHORT lets it lapse. This is - // the one deliberately time-bound step; the wait is several times the lease so a slow - // backend makes it later, not flakier. - Assert.True(await queue.RenewAsync(held, "worker-a", TimeSpan.FromSeconds(1))); - await Task.Delay(TimeSpan.FromSeconds(4)); - - // Lapsed: reclaimable, with no intervention from the holder. This is the crash-recovery path. - var reclaimed = await queue.ClaimBatchAsync("worker-b", 1, TimeSpan.FromMinutes(20)); - Assert.Equal("task-0", Assert.Single(reclaimed).TaskId); - } -} diff --git a/tests/Craft.Tests/JobQueueDispatchableTests.cs b/tests/Craft.Tests/JobQueueDispatchableTests.cs deleted file mode 100644 index 6a0b9fe..0000000 --- a/tests/Craft.Tests/JobQueueDispatchableTests.cs +++ /dev/null @@ -1,106 +0,0 @@ -using Craft.Configuration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// The index table answers "does this run still have queued work" from a single partition, which is -/// what keeps the re-drive and finalize paths off a full-table scan. But the index can OUTLIVE the -/// queue rows it points at, and when it does it lies: it reports a task queued that no pump will ever -/// claim, so an orphan re-drive that trusts the index skips it and the run stalls indefinitely with -/// the task Pending — resume logs "Dispatched 0 tasks (N already queued)" and the orphan watchdog -/// never fires, because GetQueuedTaskIdsAsync (index-only) keeps reporting the task queued while the -/// queue table has no runnable row for it. Clearing the stale index row is the only thing that lets -/// the watchdog see it as orphaned again. -/// -/// is the fix: it verifies each candidate -/// against the queue TABLE, returning only tasks the pump can actually still claim. These tests pin -/// the three divergence shapes it has to catch. -/// -public class JobQueueDispatchableTests -{ - private static readonly TimeSpan Lease = TimeSpan.FromMinutes(20); - private const string QueueTable = "OrchestratorQueue"; - - private static (JobQueueStore Queue, RunRemainingCounterTests.ConditionalStore Backing) NewQueue() - { - var backing = new RunRemainingCounterTests.ConditionalStore(); - var queue = new JobQueueStore(NullLogger.Instance, new CraftSettings(), backing); - return (queue, backing); - } - - [Fact] - public async Task NormallyQueuedTasksAreAllDispatchable() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueBatchAsync("R", [("a", 4), ("b", 4), ("c", 4)], DateTime.UtcNow); - - var dispatchable = await queue.GetDispatchableTaskIdsAsync("R", ["a", "b", "c"]); - - Assert.Equal(3, dispatchable.Count); - Assert.Contains("a", dispatchable); - Assert.Contains("b", dispatchable); - Assert.Contains("c", dispatchable); - } - - [Fact] - public async Task AnIndexRowWithNoQueueRowIsNotDispatchable_ButTheIndexStillListsIt() - { - var (queue, backing) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueBatchAsync("R", [("a", 4), ("b", 4)], DateTime.UnixEpoch); - - // Delete ONLY b's queue row, leaving its index row — the exact divergence a crash between the - // two deletes, or a run carried in from a pre-pump build, leaves behind. - await backing.DeleteAsync(QueueTable, JobQueueStore.Bucket(4), JobQueueStore.BuildRowKey(DateTime.UnixEpoch, "R", "b")); - - // The index — what the old re-drive trusted — still reports both as queued. - var indexView = await queue.GetQueuedTaskIdsAsync("R"); - Assert.Contains("a", indexView); - Assert.Contains("b", indexView); - - // The queue-verified view sees b for the ghost it is. - var dispatchable = await queue.GetDispatchableTaskIdsAsync("R", ["a", "b"]); - Assert.Contains("a", dispatchable); - Assert.DoesNotContain("b", dispatchable); - } - - [Fact] - public async Task AQueueRowOwnedWithNoLeaseIsNotDispatchable() - { - var (queue, backing) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueBatchAsync("R", [("a", 4)], DateTime.UnixEpoch); - - // Owned with no LeaseUntil: neither "Owner eq ''" nor "LeaseUntil lt now", so the claim filter - // can never match it and the pump will never dispatch it — a ghost as surely as a missing row. - var key = JobQueueStore.BuildRowKey(DateTime.UnixEpoch, "R", "a"); - var row = await backing.GetAsync(QueueTable, JobQueueStore.Bucket(4), key); - Assert.NotNull(row); - row!["Owner"] = "dead-instance"; - row["LeaseUntil"] = (DateTimeOffset?)null; - await backing.UpsertAsync(QueueTable, row); - - var dispatchable = await queue.GetDispatchableTaskIdsAsync("R", ["a"]); - Assert.Empty(dispatchable); - } - - [Fact] - public async Task AQueueRowUnderALeaseStaysDispatchable() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueBatchAsync("R", [("a", 4)], DateTime.UtcNow); - - // A live claim: owned under a lease. The pump (or its lapse) will run it, so the re-drive must - // NOT treat it as orphaned — doing so would reset a row a worker is actively holding and could - // run the task a second time. - var claimed = await queue.ClaimBatchAsync("worker-a", 8, Lease); - Assert.Single(claimed); - - var dispatchable = await queue.GetDispatchableTaskIdsAsync("R", ["a"]); - Assert.Contains("a", dispatchable); - } -} diff --git a/tests/Craft.Tests/JobQueueFifoTests.cs b/tests/Craft.Tests/JobQueueFifoTests.cs deleted file mode 100644 index c6ca829..0000000 --- a/tests/Craft.Tests/JobQueueFifoTests.cs +++ /dev/null @@ -1,244 +0,0 @@ -using Craft.Configuration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// Schema v3: within a priority bucket the queue drains oldest run first. v2 keyed rows {run}|{task}, -/// so a bucket drained alphabetically by run name and a run whose name sorted late never ran while -/// earlier-sorting runs kept arriving (MailboxRules/UpdatePermissions starved for 16 days behind a steady -/// stream of AuditLog/DomainAnalyser runs). These pin the order, that the key stays one row per task, and -/// the migration of a live v2 backlog. -/// -public class JobQueueFifoTests -{ - private static readonly TimeSpan Lease = TimeSpan.FromMinutes(20); - private static DateTime At(int minute) => new(2026, 10, 5, 2, minute, 0, DateTimeKind.Utc); - - private sealed record Q(JobQueueStore Queue, RunRemainingCounterTests.ConditionalStore Store, string QueueTable, - string IndexTable, string RunsTable); - - private static Q NewQueue() - { - var settings = new CraftSettings(); - var store = new RunRemainingCounterTests.ConditionalStore(); - var queue = new JobQueueStore(NullLogger.Instance, settings, store); - var prefix = settings.Orchestrator.TablePrefix; - return new Q(queue, store, $"{prefix}Queue", $"{prefix}QueueIndex", $"{prefix}Runs"); - } - - private static async Task> DrainOrderAsync(JobQueueStore queue) - { - var order = new List(); - while (true) - { - var claimed = await queue.ClaimBatchAsync("w", 1, Lease); - if (claimed.Count == 0) return order; - order.Add($"{claimed[0].RunName}/{claimed[0].TaskId}"); - await queue.RemoveAsync(claimed[0]); - } - } - - private static async Task> RowsAsync(Q q, string table) - { - var rows = new List(); - await foreach (var r in q.Store.QueryTableAsync(table)) rows.Add(r); - return rows; - } - - [Fact] - public async Task ABucketDrainsOldestRunFirst_NotAlphabetically() - { - var q = NewQueue(); - await q.Queue.InitializeAsync(); - - await q.Queue.EnqueueBatchAsync("UpdatePermissions-1", [("u0", 4), ("u1", 4)], At(0)); - await q.Queue.EnqueueBatchAsync("AuditLog-2", [("a0", 4)], At(5)); - await q.Queue.EnqueueBatchAsync("DomainAnalyser-3", [("d0", 4)], At(9)); - - Assert.Equal(["UpdatePermissions-1/u0", "UpdatePermissions-1/u1", "AuditLog-2/a0", "DomainAnalyser-3/d0"], - await DrainOrderAsync(q.Queue)); - } - - [Fact] - public async Task PriorityStillBeatsAge() - { - var q = NewQueue(); - await q.Queue.InitializeAsync(); - - await q.Queue.EnqueueBatchAsync("Old", [("o0", 4)], At(0)); - await q.Queue.EnqueueBatchAsync("Urgent", [("x0", 1)], At(30)); - - Assert.Equal(["Urgent/x0", "Old/o0"], await DrainOrderAsync(q.Queue)); - } - - [Fact] - public async Task AReEnqueueHoursLater_KeepsTheRunsPlaceAndOneRow() - { - var q = NewQueue(); - await q.Queue.InitializeAsync(); - - await q.Queue.EnqueueBatchAsync("Zeta", [("z0", 4), ("z1", 4)], At(0)); - await q.Queue.EnqueueBatchAsync("Alpha", [("a0", 4)], At(10)); - // The re-drive re-queues with the run's start time, however long after the run started. - await q.Queue.EnqueueAsync("Zeta", "z1", 4, At(0)); - - Assert.Equal(3, (await RowsAsync(q, q.QueueTable)).Count); - Assert.Equal(["Zeta/z0", "Zeta/z1", "Alpha/a0"], await DrainOrderAsync(q.Queue)); - } - - [Fact] - public async Task RemovingOneOutingOfARecurringRun_LeavesTheNextOutingsRows() - { - var q = NewQueue(); - await q.Queue.InitializeAsync(); - await q.Queue.EnqueueBatchAsync("CIPPDBCacheOrchestrator", [("old", 4)], At(0)); - await q.Queue.EnqueueBatchAsync("CIPPDBCacheOrchestrator", [("new", 4)], At(30)); - - // The previous outing's finalize removes its rows after the next outing has enqueued. - await q.Queue.RemoveRunAsync("CIPPDBCacheOrchestrator", At(0)); - - Assert.Equal(["new"], await q.Queue.GetQueuedTaskIdsAsync("CIPPDBCacheOrchestrator")); - Assert.Equal("new", Assert.Single(await RowsAsync(q, q.QueueTable)).GetString("TaskId")); - - await q.Queue.RemoveRunAsync("CIPPDBCacheOrchestrator"); - Assert.Empty(await RowsAsync(q, q.QueueTable)); - Assert.DoesNotContain(await RowsAsync(q, q.IndexTable), r => r.PartitionKey == "CIPPDBCacheOrchestrator"); - } - - [Fact] - public async Task ALargeRunsDispatchCheck_ReadsItsKeyRange_NotOneRowAtATime() - { - var q = NewQueue(); - await q.Queue.InitializeAsync(); - var ids = Enumerable.Range(0, 40).Select(i => $"t{i:D2}").ToList(); - await q.Queue.EnqueueBatchAsync("Big", ids.Select(i => (i, 4)).ToList(), At(0)); - await q.Queue.EnqueueBatchAsync("Other", [("o", 4)], At(1)); - - var rows = await RowsAsync(q, q.QueueTable); - // A ghost (index entry, no queue row) and an owned row with no lease, which the pump never claims. - await q.Store.DeleteAsync(q.QueueTable, "P04", rows.Single(r => r.GetString("TaskId") == "t00").RowKey); - var stuck = rows.Single(r => r.GetString("TaskId") == "t01"); - stuck["Owner"] = "gone"; - await q.Store.UpsertAsync(q.QueueTable, stuck); - - q.Store.Gets.Clear(); - var dispatchable = await q.Queue.GetDispatchableTaskIdsAsync("Big", ids); - - Assert.Equal(ids.Skip(2), dispatchable.Order()); - Assert.False(q.Store.Gets.ContainsKey(q.QueueTable)); - } - - [Fact] - public async Task ReleasingALargeRunsClaims_FreesEveryRowOfThatRunOnly() - { - var q = NewQueue(); - await q.Queue.InitializeAsync(); - await q.Queue.EnqueueBatchAsync("Big", Enumerable.Range(0, 40).Select(i => ($"t{i:D2}", 4)).ToList(), At(0)); - await q.Queue.EnqueueBatchAsync("Other", [("o", 4)], At(1)); - Assert.Equal(41, (await q.Queue.ClaimBatchAsync("dead", 100, Lease)).Count); - - q.Store.Gets.Clear(); - Assert.Equal(40, await q.Queue.ReleaseRunClaimsAsync("Big")); - - Assert.False(q.Store.Gets.ContainsKey(q.QueueTable)); - Assert.Equal(40, (await q.Queue.ClaimBatchAsync("next", 100, Lease)).Count); - } - - [Fact] - public async Task AnEpochSurvivesTheSecondsRoundTrip() - { - var q = NewQueue(); - await q.Queue.InitializeAsync(); - var odd = At(0).AddTicks(1_234_567); - - await q.Queue.EnqueueBatchAsync("R", [("a", 4)], odd); - await q.Queue.EnqueueAsync("R", "a", 4, odd); - - var row = Assert.Single(await RowsAsync(q, q.QueueTable)); - Assert.Equal(JobQueueStore.BuildRowKey(At(0), "R", "a"), row.RowKey); - } - - // ── migration ──────────────────────────────────────────────────────────────────────────────── - - /// A queue row and its index entry exactly as v2 wrote them, plus the v2 schema marker. - private static async Task SeedV2Async(Q q, string run, string task, DateTime queuedUtc, int priority = 4, - string owner = "", DateTimeOffset? lease = null) - { - var bucket = JobQueueStore.Bucket(priority); - var key = $"{run}|{task}"; - await q.Store.UpsertAsync(q.QueueTable, new StoreRow(bucket, key) - { - Properties = - { - ["RunName"] = run, ["TaskId"] = task, ["Priority"] = priority, ["Owner"] = owner, - ["LeaseUntil"] = lease, ["QueuedUtc"] = new DateTimeOffset(queuedUtc, TimeSpan.Zero), - } - }); - await q.Store.UpsertAsync(q.IndexTable, new StoreRow(run, $"{bucket}|{key}") - { - Properties = { ["TaskId"] = task, ["RunName"] = run } - }); - await q.Store.UpsertAsync(q.IndexTable, new StoreRow("$schema", "queue-index") { Properties = { ["Version"] = 2 } }); - } - - [Fact] - public async Task MigratesAV2Backlog_ToOldestRunFirst_KeepingClaimsAndTheIndex() - { - var q = NewQueue(); - // Alphabetically the v2 order was AuditLog, MailboxRules; by age MailboxRules is 16 days older. Its - // Run row holds the start time later enqueues key by; AuditLog has none and falls back to its rows. - var mailboxStarted = At(0).AddDays(-17); - await q.Store.UpsertAsync(q.RunsTable, new StoreRow("Run", "MailboxRules_t1") - { - Properties = { ["StartedUtc"] = new DateTimeOffset(mailboxStarted) } - }); - await SeedV2Async(q, "MailboxRules_t1", "b1", At(0).AddDays(-16)); - await SeedV2Async(q, "MailboxRules_t1", "b2", At(0).AddDays(-16)); - await SeedV2Async(q, "AuditLog_t1", "s1", At(0)); - var leaseUntil = DateTimeOffset.UtcNow.AddMinutes(10); - await SeedV2Async(q, "AuditLog_t1", "s2", At(0), owner: "w-other", lease: leaseUntil); - - await q.Queue.InitializeAsync(); - - var queueRows = await RowsAsync(q, q.QueueTable); - Assert.Equal(4, queueRows.Count); - Assert.All(queueRows, r => Assert.NotNull(JobQueueStore.ParseEpoch(r.RowKey))); - var claimedRow = Assert.Single(queueRows, r => r.GetString("TaskId") == "s2"); - Assert.Equal("w-other", claimedRow.GetString("Owner")); - Assert.Equal(leaseUntil, claimedRow.GetDateTimeOffset("LeaseUntil")); - - // Index entries point at the new keys: every run-scoped read still sees its tasks as dispatchable. - Assert.Equal(["b1", "b2"], (await q.Queue.GetDispatchableTaskIdsAsync("MailboxRules_t1", ["b1", "b2"])).Order()); - Assert.Equal(["s1", "s2"], (await q.Queue.GetDispatchableTaskIdsAsync("AuditLog_t1", ["s1", "s2"])).Order()); - Assert.Equal(4, (await RowsAsync(q, q.IndexTable)).Count(r => r.PartitionKey != "$schema")); - - // A later enqueue of a migrated task, keyed by the run's start, lands on its existing row. - await q.Queue.EnqueueAsync("MailboxRules_t1", "b2", 4, mailboxStarted); - Assert.Equal(4, (await RowsAsync(q, q.QueueTable)).Count); - - Assert.Equal(["MailboxRules_t1/b1", "MailboxRules_t1/b2", "AuditLog_t1/s1"], await DrainOrderAsync(q.Queue)); - } - - [Fact] - public async Task AMigrationInterruptedPartWay_ConvergesOnRerun() - { - var q = NewQueue(); - await SeedV2Async(q, "Run", "t1", At(0)); - await SeedV2Async(q, "Run", "t2", At(0)); - // t1 already re-keyed by a pass that crashed before deleting its old row or finishing t2. - var epoch = At(0).AddMinutes(-3); - var v3Key = JobQueueStore.BuildRowKey(epoch, "Run", "t1"); - await q.Store.UpsertAsync(q.QueueTable, new StoreRow("P04", v3Key) - { - Properties = { ["RunName"] = "Run", ["TaskId"] = "t1", ["Priority"] = 4, ["Owner"] = "", ["QueuedUtc"] = new DateTimeOffset(At(0), TimeSpan.Zero) } - }); - - await q.Queue.InitializeAsync(); - - var rows = await RowsAsync(q, q.QueueTable); - Assert.Equal(2, rows.Count); - Assert.All(rows, r => Assert.Equal(epoch, JobQueueStore.ParseEpoch(r.RowKey))); - } -} diff --git a/tests/Craft.Tests/JobQueueIndexBackfillTests.cs b/tests/Craft.Tests/JobQueueIndexBackfillTests.cs deleted file mode 100644 index 243b0ec..0000000 --- a/tests/Craft.Tests/JobQueueIndexBackfillTests.cs +++ /dev/null @@ -1,201 +0,0 @@ -using Craft.Configuration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// The one-time build of the run index, for a queue table that predates it. -/// -/// This is the only part of the index change that touches a queue nobody wrote through the new enqueue -/// path, and it runs against the worst queue in the estate: the instance that motivated the change was -/// carrying ~743,000 rows across 12,028 runs. Two properties matter and neither is observable from the -/// happy path. -/// -/// 1. It actually indexes what is already there. Without this every pre-existing run reads as having -/// no queued tasks, and the orphan re-drive re-queues a backlog that already has rows — the -/// duplicate-execution failure the queue exists to prevent. -/// 2. It runs exactly once, ever. It is a full pass over the queue table, awaited before the service -/// serves traffic, so a version that re-ran on every start would add a multi-minute stall to every -/// restart of the largest instances. -/// -public class JobQueueIndexBackfillTests -{ - private static DateTime At(int minute) => new(2026, 8, 9, 2, minute, 0, DateTimeKind.Utc); - - private static (JobQueueStore Queue, RunRemainingCounterTests.ConditionalStore Store, string QueueTable) NewQueue() - { - var settings = new CraftSettings(); - var store = new RunRemainingCounterTests.ConditionalStore(); - var queue = new JobQueueStore(NullLogger.Instance, settings, store); - return (queue, store, $"{settings.Orchestrator.TablePrefix}Queue"); - } - - /// - /// A queue row exactly as the v1 code wrote it: the legacy time-prefixed key - /// ({ticks:D19}-{run}-{task}), no QueuedUtc property, no index row anywhere. This is what - /// the v2 migration must re-key. - /// - private static Task SeedLegacyRowAsync(RunRemainingCounterTests.ConditionalStore store, string queueTable, - string runName, string taskId, int priority, DateTime queuedUtc) - { - var legacyKey = $"{queuedUtc.Ticks.ToString("D19", System.Globalization.CultureInfo.InvariantCulture)}-{runName}-{taskId}"; - return store.UpsertAsync(queueTable, - new StoreRow(JobQueueStore.Bucket(priority), legacyKey) - { - Properties = - { - ["RunName"] = runName, - ["TaskId"] = taskId, - ["Priority"] = priority, - ["Owner"] = "", - ["LeaseUntil"] = (DateTimeOffset?)null, - } - }); - } - - [Fact] - public async Task BuildsTheIndexForAQueueThatPredatesIt() - { - var (queue, store, queueTable) = NewQueue(); - - await SeedLegacyRowAsync(store, queueTable, "run-a", "task-0", 4, At(0)); - await SeedLegacyRowAsync(store, queueTable, "run-a", "task-1", 4, At(0)); - await SeedLegacyRowAsync(store, queueTable, "run-b", "task-9", 0, At(1)); - - await queue.InitializeAsync(); - - Assert.Equal(["task-0", "task-1"], - (await queue.GetQueuedTaskIdsAsync("run-a")).OrderBy(x => x, StringComparer.Ordinal)); - Assert.Equal(["task-9"], await queue.GetQueuedTaskIdsAsync("run-b")); - } - - /// - /// v2 re-keys a legacy time-prefixed row to the deterministic {run}|{task} scheme, carrying its - /// enqueue time into the QueuedUtc property, deleting the old row, and leaving the task claimable once. - /// - [Fact] - public async Task MigratesLegacyRowsToTheDeterministicKeyScheme() - { - var (queue, store, queueTable) = NewQueue(); - await SeedLegacyRowAsync(store, queueTable, "run-a", "task-0", 4, At(3)); - - await queue.InitializeAsync(); - - var rows = new List(); - await foreach (var row in store.QueryTableAsync(queueTable)) rows.Add(row); - - // The legacy row is gone; one row remains, keyed deterministically and carrying QueuedUtc. - var only = Assert.Single(rows); - Assert.Equal(JobQueueStore.BuildRowKey(At(3), "run-a", "task-0"), only.RowKey); - Assert.Equal(new DateTimeOffset(At(3), TimeSpan.Zero), only.GetDateTimeOffset("QueuedUtc")); - - // And it is still claimable, exactly once. - Assert.Equal("task-0", - Assert.Single(await queue.ClaimBatchAsync("w", 8, TimeSpan.FromMinutes(20))).TaskId); - } - - [Fact] - public async Task RunsOnceAndNeverAgain() - { - var (queue, store, queueTable) = NewQueue(); - - await SeedLegacyRowAsync(store, queueTable, "run-a", "task-0", 4, At(0)); - await queue.InitializeAsync(); - Assert.Equal(["task-0"], await queue.GetQueuedTaskIdsAsync("run-a")); - - // A second un-indexed row, then a fresh store instance over the SAME tables so the in-process - // _initialized latch cannot be what short-circuits it. Only the persisted marker can. - await SeedLegacyRowAsync(store, queueTable, "run-a", "task-1", 4, At(2)); - - var second = new JobQueueStore(NullLogger.Instance, new CraftSettings(), store); - await second.InitializeAsync(); - - // Deliberately asserting the un-indexed row is NOT picked up: that is what proves the backfill - // short-circuited rather than silently re-running. A rebuild would report both tasks. - Assert.Equal(["task-0"], await second.GetQueuedTaskIdsAsync("run-a")); - } - - [Fact] - public async Task IndexedRowsSurviveARemoveRun() - { - var (queue, store, queueTable) = NewQueue(); - - await SeedLegacyRowAsync(store, queueTable, "run-a", "task-0", 4, At(0)); - await SeedLegacyRowAsync(store, queueTable, "run-b", "task-9", 4, At(0)); - await queue.InitializeAsync(); - - await queue.RemoveRunAsync("run-a"); - - Assert.Empty(await queue.GetQueuedTaskIdsAsync("run-a")); - Assert.Equal(["task-9"], await queue.GetQueuedTaskIdsAsync("run-b")); - - // The backfilled queue rows must be gone too, not just their index entries — otherwise a - // cancelled run's work stays claimable. - var left = new List(); - await foreach (var row in store.QueryTableAsync(queueTable)) - { - if (row.GetString("RunName") == "run-a") left.Add(row.RowKey); - } - Assert.Empty(left); - } - - /// - /// Run names carry a user-supplied scheduled-task name, and Azure Tables rejects '/', '\', '#', '?' - /// and control characters in a key. A real one in production is - /// "UserTaskOrchestrator_AllTenants: Alert on Huntress Rogue Apps detected-{guid}"; nothing stops the - /// next one containing a slash. Unescaped, the index write throws and the run has no index at all. - /// - [Theory] - [InlineData("run/with/slashes")] - [InlineData("run?with=query")] - [InlineData(@"run\with\backslash")] - [InlineData("run#with-hash")] - [InlineData("run%already-escaped")] - [InlineData("UserTaskOrchestrator_AllTenants: Alert on 100% CPU? detected-abc123")] - public async Task RunNamesWithKeyIllegalCharactersRoundTrip(string runName) - { - var (queue, _, _) = NewQueue(); - await queue.InitializeAsync(); - - await queue.EnqueueBatchAsync(runName, [("task-0", 4)], At(0)); - await queue.EnqueueBatchAsync("run-plain", [("task-9", 4)], At(0)); - - var partition = JobQueueStore.IndexPartition(runName); - Assert.DoesNotContain(partition, c => c is '/' or '\\' or '#' or '?' || char.IsControl(c)); - - Assert.Equal(["task-0"], await queue.GetQueuedTaskIdsAsync(runName)); - Assert.Equal(["task-9"], await queue.GetQueuedTaskIdsAsync("run-plain")); - } - - /// Distinct run names must not collide once escaped, or one run's cleanup drops another's. - [Fact] - public async Task EscapingDoesNotCollideAcrossRunNames() - { - // "a/b" escapes to "a%2Fb"; a run literally named "a%2Fb" must land somewhere else, which is why - // '%' is itself escaped. - Assert.NotEqual(JobQueueStore.IndexPartition("a/b"), JobQueueStore.IndexPartition("a%2Fb")); - - var (queue, _, _) = NewQueue(); - await queue.InitializeAsync(); - - await queue.EnqueueBatchAsync("a/b", [("slash", 4)], At(0)); - await queue.EnqueueBatchAsync("a%2Fb", [("literal", 4)], At(0)); - - Assert.Equal(["slash"], await queue.GetQueuedTaskIdsAsync("a/b")); - Assert.Equal(["literal"], await queue.GetQueuedTaskIdsAsync("a%2Fb")); - } - - [Fact] - public void IndexRowKeySplitsAtTheBucketBoundary_EvenWithAPipeInTheRowKey() - { - // The queue row key embeds the run name, so a '|' can appear inside it. Splitting on the first - // '|' rather than at the fixed bucket width would address the wrong queue row. - const string QueueRowKey = "0000000638000000000000000-run|odd-task-0"; - var split = JobQueueStore.SplitIndexRowKey(JobQueueStore.IndexRowKey("P04", QueueRowKey)); - - Assert.NotNull(split); - Assert.Equal("P04", split!.Value.Bucket); - Assert.Equal(QueueRowKey, split.Value.QueueRowKey); - } -} diff --git a/tests/Craft.Tests/JobQueuePumpBackoffTests.cs b/tests/Craft.Tests/JobQueuePumpBackoffTests.cs deleted file mode 100644 index 2375fa8..0000000 --- a/tests/Craft.Tests/JobQueuePumpBackoffTests.cs +++ /dev/null @@ -1,276 +0,0 @@ -using Craft.Configuration; -using Craft.Orchestration; -using Craft.PowerShellHost; -using Craft.Storage; -using Microsoft.Extensions.Configuration; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -[CollectionDefinition(Name, DisableParallelization = true)] -public class PumpTiming -{ - public const string Name = "job-queue-pump-timing"; -} - -/// -/// The pump polls storage for work, so an idle instance was scanning the queue table once a second -/// forever. Backing off fixes that, but the interval is also a hard throughput ceiling — a refill hands -/// over at most batchSize jobs per tick, so a flat 10s poll caps the whole system at batchSize/10 tasks -/// per second. On a 7,336-task fan-out that is hours of waiting no matter how fast the tasks are. -/// -/// So the backoff has to key off the right signal. "Claimed nothing this tick" is NOT idleness: the -/// pump also claims nothing while its buffer is above the low-water mark, which is precisely when a -/// busy run is about to need its next batch. Idle means claimed nothing AND holding nothing. -/// -/// These tests pin both directions — that a quiet pump slows down, and that a working one does not. -/// They count scans in sub-second wall-clock windows, so they run alone: sharing a small CI runner with -/// parallel collections starved the 100ms ticks enough to read as a backoff. -/// -[Collection(PumpTiming.Name)] -public class JobQueuePumpBackoffTests -{ - private static (JobQueuePump Pump, JobQueueStore Queue, JobManager Jobs) NewPump( - int pollMs, int idlePollMs, int backlog = 0, int batch = 4, int lowWater = 2, int? poolSize = null) - { - var settings = new CraftSettings(); - // Separate knobs on purpose: with batch == pool every claimed job starts immediately, the buffer - // empties into the running state and the pump keeps claiming. A pool SMALLER than the batch is - // what leaves work sitting in the buffer — the state where the pump holds work but claims none. - settings.Worker.BgPoolSize = poolSize ?? batch; - - var config = new ConfigurationBuilder().AddInMemoryCollection(new Dictionary - { - ["JobQueueBatchSize"] = batch.ToString(System.Globalization.CultureInfo.InvariantCulture), - ["JobQueueLowWaterMark"] = lowWater.ToString(System.Globalization.CultureInfo.InvariantCulture), - ["JobQueuePollIntervalMs"] = pollMs.ToString(System.Globalization.CultureInfo.InvariantCulture), - ["JobQueueIdlePollIntervalMs"] = idlePollMs.ToString(System.Globalization.CultureInfo.InvariantCulture), - }).Build(); - - var backing = new CountingStore(); - var queue = new JobQueueStore(NullLogger.Instance, settings, backing); - queue.InitializeAsync().GetAwaiter().GetResult(); - - if (backlog > 0) - { - queue.EnqueueBatchAsync("run", - Enumerable.Range(0, backlog).Select(i => ($"task-{i:D5}", 4)).ToList(), - new DateTime(2026, 8, 9, 2, 0, 0, DateTimeKind.Utc)).GetAwaiter().GetResult(); - } - - backing.Scans = 0; // ignore the enqueue traffic - - var repo = new ScriptRepository(NullLogger.Instance, settings); - var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); - var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); - var jobs = new JobManager(NullLogger.Instance, settings, limiter); - - var pump = new JobQueuePump(NullLogger.Instance, queue, jobs, config, settings); - return (pump, queue, jobs); - } - - private static async Task PumpFor(JobQueuePump pump, int ms) - { - await pump.StartAsync(CancellationToken.None); - await Task.Delay(ms); - await Task.WhenAny(pump.StopAsync(CancellationToken.None), Task.Delay(3000)); - } - - /// Counts table scans, which is exactly what the backoff is meant to reduce. - private sealed class CountingStore : ICraftTableStore - { - private readonly Dictionary> _tables = new(); - private readonly object _sync = new(); - - public int Scans; - - public Task PingAsync(CancellationToken ct = default) => Task.CompletedTask; - - public Task EnsureTableAsync(string table, CancellationToken ct = default) - { - lock (_sync) { if (!_tables.ContainsKey(table)) _tables[table] = new(); } - return Task.CompletedTask; - } - - public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) - { - lock (_sync) - { - if (!_tables.ContainsKey(table)) _tables[table] = new(); - _tables[table][(row.PartitionKey, row.RowKey)] = row; - } - return Task.CompletedTask; - } - - public Task UpsertBatchAsync(string table, string pk, IReadOnlyList rows, CancellationToken ct = default) - { - lock (_sync) - { - if (!_tables.ContainsKey(table)) _tables[table] = new(); - foreach (var r in rows) _tables[table][(r.PartitionKey, r.RowKey)] = r; - } - return Task.CompletedTask; - } - - public Task TryReplaceBatchAsync(string table, string pk, IReadOnlyList rows, - CancellationToken ct = default) - { - lock (_sync) - { - if (!_tables.ContainsKey(table)) _tables[table] = new(); - foreach (var r in rows) _tables[table][(r.PartitionKey, r.RowKey)] = r; - } - return Task.FromResult(true); - } - - public Task GetAsync(string table, string pk, string rk, CancellationToken ct = default) - { - lock (_sync) - return Task.FromResult(_tables.TryGetValue(table, out var t) - && t.TryGetValue((pk, rk), out var r) ? r : null); - } - - public async IAsyncEnumerable QueryPartitionAsync(string table, string pk, - [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) - { - List snap; - lock (_sync) - snap = _tables.TryGetValue(table, out var t) - ? t.Where(k => k.Key.Item1 == pk).Select(k => k.Value).ToList() - : new List(); - foreach (var r in snap) { yield return r; await Task.Yield(); } - } - - public async IAsyncEnumerable QueryTableAsync(string table, - [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) - { - Interlocked.Increment(ref Scans); - List snap; - lock (_sync) snap = _tables.TryGetValue(table, out var t) ? t.Values.ToList() : new List(); - foreach (var r in snap) { yield return r; await Task.Yield(); } - } - - public Task DeleteAsync(string table, string pk, string rk, CancellationToken ct = default) - { - lock (_sync) { if (_tables.TryGetValue(table, out var t)) t.Remove((pk, rk)); } - return Task.CompletedTask; - } - - public Task DeletePartitionAsync(string table, string pk, CancellationToken ct = default) - { - lock (_sync) - { - if (!_tables.TryGetValue(table, out var t)) return Task.CompletedTask; - foreach (var k in t.Keys.Where(k => k.Item1 == pk).ToList()) t.Remove(k); - } - return Task.CompletedTask; - } - } - - [Fact] - public async Task AnIdlePump_BacksOff_InsteadOfScanningEveryTick() - { - // Empty queue: every tick claims nothing and holds nothing. - var (pump, queue, _) = NewPump(pollMs: 100, idlePollMs: 2000); - var backing = (CountingStore)GetBacking(queue); - - await PumpFor(pump, 900); - - // At a flat 100ms this window is ~9 scans. Doubling (100/200/400/800/1600...) allows about 4. - Assert.True(backing.Scans <= 5, - $"idle pump scanned storage {backing.Scans} times in 900ms — it is not backing off"); - Assert.True(backing.Scans >= 1, "idle pump never polled at all"); - } - - /// - /// The regression a naive backoff would cause, measured where it actually shows: throughput while a - /// consumer is draining the buffer. - /// - /// A refill hands over at most batchSize jobs, and only happens once per tick, so the poll interval - /// is a hard ceiling of batchSize/interval. If the pump backed off because a tick claimed nothing — - /// which it legitimately does whenever the buffer is above the low-water mark — a busy run would be - /// throttled to batchSize per IDLE interval instead. Here that is the difference between ~9 refills - /// in the window and ~4. - /// - /// Scans rather than polls is the right unit: RefillAsync returns without touching storage while the - /// buffer is full, so a scan happens exactly when the pump needed more work. - /// - [Fact] - public async Task APumpWhoseBufferIsDraining_RefillsAtTheBaseInterval() - { - var (pump, queue, jobs) = NewPump(pollMs: 100, idlePollMs: 2000, backlog: 200); - var backing = (CountingStore)GetBacking(queue); - - // A consumer, so the buffer actually draws down and refills are needed — without one the pump - // fills once and correctly never scans again. - // The window opens once the consumer has run a job, so its startup is not billed to the pump. - var draining = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); - jobs.SetWorkResolver((_, _) => - { - draining.TrySetResult(); - return Task.FromResult?>(_ => Task.CompletedTask); - }); - _ = Task.Run(() => jobs.StartAsync(CancellationToken.None)); - - await pump.StartAsync(CancellationToken.None); - await draining.Task.WaitAsync(TimeSpan.FromSeconds(10)); - var before = backing.Scans; - await Task.Delay(900); - var scans = backing.Scans - before; - await Task.WhenAny(pump.StopAsync(CancellationToken.None), Task.Delay(3000)); - await jobs.StopAsync(CancellationToken.None); - - Assert.True(scans >= 6, - $"a draining buffer was only refilled {scans} times in 900ms — the pump backed off " + - "while work was flowing, which caps throughput at batchSize per idle interval"); - } - - /// - /// The case that decides the idle SIGNAL, as opposed to the backoff curve. - /// - /// While long tasks run, the buffer stays above the low-water mark and the pump claims nothing tick - /// after tick. Treating "claimed nothing" as idle backs the loop off during exactly that stretch, and - /// the delay is then paid at the worst moment: the buffer finally drains and the refill that should - /// have taken one base interval takes an idle one. That is why idleness requires holding nothing as - /// well as claiming nothing. - /// - /// Long tasks are modelled by blocking the consumer, then released so the buffer drains. The window - /// after release is short enough that a backed-off pump cannot have polled in it at all. - /// - [Fact] - public async Task AfterALongStretchOfHoldingWork_TheNextRefillIsStillPrompt() - { - // Batch 8 into a pool of 1: one job runs, seven sit in the buffer above the low-water mark, so - // the pump claims nothing while still holding claims. That is the stretch under test. - var (pump, queue, jobs) = NewPump(pollMs: 100, idlePollMs: 3000, backlog: 200, batch: 8, poolSize: 1); - var backing = (CountingStore)GetBacking(queue); - - var release = new SemaphoreSlim(0); - jobs.SetWorkResolver((_, _) => Task.FromResult?>( - async _ => await release.WaitAsync(TimeSpan.FromSeconds(10), CancellationToken.None))); - _ = Task.Run(() => jobs.StartAsync(CancellationToken.None)); - - await pump.StartAsync(CancellationToken.None); - - // Tasks are stuck, so the buffer stays full and nothing is claimed for many ticks. - await Task.Delay(800); - var scansWhileHolding = backing.Scans; - - // Let everything finish; the buffer now drains and the pump must top it up at the base interval. - release.Release(1000); - await Task.Delay(300); - var scansAfterDrain = backing.Scans - scansWhileHolding; - - await Task.WhenAny(pump.StopAsync(CancellationToken.None), Task.Delay(3000)); - await jobs.StopAsync(CancellationToken.None); - - Assert.True(scansAfterDrain >= 2, - $"only {scansAfterDrain} refill(s) in the 300ms after the buffer drained — the pump had backed " + - "off during the stretch where it held work but claimed nothing, so the next batch was late"); - } - - private static object GetBacking(JobQueueStore queue) => - typeof(JobQueueStore).GetField("_store", - System.Reflection.BindingFlags.NonPublic | System.Reflection.BindingFlags.Instance)! - .GetValue(queue)!; -} diff --git a/tests/Craft.Tests/JobQueuePumpTests.cs b/tests/Craft.Tests/JobQueuePumpTests.cs deleted file mode 100644 index 60f5cf5..0000000 --- a/tests/Craft.Tests/JobQueuePumpTests.cs +++ /dev/null @@ -1,224 +0,0 @@ -using Craft.Configuration; -using Craft.Orchestration; -using Craft.PowerShellHost; -using Craft.Storage; -using Microsoft.Extensions.Configuration; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// The pump that keeps the in-memory queue small by feeding it from storage a batch at a time. -/// -/// The property under test is the one the whole exercise is for: however deep the backlog, this process -/// holds a worker-pool-sized buffer and no more. A run of 7,336 tasks is 7,336 rows in storage and a -/// handful of objects here. -/// -/// It is a separate pump rather than a change to the dispatch loop on purpose — that loop owns the -/// limiter slot lifecycle whose invariants were written to close a leak that wedged production for 28 -/// hours, and feeding it through the enqueue path it already has leaves all of that untouched. -/// -public class JobQueuePumpTests -{ - private static (JobQueuePump Pump, JobQueueStore Queue, JobManager Jobs) NewPump( - int batch = 4, int lowWater = 2, int backlog = 0) - { - var settings = new CraftSettings(); - settings.Worker.BgPoolSize = batch; - - var config = new ConfigurationBuilder().AddInMemoryCollection(new Dictionary - { - ["JobQueueBatchSize"] = batch.ToString(System.Globalization.CultureInfo.InvariantCulture), - ["JobQueueLowWaterMark"] = lowWater.ToString(System.Globalization.CultureInfo.InvariantCulture), - ["JobQueuePollIntervalMs"] = "100", - }).Build(); - - var backing = new RunRemainingCounterTests.ConditionalStore(); - var queue = new JobQueueStore(NullLogger.Instance, settings, backing); - queue.InitializeAsync().GetAwaiter().GetResult(); - - if (backlog > 0) - { - queue.EnqueueBatchAsync("StandardsApply", - Enumerable.Range(0, backlog).Select(i => ($"task-{i:D5}", 4)).ToList(), - new DateTime(2026, 8, 9, 2, 0, 0, DateTimeKind.Utc)).GetAwaiter().GetResult(); - } - - var repo = new ScriptRepository(NullLogger.Instance, settings); - var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); - var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); - var jobs = new JobManager(NullLogger.Instance, settings, limiter); - - var pump = new JobQueuePump(NullLogger.Instance, queue, jobs, config, settings); - return (pump, queue, jobs); - } - - /// Run the pump without a dispatch loop consuming, so the buffer state is observable. - private static async Task PumpFor(JobQueuePump pump, int ms) - { - await pump.StartAsync(CancellationToken.None); - await Task.Delay(ms); - await Task.WhenAny(pump.StopAsync(CancellationToken.None), Task.Delay(3000)); - } - - /// - /// THE GUARANTEE. A backlog far larger than the pool must not end up in memory. Before this, all - /// 7,336 descriptors of a StandardsApply run sat in the process; now the table holds them. - /// - [Fact] - public async Task DeepBacklogNeverLandsInMemoryAllAtOnce() - { - var (pump, _, jobs) = NewPump(batch: 4, lowWater: 2, backlog: 500); - - await PumpFor(pump, 500); - - // Nothing is consuming, so the buffer sits at whatever one refill put there — never the backlog. - Assert.InRange(jobs.QueuedCount, 1, 8); - } - - [Fact] - public async Task RefillsOnlyOnceTheBufferHasDrawnDown() - { - var (pump, _, jobs) = NewPump(batch: 4, lowWater: 2, backlog: 100); - - await PumpFor(pump, 400); - var afterFirst = jobs.QueuedCount; - - // Above the low-water mark, so repeated cycles must not keep claiming. - Assert.InRange(afterFirst, 1, 8); - - await PumpFor(pump, 400); - Assert.InRange(jobs.QueuedCount, 1, 8); - } - - [Fact] - public async Task ClaimsNothingWhenTheQueueIsEmpty() - { - var (pump, _, jobs) = NewPump(backlog: 0); - - await PumpFor(pump, 300); - - Assert.Equal(0, jobs.QueuedCount); - } - - /// - /// Claimed rows stay in storage until the work is done. Deleting on claim would take the task with it - /// if this instance died holding the batch — the lease, not deletion, is what stops a double run. - /// - [Fact] - public async Task ClaimedRowsSurviveUntilTheWorkFinishes() - { - var (pump, queue, jobs) = NewPump(batch: 3, lowWater: 2, backlog: 3); - - await PumpFor(pump, 300); - Assert.True(jobs.QueuedCount > 0, "the pump should have claimed a batch"); - - // Still owned by us and still present, so nobody else can take them... - Assert.Empty(await queue.ClaimBatchAsync("someone-else", 3, TimeSpan.FromMinutes(20))); - - // ...and once the leases lapse they are reclaimable, which is the crash-recovery path. - var reclaimed = await queue.ClaimBatchAsync("someone-else", 3, TimeSpan.FromMinutes(20)); - Assert.Empty(reclaimed); - } - - [Fact] - public async Task SurvivesAStoreThatThrows() - { - var settings = new CraftSettings(); - settings.Worker.BgPoolSize = 4; - var config = new ConfigurationBuilder().AddInMemoryCollection(new Dictionary - { - ["JobQueuePollIntervalMs"] = "100", - }).Build(); - - var repo = new ScriptRepository(NullLogger.Instance, settings); - var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); - var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); - var jobs = new JobManager(NullLogger.Instance, settings, limiter); - - // A store whose every read throws — a pump that dies here would look exactly like an empty queue. - var queue = new JobQueueStore(NullLogger.Instance, settings, new ThrowingStore()); - var pump = new JobQueuePump(NullLogger.Instance, queue, jobs, config, settings); - - var ex = await Record.ExceptionAsync(() => PumpFor(pump, 350)); - - Assert.Null(ex); - } - - /// - /// The wake signal: an enqueue must start the pump now, not on its next poll tick. With a - /// deliberately long poll interval, a task queued after the pump has gone quiet is still claimed - /// promptly — which can only happen if the enqueue woke it. Without the signal this waits the full - /// poll interval, which on a cold system had backed off toward its idle ceiling. - /// - [Fact] - public async Task AnEnqueueWakesThePumpBeforeTheNextPollTick() - { - var settings = new CraftSettings(); - settings.Worker.BgPoolSize = 4; - var config = new ConfigurationBuilder().AddInMemoryCollection(new Dictionary - { - ["JobQueueBatchSize"] = "4", - ["JobQueueLowWaterMark"] = "2", - // Long enough that a poll-driven claim would miss the assertion window by an order of magnitude. - ["JobQueuePollIntervalMs"] = "5000", - ["JobQueueIdlePollIntervalMs"] = "5000", - }).Build(); - - var backing = new RunRemainingCounterTests.ConditionalStore(); - var queue = new JobQueueStore(NullLogger.Instance, settings, backing); - await queue.InitializeAsync(); - - var repo = new ScriptRepository(NullLogger.Instance, settings); - var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); - var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); - var jobs = new JobManager(NullLogger.Instance, settings, limiter); - var pump = new JobQueuePump(NullLogger.Instance, queue, jobs, config, settings); - - await pump.StartAsync(CancellationToken.None); - try - { - // Let the first (empty) cycle run and the pump settle into its long wait. - await Task.Delay(200); - Assert.Equal(0, jobs.QueuedCount); - - await queue.EnqueueBatchAsync("run", - new[] { ("task-1", 4) }, - new DateTime(2026, 8, 9, 2, 0, 0, DateTimeKind.Utc)); - - // Well under the 5s poll: only the wake can explain a claim this fast. - var claimed = await WaitUntilAsync(() => jobs.QueuedCount > 0, TimeSpan.FromMilliseconds(1500)); - Assert.True(claimed, - "the pump did not claim the enqueued task within 1.5s despite a 5s poll — the wake signal did not fire"); - } - finally - { - await Task.WhenAny(pump.StopAsync(CancellationToken.None), Task.Delay(3000)); - } - } - - private static async Task WaitUntilAsync(Func condition, TimeSpan timeout) - { - var sw = System.Diagnostics.Stopwatch.StartNew(); - while (sw.Elapsed < timeout) - { - if (condition()) return true; - await Task.Delay(20); - } - return condition(); - } - - private sealed class ThrowingStore : ICraftTableStore - { - public Task PingAsync(CancellationToken ct = default) => Task.CompletedTask; - public Task EnsureTableAsync(string table, CancellationToken ct = default) => Task.CompletedTask; - public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) => throw new InvalidOperationException("store down"); - public Task UpsertBatchAsync(string table, string pk, IReadOnlyList rows, CancellationToken ct = default) => throw new InvalidOperationException("store down"); - public Task TryReplaceBatchAsync(string table, string pk, IReadOnlyList rows, CancellationToken ct = default) => throw new InvalidOperationException("store down"); - public Task GetAsync(string table, string pk, string rk, CancellationToken ct = default) => throw new InvalidOperationException("store down"); - public IAsyncEnumerable QueryPartitionAsync(string table, string pk, CancellationToken ct = default) => throw new InvalidOperationException("store down"); - public IAsyncEnumerable QueryTableAsync(string table, CancellationToken ct = default) => throw new InvalidOperationException("store down"); - public Task DeleteAsync(string table, string pk, string rk, CancellationToken ct = default) => throw new InvalidOperationException("store down"); - public Task DeletePartitionAsync(string table, string pk, CancellationToken ct = default) => throw new InvalidOperationException("store down"); - } -} diff --git a/tests/Craft.Tests/JobQueueRetentionTests.cs b/tests/Craft.Tests/JobQueueRetentionTests.cs deleted file mode 100644 index 678919c..0000000 --- a/tests/Craft.Tests/JobQueueRetentionTests.cs +++ /dev/null @@ -1,280 +0,0 @@ -using Craft.Configuration; -using Craft.Orchestration; -using Craft.PowerShellHost; -using Microsoft.Extensions.Configuration; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// Heap-delta measurement cannot share a process with concurrently-allocating tests — xUnit runs -/// collections in parallel, and another collection's allocations land in the same -/// GC.GetTotalMemory window. Observed swinging the same measurement from 725 to 1181 B/task. -/// This collection runs alone. -/// -[CollectionDefinition(RetentionMeasurement.Name, DisableParallelization = true)] -public class RetentionMeasurement -{ - public const string Name = "retention-measurement"; -} - -/// -/// Measures what the JobManager queue actually retains per queued job. -/// -/// The claim under test is that queue entries pin whole graphs and are -/// therefore a primary driver of heap growth under a deep backlog (production peaked at 783 queued with -/// 3.7-hour waits). Retention is measured, not assumed: each case builds the graph, settles the GC, and -/// diffs GC.GetTotalMemory(forceFullCollection: true). -/// -/// The measurement deliberately holds ONE run graph alive across every case, mirroring -/// OrchestratorService._activeRuns — which pins the run from DispatchPendingTasks -/// (TryAdd) until FinalizeRunAsync (TryRemove), i.e. for the entire time its tasks sit queued. -/// So the reported delta is the queue's MARGINAL cost, which is the only part a descriptor rewrite -/// can actually reclaim. -/// -[Collection(RetentionMeasurement.Name)] -public class JobQueueRetentionTests -{ - private const int Jobs = 5000; - - /// Bytes below which a per-job cost is not worth a redesign on a 2398MB heap cap. - private const int NoiseFloorBytesPerJob = 32; - - private static JobManager NewJobManager(int bgPoolSize = 8) - { - var settings = new CraftSettings(); - settings.Worker.BgPoolSize = bgPoolSize; - var config = new ConfigurationBuilder().AddInMemoryCollection([]).Build(); - var repo = new ScriptRepository(NullLogger.Instance, settings); - var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); - var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); - return new JobManager(NullLogger.Instance, settings, limiter); - } - - /// A task payload shaped like a real CIPP fan-out item (tenant + collection + queue refs). - private static OrchestratorTaskItem NewTask(int i) => new() - { - Id = $"Graph_tenant{i:D4}.onmicrosoft.com", - Status = "Pending", - Parameters = new Dictionary - { - ["TenantFilter"] = $"tenant{i:D4}.onmicrosoft.com", - ["CollectionType"] = "Graph", - ["Name"] = $"CIPPDBCacheRun-Graph_tenant{i:D4}.onmicrosoft.com", - ["FunctionName"] = "Invoke-CIPPDBCacheTask", - ["QueueName"] = "cippdbcache", - ["BatchNumber"] = 1L, - }, - }; - - private static OrchestratorRun NewRun(int taskCount) - { - var run = new OrchestratorRun - { - Name = "CIPPDBCacheRun", - Status = "Running", - Priority = 5, - StartedUtc = DateTime.UtcNow, - TaskScriptName = "Invoke-CIPPDBCacheTask", - }; - for (var i = 0; i < taskCount; i++) run.Tasks.Add(NewTask(i)); - return run; - } - - private static void Settle() - { - for (var i = 0; i < 3; i++) - { - GC.Collect(2, GCCollectionMode.Forced, blocking: true, compacting: true); - GC.WaitForPendingFinalizers(); - } - } - - /// - /// Heap bytes retained by whatever returns, over and above the state - /// already alive when it is called. - /// - /// Best-of-N, taking the MINIMUM: measurement error here is one-sided. Anything else allocating - /// during the window — the test host, finalizers, JIT on the first pass — can only inflate the - /// delta, never shrink it below the true retained size. So the smallest observation is the closest - /// to truth, and averaging would bake the noise in. - /// - private static long RetainedBytes(Func build, int trials = 3) - { - var best = long.MaxValue; - - for (var i = 0; i < trials; i++) - { - // Drop the previous trial's graph and settle BEFORE sampling the floor. Without this the - // prior graph is still rooted at `before` and reclaimed inside the window, which drives the - // delta negative — that is how this reported "0 bytes" for a 3.6MB run graph. - Held = null; - Settle(); - - var before = GC.GetTotalMemory(forceFullCollection: true); - Held = build(); - Settle(); - var after = GC.GetTotalMemory(forceFullCollection: true); - - var delta = after - before; - if (delta > 0) best = Math.Min(best, delta); - } - - Held = null; - Assert.NotEqual(long.MaxValue, best); // every trial was non-positive ⇒ the measurement is broken - return best; - } - - /// - /// Roots the graph under measurement across the sampling window. A static field rather than a local - /// so the JIT cannot treat it as dead before GC.GetTotalMemory runs, and so the previous - /// trial's graph can be explicitly released. - /// - private static object? Held; - - /// - /// TODAY: every queued job carries a closure capturing (run, task, taskPath, this) — the shape - /// built by OrchestratorService.DispatchSingleTask. - /// - [Fact] - public void ClosureQueue_MarginalRetentionPerJob_IsMeasured() - { - var run = NewRun(Jobs); // held alive throughout, exactly as _activeRuns does - const string taskPath = "/app/API/Modules/CIPP/Invoke-CIPPDBCacheTask.ps1"; - var sink = new object(); // stands in for the captured `this` (OrchestratorService) - - var bytes = RetainedBytes(() => - { - var jm = NewJobManager(); - foreach (var task in run.Tasks) - { - var captured = task; - jm.Enqueue( - name: $"{run.Name}-{captured.Id}", - priority: run.Priority, - runName: run.Name, - work: _ => - { - // Same captures as the production closure: run graph, task, script path, service. - GC.KeepAlive(run); - GC.KeepAlive(captured); - GC.KeepAlive(taskPath); - GC.KeepAlive(sink); - return Task.CompletedTask; - }); - } - return jm; - }); - - GC.KeepAlive(run); - var perJob = bytes / (double)Jobs; - Assert.Equal(Jobs, NewJobManagerQueueDepthProbe(run)); // sanity: all enqueued, none dispatched - Assert.InRange(perJob, NoiseFloorBytesPerJob, 4096); - TestOutput($"closure queue: {bytes:N0} bytes for {Jobs:N0} jobs = {perJob:N0} bytes/job"); - } - - /// - /// AFTER: the queue carries only — (runName, taskId, priority) — and the - /// work is rehydrated at dispatch. No closure, no delegate, no _pendingWork entry, and no - /// Guid-suffixed second id string. - /// - [Fact] - public void DescriptorQueue_RetainsLessPerJob_ThanTheClosureQueue() - { - var run = NewRun(Jobs); - const string taskPath = "/app/API/Modules/CIPP/Invoke-CIPPDBCacheTask.ps1"; - var sink = new object(); - - var closureBytes = RetainedBytes(() => - { - var jm = NewJobManager(); - foreach (var task in run.Tasks) - { - var captured = task; - jm.Enqueue($"{run.Name}-{captured.Id}", run.Priority, _ => - { - GC.KeepAlive(run); GC.KeepAlive(captured); GC.KeepAlive(taskPath); GC.KeepAlive(sink); - return Task.CompletedTask; - }, run.Name); - } - return jm; - }); - - var descriptorBytes = RetainedBytes(() => - { - var jm = NewJobManager(); - foreach (var task in run.Tasks) - jm.Enqueue(new JobDescriptor(run.Name, task.Id, run.Priority), $"{run.Name}-{task.Id}"); - return jm; - }); - - GC.KeepAlive(run); - var closurePerJob = closureBytes / (double)Jobs; - var descriptorPerJob = descriptorBytes / (double)Jobs; - var saved = 1 - (descriptorPerJob / closurePerJob); - - TestOutput($"BEFORE (closure): {closurePerJob:N0} B/job ({closureBytes:N0} total)"); - TestOutput($"AFTER (descriptor): {descriptorPerJob:N0} B/job ({descriptorBytes:N0} total)"); - TestOutput($"reduction: {saved:P0} ({closurePerJob - descriptorPerJob:N0} B/job)"); - - Assert.True(descriptorBytes < closureBytes, - $"descriptor queue ({descriptorBytes:N0}B) must retain less than the closure queue ({closureBytes:N0}B)"); - } - - /// - /// The load-bearing measurement, and the one that falsifies "the queue is the memory problem". - /// - /// The run graph a closure captures is ALREADY pinned by _activeRuns from - /// DispatchPendingTasks to FinalizeRunAsync, so a descriptor rewrite reclaims the - /// queue's own bookkeeping and nothing else. Measured, both are the same order of magnitude - /// (~730 B), and at production's observed peak of 783 queued jobs the ENTIRE queue is well under - /// 1 MB — 0.02% of the 2398 MB DOTNET_GCHeapHardLimit. The queue cannot be what OOMs the - /// container; this test fails if that ever stops being true. - /// - [Fact] - public void QueueRetention_AtProductionPeak_IsNegligibleAgainstTheHeapCap() - { - const int ProductionPeakQueueDepth = 783; - const long HeapHardLimitBytes = 0x95E00000; // build/Dockerfile DOTNET_GCHeapHardLimit = 2398 MB - - var runBytes = RetainedBytes(() => NewRun(Jobs)); - - var run = NewRun(Jobs); - var queueBytes = RetainedBytes(() => - { - var jm = NewJobManager(); - foreach (var task in run.Tasks) - { - var captured = task; - jm.Enqueue($"{run.Name}-{captured.Id}", run.Priority, _ => - { - GC.KeepAlive(run); GC.KeepAlive(captured); - return Task.CompletedTask; - }, run.Name); - } - return jm; - }); - GC.KeepAlive(run); - - var perJob = queueBytes / (double)Jobs; - var atPeak = (long)(perJob * ProductionPeakQueueDepth); - var shareOfCap = atPeak / (double)HeapHardLimitBytes; - - TestOutput($"run graph ({Jobs:N0} tasks): {runBytes:N0} bytes ({runBytes / (double)Jobs:N0} B/task, pinned by _activeRuns)"); - TestOutput($"queue bookkeeping: {queueBytes:N0} bytes ({perJob:N0} B/job, reclaimable)"); - TestOutput($"at production peak ({ProductionPeakQueueDepth} queued): {atPeak:N0} bytes = {shareOfCap:P3} of the 2398MB cap"); - - Assert.True(shareOfCap < 0.01, - $"queue at peak is {shareOfCap:P2} of the heap cap ({atPeak:N0}B) — if this ever exceeds 1% " + - "the queue really has become a memory driver and this analysis needs redoing"); - } - - private static int NewJobManagerQueueDepthProbe(OrchestratorRun run) - { - var jm = NewJobManager(); - foreach (var t in run.Tasks) jm.Enqueue(t.Id, run.Priority, _ => Task.CompletedTask, run.Name); - return jm.QueuedCount; - } - - private static void TestOutput(string message) => Console.WriteLine($"[retention] {message}"); -} diff --git a/tests/Craft.Tests/JobQueueRunLookupTests.cs b/tests/Craft.Tests/JobQueueRunLookupTests.cs deleted file mode 100644 index b5e68bd..0000000 --- a/tests/Craft.Tests/JobQueueRunLookupTests.cs +++ /dev/null @@ -1,142 +0,0 @@ -using Craft.Configuration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// The lookup that tells a re-drive "waiting" from "lost". -/// -/// RedrivePendingTasks used to decide a Pending task was orphaned when the JobManager did not have it -/// queued or running. Under the pump that is what a BACKLOG is — the pump buffers a worker-pool-sized -/// slice and leaves the rest in storage — so the re-drive re-queued the whole un-started backlog every -/// 60 seconds. Queue RowKeys are prefixed with the enqueue timestamp, so each pass added another row -/// for the same task instead of updating the first, and every copy was independently claimable. -/// -/// Measured live before the fix, on one 124-task run: re-drives of 92, 60, 60, 52, 44, 36 tasks on -/// consecutive ticks, six queue rows for a single task exactly 60s apart, and that task executing six -/// times. -/// -public class JobQueueRunLookupTests -{ - private static JobQueueStore NewQueue() - { - var settings = new CraftSettings(); - var queue = new JobQueueStore(NullLogger.Instance, settings, - new RunRemainingCounterTests.ConditionalStore()); - queue.InitializeAsync().GetAwaiter().GetResult(); - return queue; - } - - private static DateTime At(int minute) => new(2026, 8, 9, 2, minute, 0, DateTimeKind.Utc); - - [Fact] - public async Task QueuedTaskIds_ReportsWaitingWork_SoABacklogIsNotMistakenForOrphans() - { - var queue = NewQueue(); - await queue.EnqueueBatchAsync("run-a", - [("task-0", 4), ("task-1", 4), ("task-2", 4)], At(0)); - await queue.EnqueueBatchAsync("other-run", [("task-9", 4)], At(0)); - - var ids = await queue.GetQueuedTaskIdsAsync("run-a"); - - Assert.Equal(["task-0", "task-1", "task-2"], ids.OrderBy(x => x, StringComparer.Ordinal)); - Assert.DoesNotContain("task-9", ids); - } - - [Fact] - public async Task AClaimedTaskIsStillReported_BecauseItIsRunning_NotLost() - { - // The re-drive must not re-queue work that has been handed to a worker: it is the case that - // produced duplicate rows for tasks already executing. - var queue = NewQueue(); - await queue.EnqueueBatchAsync("run-a", [("task-0", 4), ("task-1", 4)], At(0)); - - var claimed = await queue.ClaimBatchAsync("worker-a", 1, TimeSpan.FromMinutes(20)); - Assert.Single(claimed); - - var ids = await queue.GetQueuedTaskIdsAsync("run-a"); - - Assert.Contains(claimed[0].TaskId, ids); - Assert.Equal(2, ids.Count); - } - - [Fact] - public async Task ATaskWhoseRowIsGone_IsNotReported_SoARealOrphanIsStillRecoverable() - { - // The guard has to stay specific — a task whose row genuinely vanished must still be re-driven, - // which is the whole reason the re-drive exists. - var queue = NewQueue(); - await queue.EnqueueBatchAsync("run-a", [("task-0", 4), ("task-1", 4)], At(0)); - - var claimed = await queue.ClaimBatchAsync("worker-a", 2, TimeSpan.FromMinutes(20)); - await queue.RemoveAsync(claimed.Single(c => c.TaskId == "task-0")); - - var ids = await queue.GetQueuedTaskIdsAsync("run-a"); - - Assert.DoesNotContain("task-0", ids); - Assert.Contains("task-1", ids); - } - - [Fact] - public async Task ARunWithNothingQueued_ReportsNothing() - { - var queue = NewQueue(); - Assert.Empty(await queue.GetQueuedTaskIdsAsync("run-with-no-rows")); - } - - // ── Releasing the claims a crashed process was holding ──────────────────────────────────────── - - /// - /// After a crash the dead process's claims are still live as far as storage is concerned, so nothing - /// can pick those rows up until the lease lapses — up to 30 minutes by default. Since re-dispatch - /// now declines to write duplicate rows for tasks that already have one, the run simply stalls. - /// Recovery has to hand the claims back. - /// - [Fact] - public async Task ReleasingARunsClaims_MakesItsRowsClaimableAgain() - { - var queue = NewQueue(); - await queue.EnqueueBatchAsync("crashed-run", [("task-0", 4), ("task-1", 4)], At(0)); - - // A long lease, as the pump takes: without a release these are untouchable for its full duration. - var held = await queue.ClaimBatchAsync("dead-worker", 2, TimeSpan.FromMinutes(30)); - Assert.Equal(2, held.Count); - Assert.Empty(await queue.ClaimBatchAsync("new-worker", 2, TimeSpan.FromMinutes(30))); - - var released = await queue.ReleaseRunClaimsAsync("crashed-run"); - - Assert.Equal(2, released); - Assert.Equal(2, (await queue.ClaimBatchAsync("new-worker", 2, TimeSpan.FromMinutes(30))).Count); - } - - /// Releasing frees the existing rows rather than adding more — the duplicate-row bug again. - [Fact] - public async Task ReleasingClaims_DoesNotCreateExtraRows() - { - var queue = NewQueue(); - await queue.EnqueueBatchAsync("crashed-run", [("task-0", 4), ("task-1", 4)], At(0)); - await queue.ClaimBatchAsync("dead-worker", 2, TimeSpan.FromMinutes(30)); - - await queue.ReleaseRunClaimsAsync("crashed-run"); - - var ids = await queue.GetQueuedTaskIdsAsync("crashed-run"); - Assert.Equal(["task-0", "task-1"], ids.OrderBy(x => x, StringComparer.Ordinal)); - Assert.Equal(2, (await queue.ClaimBatchAsync("new-worker", 10, TimeSpan.FromMinutes(30))).Count); - } - - /// Another run's claims are left alone — recovery is per run. - [Fact] - public async Task ReleasingOneRunsClaims_LeavesOtherRunsHeld() - { - var queue = NewQueue(); - await queue.EnqueueBatchAsync("crashed-run", [("task-0", 4)], At(0)); - await queue.EnqueueBatchAsync("healthy-run", [("task-9", 4)], At(1)); - await queue.ClaimBatchAsync("worker", 2, TimeSpan.FromMinutes(30)); - - Assert.Equal(1, await queue.ReleaseRunClaimsAsync("crashed-run")); - - var reclaimed = await queue.ClaimBatchAsync("new-worker", 10, TimeSpan.FromMinutes(30)); - Assert.Equal("task-0", Assert.Single(reclaimed).TaskId); - } -} diff --git a/tests/Craft.Tests/JobQueueStatusReaderTests.cs b/tests/Craft.Tests/JobQueueStatusReaderTests.cs index 0dea7d8..5342bfc 100644 --- a/tests/Craft.Tests/JobQueueStatusReaderTests.cs +++ b/tests/Craft.Tests/JobQueueStatusReaderTests.cs @@ -8,226 +8,78 @@ namespace Craft.Tests; /// -/// The table-backed status view. Since task ownership moved into the queue table, the in-memory -/// JobManager holds only a worker-pool-sized buffer — so every status consumer that reads it as "the -/// queue" reports a 7,000-task fan-out as eight queued jobs. These tests pin the merge semantics: the -/// durable backlog is counted and listed, claimed rows are never double-counted against the local -/// records that represent them, and run sizes come from the counter row rather than from whatever -/// slice this instance happened to claim. +/// The status APIs read the durable backlog from the Ready list: every run is counted from its counts, only +/// the head is listed task by task, and work this process holds is not counted as waiting. /// public class JobQueueStatusReaderTests { - private static readonly TimeSpan Lease = TimeSpan.FromMinutes(20); - private static readonly TimeSpan Fresh = TimeSpan.Zero; - - private sealed class Fixture - { - public required RunRemainingCounterTests.ConditionalStore Backing { get; init; } - public required JobQueueStore Queue { get; init; } - public required OrchestratorTableStore Store { get; init; } - public required JobManager Jobs { get; init; } - public required JobQueueStatusReader Reader { get; init; } - } - - private static async Task NewFixtureAsync() + private static (JobQueueStatusReader Reader, WorkStore Store, JobManager Jobs) New() { var settings = new CraftSettings(); - settings.Worker.BgPoolSize = 2; var config = new ConfigurationBuilder().AddInMemoryCollection([]).Build(); var repo = new ScriptRepository(NullLogger.Instance, settings); var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); var jobs = new JobManager(NullLogger.Instance, settings, limiter); - - var backing = new RunRemainingCounterTests.ConditionalStore(); - var queue = new JobQueueStore(NullLogger.Instance, settings, backing); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); - await queue.InitializeAsync(); - await store.InitializeAsync(); - - return new Fixture - { - Backing = backing, - Queue = queue, - Store = store, - Jobs = jobs, - Reader = new JobQueueStatusReader(NullLogger.Instance, jobs, queue, store), - }; + var store = new WorkStore(NullLogger.Instance, settings, new MemoryTableStore()); + return (new JobQueueStatusReader(NullLogger.Instance, jobs, store), store, jobs); } - private static DateTime At(int minute) => new(2026, 8, 12, 3, minute, 0, DateTimeKind.Utc); - - [Fact] - public async Task SummaryCountsTheDurableBacklog_NotJustTheLocalBuffer() + private static Task Create(WorkStore s, string name, int tasks, int minute) { - var f = await NewFixtureAsync(); - await f.Queue.EnqueueBatchAsync("StandardsApply", - Enumerable.Range(0, 120).Select(i => ($"std-{i}", 4)).ToList(), At(5)); - - var summary = await f.Reader.GetSummaryAsync(); - - Assert.Equal(120, summary.Queued); - Assert.Equal(120, summary.QueuedDurable); - Assert.Equal(0, summary.QueuedLocal); - Assert.Equal(At(5), summary.OldestQueuedUtc); - } - - [Fact] - public async Task ASnapshotHoldsOnlyTheHeadOfABigQueue_ButCountsAllOfIt() - { - var f = await NewFixtureAsync(); - var total = JobQueueStatusReader.HeadRows + 500; - await f.Queue.EnqueueBatchAsync("Late", [("l", 4)], At(9)); - await f.Queue.EnqueueBatchAsync("Big", - Enumerable.Range(0, total - 1).Select(i => ($"b{i:D5}", 4)).ToList(), At(1)); - - var snap = await f.Reader.GetAsync(Fresh); - - Assert.Equal(JobQueueStatusReader.HeadRows, snap!.Rows.Count); - Assert.All(snap.Rows, r => Assert.Equal("Big", r.RunName)); - Assert.Equal(total, snap.Total); - Assert.Equal(total, snap.Unclaimed); - Assert.Equal(total - 1, snap.ByRun["Big"].Unclaimed); - Assert.Equal(1, snap.ByRun["Late"].Unclaimed); - } - - [Fact] - public async Task ClaimedRowsAreNotCountedAsQueued() - { - var f = await NewFixtureAsync(); - await f.Queue.EnqueueBatchAsync("run", - Enumerable.Range(0, 10).Select(i => ($"task-{i}", 4)).ToList(), At(0)); - var claimed = await f.Queue.ClaimBatchAsync("worker-a", 4, Lease); - Assert.Equal(4, claimed.Count); - - var summary = await f.Reader.GetSummaryAsync(); - - // The four claimed rows are some instance's buffer — represented by its local records, not by - // the backlog count. - Assert.Equal(6, summary.QueuedDurable); - Assert.Equal(6, summary.Queued); - } - - [Fact] - public async Task JobDetails_ListTheBacklog_WithoutDuplicatingLocallyClaimedWork() - { - var f = await NewFixtureAsync(); - await f.Queue.EnqueueAsync("run", "claimed-here", 4, At(0)); - await f.Queue.EnqueueAsync("run", "claimed-elsewhere", 4, At(1)); - await f.Queue.EnqueueAsync("run", "waiting", 4, At(2)); - - // Claim one locally, the way the pump does: claim the row, enqueue the descriptor locally. Which - // of the two "claimed-*" tasks comes first is no longer time-ordered under schema v2, so drive the - // local record off whatever was actually claimed rather than a hard-coded id. - var mine = await f.Queue.ClaimBatchAsync("this-node", 1, Lease); - var mineId = Assert.Single(mine).TaskId; - f.Jobs.Enqueue(new JobDescriptor("run", mineId, 4), $"run-{mineId}"); - - // Another instance's claim: a row under lease with no local record at all. - var theirs = await f.Queue.ClaimBatchAsync("other-node", 1, Lease); - var theirsId = Assert.Single(theirs).TaskId; - Assert.NotEqual(mineId, theirsId); - - var details = await f.Reader.GetJobDetailsAsync(); - - // The locally-claimed row once (its local record), the still-waiting row once (durable), the row - // claimed by the other instance not at all — its records live there. ("waiting" sorts last, so the - // two claims took the "claimed-*" pair and it is what remains.) - Assert.Equal(2, details.Count); - Assert.Single(details, d => d.Id == $"run-{mineId}"); - Assert.DoesNotContain(details, d => d.Id == $"run-{theirsId}"); - var waiting = Assert.Single(details, d => d.Id == "run-waiting"); - Assert.Equal("Queued", waiting.Status); - Assert.Equal(At(2), waiting.QueuedUtc); - Assert.True(waiting.WaitSeconds > 0); - } - - [Fact] - public async Task JobDetails_RespectStatusFilterAndLimit() - { - var f = await NewFixtureAsync(); - await f.Queue.EnqueueBatchAsync("run", - Enumerable.Range(0, 5).Select(i => ($"task-{i}", 4)).ToList(), At(0)); - - // A non-Queued filter describes claimed work, which only local records know about. - Assert.Empty(await f.Reader.GetJobDetailsAsync(status: "Running")); - - var queuedOnly = await f.Reader.GetJobDetailsAsync(status: "Queued"); - Assert.Equal(5, queuedOnly.Count); - Assert.All(queuedOnly, d => Assert.Equal("Queued", d.Status)); - - Assert.Equal(3, (await f.Reader.GetJobDetailsAsync(limit: 3)).Count); - } - - [Fact] - public async Task RunSummaries_SizeARunFromItsCounter_NotFromTheClaimedSlice() - { - var f = await NewFixtureAsync(); - - // A 10-task run: one durably finished, one claimed into this instance, eight still queued. - await f.Store.InitRemainingAsync("run", 10); - await f.Store.DecrementRemainingAsync("run", 1); - await f.Queue.EnqueueBatchAsync("run", - Enumerable.Range(0, 9).Select(i => ($"task-{i}", 4)).ToList(), At(0)); - var claimed = await f.Queue.ClaimBatchAsync("this-node", 1, Lease); - f.Jobs.Enqueue(new JobDescriptor("run", claimed[0].TaskId, 4), $"run-{claimed[0].TaskId}"); - - var summary = Assert.Single(await f.Reader.GetRunSummariesAsync(), s => s.Name == "run"); - - Assert.Equal(10, summary.Total); - Assert.Equal(9, summary.Queued); // 8 unclaimed + 1 buffered locally - Assert.Equal(1, summary.Completed); // Total − Remaining, durable across restarts + var started = new DateTime(2026, 10, 5, 3, minute, 0, DateTimeKind.Utc); + return s.CreateRunAsync(new RunHeader + { + RunKey = WorkStore.RunKeyFor(name, started), + Name = name, + StartedUtc = started, + TaskScriptName = "Invoke-CraftTask", + }, Enumerable.Range(0, tasks).Select(i => new WorkStore.NewTask($"t{i}", [])).ToList()); } [Fact] - public async Task RunSummaries_SynthesizeARunTheJobManagerHasNeverSeen() + public async Task EveryRunIsCounted_ButOnlyTheHeadIsListed() { - var f = await NewFixtureAsync(); - await f.Store.InitRemainingAsync("cold-run", 50); - await f.Queue.EnqueueBatchAsync("cold-run", - Enumerable.Range(0, 50).Select(i => ($"task-{i}", 3)).ToList(), At(0)); + var (reader, store, _) = New(); + for (var i = 0; i < 60; i++) await Create(store, $"Run{i:D2}", 50, i % 60); - var summary = Assert.Single(await f.Reader.GetRunSummariesAsync(), s => s.Name == "cold-run"); + var snap = (await reader.GetAsync())!; - Assert.Equal(50, summary.Total); - Assert.Equal(50, summary.Queued); - Assert.Equal(3, summary.Priority); - Assert.Equal(0, summary.Running); + Assert.Equal(3_000, snap.Total); + Assert.Equal(3_000, snap.Unclaimed); + Assert.Equal(60, snap.ByRun.Count); + Assert.Equal(JobQueueStatusReader.HeadRows, snap.Rows.Count); + Assert.Equal("Run00", snap.Rows[0].RunName); } [Fact] - public async Task AFailedRefreshServesThePreviousSnapshot() + public async Task WorkThisProcessHolds_IsNotCountedAsWaiting() { - var f = await NewFixtureAsync(); - await f.Queue.EnqueueAsync("run", "task-0", 4, At(0)); - - var first = await f.Reader.GetAsync(Fresh); - Assert.NotNull(first); - Assert.Equal(1, first!.Unclaimed); - - f.Backing.OnBeforeQuery = () => throw new InvalidOperationException("storage is down"); - var second = await f.Reader.GetAsync(Fresh); - - // Stale data with an honest timestamp, not an exception on a health endpoint. - Assert.Same(first, second); + var (reader, store, jobs) = New(); + var run = await Create(store, "Busy", 5, 0); + foreach (var c in await store.ClaimAsync(run.RunKey, 2, "me", TimeSpan.FromMinutes(5), false)) + jobs.Enqueue(new JobDescriptor("Busy", c.TaskId, 4) { RunKey = c.RunKey, Seq = c.Seq }, $"Busy-{c.TaskId}"); + + var snap = (await reader.GetAsync())!; + + Assert.Equal(5, snap.Total); + Assert.Equal(3, snap.Unclaimed); + Assert.Equal(3, snap.Rows.Count); + var summary = await reader.GetSummaryAsync(); + Assert.Equal(3, summary.QueuedDurable); } [Fact] - public async Task SnapshotAggregatesPerRun() + public async Task AFinishedRun_LeavesTheBacklog() { - var f = await NewFixtureAsync(); - await f.Queue.EnqueueBatchAsync("run-a", - Enumerable.Range(0, 3).Select(i => ($"a-{i}", 4)).ToList(), At(0)); - await f.Queue.EnqueueBatchAsync("run-b", - Enumerable.Range(0, 2).Select(i => ($"b-{i}", 1)).ToList(), At(1)); + var (reader, store, _) = New(); + var run = await Create(store, "Quick", 2, 0); + var claims = await store.ClaimAsync(run.RunKey, 2, "w", TimeSpan.FromMinutes(5), false); + await store.FinishAsync(run.RunKey, claims.Select(c => new WorkStore.Finish(c.Seq, "Completed", Owner: "w")).ToList()); - var snap = await f.Reader.GetAsync(Fresh); + var snap = (await reader.GetAsync())!; - Assert.NotNull(snap); - Assert.Equal(5, snap!.Total); - Assert.Equal(3, snap.ByRun["run-a"].Unclaimed); - Assert.Equal(2, snap.ByRun["run-b"].Unclaimed); - Assert.Equal(1, snap.ByRun["run-b"].MinPriority); - Assert.Equal(At(0), snap.OldestUnclaimedUtc); + Assert.Equal(0, snap.Total); + Assert.Empty(snap.ByRun); } } diff --git a/tests/Craft.Tests/JobQueueStatusRefreshCostTests.cs b/tests/Craft.Tests/JobQueueStatusRefreshCostTests.cs deleted file mode 100644 index 330cd7f..0000000 --- a/tests/Craft.Tests/JobQueueStatusRefreshCostTests.cs +++ /dev/null @@ -1,215 +0,0 @@ -using Craft.Configuration; -using Craft.Orchestration; -using Craft.PowerShellHost; -using Craft.Storage; -using Microsoft.Extensions.Configuration; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// What a status refresh COSTS, as opposed to what it reports. -/// -/// The status snapshot is polled — the stats sampler on its timer, the worker-health page at a few -/// hertz, the perf harness at 4 Hz — and its scan is proportional to the backlog. Two properties kept -/// that from being sustainable on a large instance, and neither is visible from the values returned: -/// -/// 1. The snapshot did a per-run counter read while aggregating, so a backlog spanning 12,028 runs -/// cost 12,028 point reads per refresh — on every poll, for data only the run-summaries listing -/// ever read. -/// 2. The snapshot was stamped with the time the refresh STARTED, so a scan slower than the TTL -/// returned something already expired and the next poll immediately started another. Measured on -/// a 743,000-row queue: continuous back-to-back full scans, gated only by the single-flight lock. -/// -/// Both are cost properties, so both are asserted by counting storage calls and by reading the -/// snapshot's own age — not by checking the numbers it reports, which were correct throughout. -/// -public class JobQueueStatusRefreshCostTests -{ - /// Counts reads and can make the queue scan take a controllable amount of time. - private sealed class CountingStore(ICraftTableStore inner, string queueTable) : ICraftTableStore - { - private int _pointReads; - private int _tableScans; - - /// Point reads (GetAsync) issued since the last . - public int PointReads => Volatile.Read(ref _pointReads); - - /// Full scans of the queue table since the last . - public int QueueScans => Volatile.Read(ref _tableScans); - - /// Injected per-scan delay, to model a backlog large enough to outlast the TTL. - public TimeSpan ScanDelay { get; set; } = TimeSpan.Zero; - - public void Reset() { Volatile.Write(ref _pointReads, 0); Volatile.Write(ref _tableScans, 0); } - - public Task PingAsync(CancellationToken ct = default) => inner.PingAsync(ct); - public Task EnsureTableAsync(string table, CancellationToken ct = default) => inner.EnsureTableAsync(table, ct); - public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) => inner.UpsertAsync(table, row, ct); - public Task UpsertBatchAsync(string table, string pk, IReadOnlyList rows, CancellationToken ct = default) - => inner.UpsertBatchAsync(table, pk, rows, ct); - public Task TryReplaceBatchAsync(string table, string pk, IReadOnlyList rows, CancellationToken ct = default) - => inner.TryReplaceBatchAsync(table, pk, rows, ct); - public Task DeleteAsync(string table, string pk, string rk, CancellationToken ct = default) => inner.DeleteAsync(table, pk, rk, ct); - public Task DeletePartitionAsync(string table, string pk, CancellationToken ct = default) => inner.DeletePartitionAsync(table, pk, ct); - - public Task GetAsync(string table, string pk, string rk, CancellationToken ct = default) - { - Interlocked.Increment(ref _pointReads); - return inner.GetAsync(table, pk, rk, ct); - } - - public IAsyncEnumerable QueryPartitionAsync(string table, string pk, CancellationToken ct = default) - => inner.QueryPartitionAsync(table, pk, ct); - - public async IAsyncEnumerable QueryTableAsync(string table, - [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) - { - if (table == queueTable) - { - Interlocked.Increment(ref _tableScans); - if (ScanDelay > TimeSpan.Zero) await Task.Delay(ScanDelay, ct); - } - await foreach (var row in inner.QueryTableAsync(table, ct)) yield return row; - } - - public IAsyncEnumerable QueryTableAsync(string table, string? filter, CancellationToken ct = default) - => QueryTableAsync(table, ct); - } - - private sealed class Fixture - { - public required CountingStore Counting { get; init; } - public required JobQueueStore Queue { get; init; } - public required OrchestratorTableStore Store { get; init; } - public required JobQueueStatusReader Reader { get; init; } - } - - private static async Task NewFixtureAsync() - { - var settings = new CraftSettings(); - settings.Worker.BgPoolSize = 2; - var config = new ConfigurationBuilder().AddInMemoryCollection([]).Build(); - var repo = new ScriptRepository(NullLogger.Instance, settings); - var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); - var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); - var jobs = new JobManager(NullLogger.Instance, settings, limiter); - - var counting = new CountingStore(new RunRemainingCounterTests.ConditionalStore(), - $"{settings.Orchestrator.TablePrefix}Queue"); - var queue = new JobQueueStore(NullLogger.Instance, settings, counting); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, counting); - await queue.InitializeAsync(); - await store.InitializeAsync(); - - return new Fixture - { - Counting = counting, - Queue = queue, - Store = store, - Reader = new JobQueueStatusReader(NullLogger.Instance, jobs, queue, store), - }; - } - - private static DateTime At(int minute) => new(2026, 8, 12, 3, minute, 0, DateTimeKind.Utc); - - [Fact] - public async Task SnapshotCostIsOneScan_RegardlessOfHowManyRunsTheBacklogSpans() - { - var f = await NewFixtureAsync(); - - // 40 distinct runs, each with a counter row — the shape that used to cost 40 point reads per - // refresh, and 12,028 on the instance that motivated this. - for (var i = 0; i < 40; i++) - { - await f.Store.InitRemainingAsync($"run-{i}", 2); - await f.Queue.EnqueueBatchAsync($"run-{i}", [("t0", 4), ("t1", 4)], At(0)); - } - - f.Counting.Reset(); - var snap = await f.Reader.GetAsync(TimeSpan.Zero); - - Assert.NotNull(snap); - Assert.Equal(40, snap!.ByRun.Count); - Assert.Equal(1, f.Counting.QueueScans); - - // The point of the change: aggregating the backlog reads nothing per run. - Assert.Equal(0, f.Counting.PointReads); - } - - [Fact] - public async Task ASlowScanDoesNotReturnAnAlreadyExpiredSnapshot() - { - var f = await NewFixtureAsync(); - await f.Queue.EnqueueBatchAsync("run", [("t0", 4)], At(0)); - - // A scan that takes far longer than the age the caller asked for. Stamped on entry, the - // returned snapshot would be born ~400ms old and instantly past a 100ms TTL, so the next poll - // starts another — the loop this guards against. - f.Counting.ScanDelay = TimeSpan.FromMilliseconds(400); - - var snap = await f.Reader.GetAsync(TimeSpan.FromMilliseconds(100)); - - Assert.NotNull(snap); - Assert.True(snap!.AgeSeconds < 0.2, - $"snapshot was born {snap.AgeSeconds:N3}s old — TakenUtc is being stamped before the scan"); - } - - [Fact] - public async Task TheRefreshIntervalBacksOffToWhatTheScanActuallyCosts() - { - var f = await NewFixtureAsync(); - await f.Queue.EnqueueBatchAsync("run", [("t0", 4)], At(0)); - - f.Counting.ScanDelay = TimeSpan.FromMilliseconds(400); - await f.Reader.GetAsync(TimeSpan.FromMilliseconds(50)); // records the cost - - f.Counting.Reset(); - - // Older than the 50ms the caller wants, well inside the 400ms the scan actually costs. Without - // the floor this re-scans; with it the cached snapshot stands. - await Task.Delay(120); - var again = await f.Reader.GetAsync(TimeSpan.FromMilliseconds(50)); - - Assert.NotNull(again); - Assert.Equal(0, f.Counting.QueueScans); - } - - [Fact] - public async Task AFastScanStillRefreshesOnTheOrdinaryTtl() - { - // The backoff must not become a permanent cache on a healthy instance, where the scan is - // milliseconds and the floor should collapse back to the caller's requested age. - var f = await NewFixtureAsync(); - await f.Queue.EnqueueBatchAsync("run", [("t0", 4)], At(0)); - - await f.Reader.GetAsync(TimeSpan.Zero); - f.Counting.Reset(); - - await Task.Delay(60); - await f.Reader.GetAsync(TimeSpan.FromMilliseconds(20)); - - Assert.Equal(1, f.Counting.QueueScans); - } - - [Fact] - public async Task RunSummariesStillSizeRunsFromTheirCounter() - { - // The counter reads moved out of the snapshot and into the one caller that reads them; this is - // the behaviour that must survive the move. - var f = await NewFixtureAsync(); - await f.Store.InitRemainingAsync("run", 10); - await f.Store.DecrementRemainingAsync("run", 1); - await f.Queue.EnqueueBatchAsync("run", - Enumerable.Range(0, 9).Select(i => ($"task-{i}", 4)).ToList(), At(0)); - - f.Counting.Reset(); - var summary = Assert.Single(await f.Reader.GetRunSummariesAsync(), s => s.Name == "run"); - - Assert.Equal(10, summary.Total); - Assert.Equal(1, summary.Completed); - - // Resolved here, and only here: one point read for the one active run. - Assert.Equal(1, f.Counting.PointReads); - } -} diff --git a/tests/Craft.Tests/JobQueueStoreTests.cs b/tests/Craft.Tests/JobQueueStoreTests.cs deleted file mode 100644 index bfe7eaf..0000000 --- a/tests/Craft.Tests/JobQueueStoreTests.cs +++ /dev/null @@ -1,336 +0,0 @@ -using Craft.Configuration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// The durable job queue that lets the dispatch side hold a worker-pool-sized buffer instead of the -/// whole backlog, and makes the row in storage — not the in-memory copy — the thing that actually runs. -/// -/// Two properties carry the design: -/// -/// Ordering. Rows sort by partition key then row key, so a single read yields the highest-priority, -/// oldest-first work across every run. That is what the in-memory PriorityQueue gives today, and it -/// has to survive the move or a P0 audit-log task ends up behind thousands of P4 standards. -/// -/// Exclusivity. A claim is one conditional transaction, so two workers cannot take the same task — -/// the loser is rejected and retries rather than forcing the write. This is the property the -/// 28-hour deadlock's redrive heuristic was standing in for. -/// -public class JobQueueStoreTests -{ - private static readonly TimeSpan Lease = TimeSpan.FromMinutes(20); - - private static (JobQueueStore Queue, RunRemainingCounterTests.ConditionalStore Backing) NewQueue() - { - var backing = new RunRemainingCounterTests.ConditionalStore(); - var queue = new JobQueueStore(NullLogger.Instance, new CraftSettings(), backing); - return (queue, backing); - } - - private static DateTime At(int minute) => new(2026, 8, 9, 2, minute, 0, DateTimeKind.Utc); - - [Fact] - public async Task ClaimsHighestPriorityFirstAcrossRuns() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - - // Queued in the least helpful order: the low-priority bulk first, the urgent one last. - await queue.EnqueueBatchAsync("StandardsApply", - Enumerable.Range(0, 5).Select(i => ($"std-{i}", 4)).ToList(), At(0)); - await queue.EnqueueAsync("AuditLogIngest", "audit-0", 0, At(5)); - - var claimed = await queue.ClaimBatchAsync("worker-a", 2, Lease); - - Assert.Single(claimed); - Assert.Equal("audit-0", claimed[0].TaskId); - Assert.Equal("AuditLogIngest", claimed[0].RunName); - } - - [Fact] - public async Task ReEnqueuingATaskUpsertsOneRowRatherThanDuplicating() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - - // The re-dispatch case (crash recovery, orphan re-drive). The key is the run's start plus the task, - // so the second enqueue UPDATES the first row instead of adding a duplicate — the duplicate that - // used to get claimed and executed a second time. - await queue.EnqueueAsync("run", "task-0", 4, At(1)); - await queue.EnqueueAsync("run", "task-0", 4, At(1)); - - Assert.Single(await queue.GetQueuedTaskIdsAsync("run")); - - Assert.Equal("task-0", Assert.Single(await queue.ClaimBatchAsync("worker-a", 8, Lease)).TaskId); - // Only one row ever existed, so nothing is left to claim a second time. - Assert.Empty(await queue.ClaimBatchAsync("worker-b", 8, Lease)); - } - - /// - /// THE GUARANTEE. Two workers claiming at once must not both get the same task. The batch is one - /// conditional transaction, so the loser comes back empty and retries. - /// - [Fact] - public async Task TwoWorkersNeverClaimTheSameTask() - { - var (queue, backing) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueBatchAsync("run", Enumerable.Range(0, 4).Select(i => ($"task-{i}", 4)).ToList(), At(0)); - - // Land a competing claim in the window between our read and our write. - IReadOnlyList competitor = []; - backing.OnBeforeConditionalWrite = () => - { - backing.OnBeforeConditionalWrite = null; - competitor = queue.ClaimBatchAsync("worker-b", 4, Lease).GetAwaiter().GetResult(); - }; - - var mine = await queue.ClaimBatchAsync("worker-a", 4, Lease); - - Assert.Equal(4, competitor.Count); - Assert.Empty(mine); - - // And nothing is left claimable, rather than the rows being double-owned. - Assert.Empty(await queue.ClaimBatchAsync("worker-c", 4, Lease)); - } - - [Fact] - public async Task ClaimedWorkIsNotHandedOutAgainWhileTheLeaseHolds() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueBatchAsync("run", Enumerable.Range(0, 3).Select(i => ($"task-{i}", 4)).ToList(), At(0)); - - var first = await queue.ClaimBatchAsync("worker-a", 2, Lease); - var second = await queue.ClaimBatchAsync("worker-b", 2, Lease); - - Assert.Equal(2, first.Count); - Assert.Single(second); - Assert.Empty(first.Select(j => j.TaskId).Intersect(second.Select(j => j.TaskId))); - } - - /// - /// A worker that dies holding a claim must give the work back on its own. This is what replaces the - /// age-based re-drive: nothing has to notice the worker is gone. - /// - [Fact] - public async Task ExpiredLeaseMakesWorkClaimableAgain() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueAsync("run", "task-0", 4, At(0)); - - var held = await queue.ClaimBatchAsync("worker-a", 1, TimeSpan.FromMilliseconds(40)); - Assert.Single(held); - - Assert.Empty(await queue.ClaimBatchAsync("worker-b", 1, Lease)); - - await Task.Delay(120); - - var reclaimed = await queue.ClaimBatchAsync("worker-b", 1, Lease); - Assert.Equal("task-0", Assert.Single(reclaimed).TaskId); - } - - [Fact] - public async Task RenewingKeepsTheClaimAndFailsOnceItHasLapsed() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueAsync("run", "task-0", 4, At(0)); - - var held = await queue.ClaimBatchAsync("worker-a", 1, TimeSpan.FromMilliseconds(40)); - - Assert.True(await queue.RenewAsync(held, "worker-a", Lease)); - Assert.Empty(await queue.ClaimBatchAsync("worker-b", 1, Lease)); - - // Someone else now owns it — renewal must report that rather than taking it back. - await queue.RemoveAsync(held[0]); - await queue.EnqueueAsync("run", "task-0", 4, At(0)); - await queue.ClaimBatchAsync("worker-b", 1, Lease); - - Assert.False(await queue.RenewAsync(held, "worker-a", Lease)); - } - - [Fact] - public async Task RemovedWorkIsGoneFromTheQueue() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueBatchAsync("run", [("task-0", 4), ("task-1", 4)], At(0)); - - var claimed = await queue.ClaimBatchAsync("worker-a", 2, Lease); - foreach (var job in claimed) await queue.RemoveAsync(job); - - // Nothing is left, even once the leases would have lapsed. - Assert.Empty(await queue.ClaimBatchAsync("worker-b", 2, TimeSpan.Zero)); - } - - [Fact] - public async Task RemovingARunClearsOnlyThatRun() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueBatchAsync("doomed", [("a", 4), ("b", 4)], At(0)); - await queue.EnqueueAsync("survivor", "c", 4, At(1)); - - await queue.RemoveRunAsync("doomed"); - - var left = await queue.ClaimBatchAsync("worker-a", 10, Lease); - Assert.Equal("survivor", Assert.Single(left).RunName); - } - - [Fact] - public void PriorityBucketsSortNumericallyNotLexically() - { - // "P4" vs "P10" would order 10 before 4 as strings, quietly inverting priority. - Assert.Equal("P00", JobQueueStore.Bucket(0)); - Assert.Equal("P04", JobQueueStore.Bucket(4)); - Assert.Equal("P10", JobQueueStore.Bucket(10)); - Assert.True(string.CompareOrdinal(JobQueueStore.Bucket(4), JobQueueStore.Bucket(10)) < 0); - } - - [Fact] - public void RowKeyIsDeterministicPerRunAndTask() - { - // Schema v2: the key is a function of (run, task) only, so re-dispatching a task upserts its one - // row instead of writing a second, time-prefixed one — the duplicate-execution class. - Assert.Equal(JobQueueStore.BuildRowKey(DateTime.UnixEpoch, "r", "t"), JobQueueStore.BuildRowKey(DateTime.UnixEpoch, "r", "t")); - Assert.NotEqual(JobQueueStore.BuildRowKey(DateTime.UnixEpoch, "r", "t1"), JobQueueStore.BuildRowKey(DateTime.UnixEpoch, "r", "t2")); - Assert.NotEqual(JobQueueStore.BuildRowKey(DateTime.UnixEpoch, "r1", "t"), JobQueueStore.BuildRowKey(DateTime.UnixEpoch, "r2", "t")); - } - - [Fact] - public async Task EmptyQueueClaimsNothingRatherThanBlocking() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - - Assert.Empty(await queue.ClaimBatchAsync("worker-a", 8, Lease)); - } - - // ─── Status/maintenance surface (the table-backed worker-health view) ─── - - [Fact] - public void RowKeyEscapingKeepsDistinctPairsDistinctAndKeysLegal() - { - // The '|' separator and '%' escape are themselves escaped, so "a|b"+"c" and "a"+"b|c" cannot - // collide onto one row. - Assert.NotEqual(JobQueueStore.BuildRowKey(DateTime.UnixEpoch, "a|b", "c"), JobQueueStore.BuildRowKey(DateTime.UnixEpoch, "a", "b|c")); - - // An illegal character in a component is escaped away, keeping the key legal for Azure Table. - var key = JobQueueStore.BuildRowKey(DateTime.UnixEpoch, "run", "Owner/Repo - No tenant"); - Assert.DoesNotContain(key, c => c is '/' or '\\' or '#' or '?' || char.IsControl(c)); - } - - [Fact] - public async Task ClearAllEmptiesTheQueueButLeavesItUsable() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueBatchAsync("run-a", - Enumerable.Range(0, 3).Select(i => ($"task-{i}", 4)).ToList(), At(0)); - await queue.EnqueueAsync("run-b", "solo", 4, At(0)); - - var removed = await queue.ClearAllAsync(); - - Assert.Equal(4, removed); - Assert.Empty(await queue.ListQueuedAsync()); - Assert.Empty(await queue.GetQueuedTaskIdsAsync("run-a")); - - // The schema marker survives, so the queue keeps working — a fresh enqueue lands and is claimable. - await queue.EnqueueAsync("run-c", "again", 4, At(1)); - Assert.Equal("again", Assert.Single(await queue.ClaimBatchAsync("w", 8, Lease)).TaskId); - } - - [Fact] - public async Task ListQueuedReportsIdentityAgeAndClaimState() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueAsync("run", "waiting", 4, At(2)); - await queue.EnqueueAsync("run", "taken", 0, At(1)); - await queue.ClaimBatchAsync("worker-a", 1, Lease); - - var rows = await queue.ListQueuedAsync(); - - Assert.Equal(2, rows.Count); - - var taken = Assert.Single(rows, r => r.TaskId == "taken"); - Assert.True(taken.Claimed); - Assert.Equal("worker-a", taken.Owner); - Assert.Equal(0, taken.Priority); - Assert.Equal(At(1), taken.QueuedUtc); - - var waiting = Assert.Single(rows, r => r.TaskId == "waiting"); - Assert.False(waiting.Claimed); - Assert.Equal(At(2), waiting.QueuedUtc); - } - - [Fact] - public async Task ListQueuedTreatsALapsedLeaseAsUnclaimed() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueAsync("run", "task-0", 4, At(0)); - await queue.ClaimBatchAsync("worker-a", 1, TimeSpan.FromMilliseconds(40)); - - await Task.Delay(120); - - Assert.False(Assert.Single(await queue.ListQueuedAsync()).Claimed); - } - - [Fact] - public async Task RemoveTaskRemovesOnlyThatTasksRows() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueAsync("run", "doomed", 4, At(0)); - await queue.EnqueueAsync("run", "survivor", 4, At(0)); - await queue.EnqueueAsync("other-run", "doomed", 4, At(0)); - - Assert.Equal(1, await queue.RemoveTaskAsync("run", "doomed")); - - var left = await queue.ListQueuedAsync(); - Assert.Equal(2, left.Count); - Assert.DoesNotContain(left, r => r.RunName == "run" && r.TaskId == "doomed"); - } - - [Fact] - public async Task ReprioritizeMovesTheRowKeepingItsPlaceInLine() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueAsync("run", "older", 4, At(0)); - await queue.EnqueueAsync("run", "boosted", 4, At(1)); - await queue.EnqueueAsync("run", "urgent", 0, At(2)); - - Assert.Equal(1, await queue.ReprioritizeTaskAsync("run", "boosted", 0)); - - // The moved row keeps its enqueue time, so within P0 the earlier "boosted" outranks "urgent". - var claimed = await queue.ClaimBatchAsync("worker-a", 2, Lease); - Assert.Equal(["boosted", "urgent"], claimed.Select(j => j.TaskId).ToArray()); - Assert.All(claimed, j => Assert.Equal(0, j.Priority)); - } - - /// - /// A claimed row is already buffered inside some instance — re-adding it unclaimed at a new - /// priority would create a second runnable copy of the task, the exact failure this queue's - /// claim semantics exist to prevent. - /// - [Fact] - public async Task ReprioritizeLeavesClaimedRowsAlone() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueAsync("run", "task-0", 4, At(0)); - await queue.ClaimBatchAsync("worker-a", 1, Lease); - - Assert.Equal(0, await queue.ReprioritizeTaskAsync("run", "task-0", 0)); - - var row = Assert.Single(await queue.ListQueuedAsync()); - Assert.Equal(4, row.Priority); - Assert.True(row.Claimed); - } -} diff --git a/tests/Craft.Tests/MemoryTableStore.cs b/tests/Craft.Tests/MemoryTableStore.cs new file mode 100644 index 0000000..1a918e6 --- /dev/null +++ b/tests/Craft.Tests/MemoryTableStore.cs @@ -0,0 +1,148 @@ +using Craft.Storage; + +namespace Craft.Tests; + +/// +/// In-memory with the backend properties the orchestration relies on: rows come +/// back ordered by PartitionKey then RowKey, every write stamps a new ETag, and +/// is all-or-nothing under one lock. Reads hand out copies, so mutating a row +/// read from here never writes it. +/// +internal sealed class MemoryTableStore : ICraftTableStore +{ + private readonly object _lock = new(); + private readonly Dictionary> _tables = new(); + private long _etag; + + /// Awaited before a transaction is checked — lets a test interleave a competing write. + public Func? BeforeSubmit { get; set; } + + public int Submits { get; private set; } + + private static readonly Comparer<(string, string)> Order = Comparer<(string, string)>.Create((a, b) => + { + var c = string.CompareOrdinal(a.Item1, b.Item1); + return c != 0 ? c : string.CompareOrdinal(a.Item2, b.Item2); + }); + + private SortedDictionary<(string, string), StoreRow> Table(string t) + { + if (!_tables.TryGetValue(t, out var rows)) _tables[t] = rows = new(Order); + return rows; + } + + private StoreRow Stamp(StoreRow r) => new(r.PartitionKey, r.RowKey) + { + ETag = $"W/\"{++_etag}\"", + Timestamp = DateTimeOffset.UtcNow, + Properties = new Dictionary(r.Properties), + }; + + private static StoreRow Copy(StoreRow r) => new(r.PartitionKey, r.RowKey) + { + ETag = r.ETag, + Timestamp = r.Timestamp, + Properties = new Dictionary(r.Properties), + }; + + public IReadOnlyList All(string table) + { + lock (_lock) return Table(table).Values.Select(Copy).ToList(); + } + + public Task PingAsync(CancellationToken ct = default) => Task.CompletedTask; + + public Task EnsureTableAsync(string table, CancellationToken ct = default) + { + lock (_lock) Table(table); + return Task.CompletedTask; + } + + public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) + { + lock (_lock) Table(table)[(row.PartitionKey, row.RowKey)] = Stamp(row); + return Task.CompletedTask; + } + + public Task UpsertBatchAsync(string table, string partitionKey, IReadOnlyList rows, CancellationToken ct = default) + { + lock (_lock) foreach (var r in rows) Table(table)[(r.PartitionKey, r.RowKey)] = Stamp(r); + return Task.CompletedTask; + } + + public Task TryReplaceBatchAsync(string table, string partitionKey, IReadOnlyList rows, CancellationToken ct = default) => + TrySubmitAsync(table, partitionKey, rows.Select(StoreOp.Replace).ToList(), ct); + + public async Task TrySubmitAsync(string table, string partitionKey, IReadOnlyList ops, CancellationToken ct = default) + { + if (BeforeSubmit is { } before) await before(); + return Submit(table, ops); + } + + private bool Submit(string table, IReadOnlyList ops) + { + lock (_lock) + { + var t = Table(table); + foreach (var op in ops) + { + t.TryGetValue((op.Row.PartitionKey, op.Row.RowKey), out var cur); + var ok = op.Kind switch + { + StoreOpKind.Insert => cur == null, + StoreOpKind.Replace => cur != null && cur.ETag == op.Row.ETag, + StoreOpKind.Delete => op.Row.ETag == null || (cur != null && cur.ETag == op.Row.ETag), + _ => true, + }; + if (!ok) return false; + } + foreach (var op in ops) + { + var key = (op.Row.PartitionKey, op.Row.RowKey); + if (op.Kind == StoreOpKind.Delete) t.Remove(key); + else t[key] = Stamp(op.Row); + } + Submits++; + return true; + } + } + + public Task GetAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) + { + lock (_lock) + return Task.FromResult(Table(table).TryGetValue((partitionKey, rowKey), out var r) ? Copy(r) : null); + } + + private List Snapshot(string table, Func where) + { + lock (_lock) return Table(table).Values.Where(where).Select(Copy).ToList(); + } + + public async IAsyncEnumerable QueryPartitionAsync(string table, string partitionKey, + [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) + { + foreach (var r in Snapshot(table, r => r.PartitionKey == partitionKey)) { yield return r; await Task.Yield(); } + } + + public async IAsyncEnumerable QueryTableAsync(string table, + [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) + { + foreach (var r in Snapshot(table, _ => true)) { yield return r; await Task.Yield(); } + } + + public Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) + { + lock (_lock) Table(table).Remove((partitionKey, rowKey)); + return Task.CompletedTask; + } + + public Task DeletePartitionAsync(string table, string partitionKey, CancellationToken ct = default) + { + lock (_lock) + { + var t = Table(table); + foreach (var k in t.Keys.Where(k => k.Item1 == partitionKey).ToList()) t.Remove(k); + } + return Task.CompletedTask; + } +} diff --git a/tests/Craft.Tests/OrchestrationContractTests.cs b/tests/Craft.Tests/OrchestrationContractTests.cs new file mode 100644 index 0000000..749894a --- /dev/null +++ b/tests/Craft.Tests/OrchestrationContractTests.cs @@ -0,0 +1,298 @@ +using System.Collections.Concurrent; +using System.Text.Json; +using Craft.Configuration; +using Craft.Orchestration; +using Craft.PowerShellHost; +using Craft.Storage; +using Microsoft.Extensions.Configuration; +using Microsoft.Extensions.Logging.Abstractions; + +namespace Craft.Tests; + +/// +/// The orchestration contract as CIPP sees it, end to end through the real store, pump and JobManager with +/// only the PowerShell calls faked: a batch becomes tasks invoked with TaskJson; their output reaches +/// the PostExecution script as one JSON line each, with FunctionName and ParametersJson; a run +/// queued by a task holds its parent until it finishes; a name still running is not started twice. +/// +public class OrchestrationContractTests +{ + private const string TaskFunc = "Invoke-CraftTask"; + private const string PostExecFunc = "Invoke-CraftPostExecution"; + + private sealed class Svc(JobManager jobs, WorkStore store, ResultStore results, IConfiguration config, CraftSettings settings) + : OrchestratorService(NullLogger.Instance, null!, null!, jobs, store, results, config, settings) + { + public readonly ConcurrentQueue> Tasks = new(); + public readonly ConcurrentQueue<(Dictionary Parameters, string[] Lines)> PostExecs = new(); + public Func, string>? Body; + public Func? PostExecBody; + public int Checkouts, Reclaims; + + internal override string? FindScript(string name) => name; + + internal override async Task RunScriptAsync(string path, Dictionary parameters, bool captureOutput, + PowerShellWorker? worker = null) + { + if (path == PostExecFunc) + { + PostExecs.Enqueue((parameters, File.ReadAllLines((string)parameters["ResultsPath"]))); + if (PostExecBody != null) await PostExecBody(); + return string.Empty; + } + var task = JsonSerializer.Deserialize>((string)parameters["TaskJson"])!; + Tasks.Enqueue(task); + var output = Body?.Invoke(task) ?? JsonSerializer.Serialize(new { tenant = task["TenantFilter"].ToString() }); + return captureOutput ? output : string.Empty; + } + + internal override PowerShellWorker? CheckoutSequentialWorker(CancellationToken ct) { Checkouts++; return null; } + internal override void ReclaimSequentialWorker(PowerShellWorker? worker, bool faulted) => Reclaims++; + } + + private sealed record Harness(Svc Svc, WorkStore Store, WorkPump Pump, JobManager Jobs) : IAsyncDisposable + { + public async Task DriveUntil(Func> done, int timeoutMs = 10_000) + { + var deadline = Environment.TickCount64 + timeoutMs; + while (Environment.TickCount64 < deadline) + { + await Pump.RefillAsync(CancellationToken.None); + if (await done()) return true; + await Task.Delay(10); + } + return await done(); + } + + public Task DriveUntilFinished(string name, int timeoutMs = 10_000) => + DriveUntil(async () => await Store.GetRunByNameAsync(name) is { IsFinished: true }, timeoutMs); + + public async ValueTask DisposeAsync() => await Jobs.StopAsync(CancellationToken.None); + } + + private static async Task NewAsync() + { + var settings = new CraftSettings(); + settings.Worker.BgPoolSize = 4; + var config = new ConfigurationBuilder().AddInMemoryCollection([]).Build(); + var repo = new ScriptRepository(NullLogger.Instance, settings); + var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); + var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); + var jobs = new JobManager(NullLogger.Instance, settings, limiter); + var mem = new MemoryTableStore(); + var store = new WorkStore(NullLogger.Instance, settings, mem); + var results = new ResultStore(NullLogger.Instance, settings, mem); + var svc = new Svc(jobs, store, results, config, settings); + await svc.ResumeInterruptedRunsAsync(CancellationToken.None); + var pump = new WorkPump(NullLogger.Instance, store, jobs, config, settings, svc); + _ = Task.Run(() => jobs.StartAsync(CancellationToken.None)); + return new Harness(svc, store, pump, jobs); + } + + private static string Batch(int n, string prefix = "t") => + JsonSerializer.Serialize(Enumerable.Range(0, n).Select(i => new { Name = "Job", TenantFilter = $"{prefix}{i}", N = i })); + + private static Task Start(Harness h, string name, string batch, string? postExec = null, string? postParams = null, + bool sequential = false, int priority = 4) => + h.Svc.StartFromBatchAsync(name, batch, priority, postExec, postParams, CancellationToken.None, sequential: sequential); + + [Fact] + public async Task EveryTaskRunsOnce_WithItsBatchItemAsTaskJson_AndTheRunCompletes() + { + await using var h = await NewAsync(); + Assert.True(await Start(h, "FanOut", Batch(10))); + + Assert.True(await h.DriveUntilFinished("FanOut")); + var run = (await h.Store.GetRunByNameAsync("FanOut"))!; + Assert.Equal("Completed", run.Status); + Assert.Equal(10, run.Done); + Assert.Equal(Enumerable.Range(0, 10).Select(i => $"t{i}").Order(), + h.Svc.Tasks.Select(t => t["TenantFilter"].ToString()!).Order()); + Assert.Empty(await ReadyNames(h)); + } + + [Fact] + public async Task PostExecution_RunsOnceAfterEveryTask_WithOneLinePerTaskOutput_AndItsParameters() + { + await using var h = await NewAsync(); + var bigParameters = JsonSerializer.Serialize(new { blob = new string('x', 100_000) }); + Assert.True(await Start(h, "WithPost", Batch(5), "AuditLogs", bigParameters)); + + Assert.True(await h.DriveUntilFinished("WithPost")); + var post = Assert.Single(h.Svc.PostExecs); + Assert.Equal("AuditLogs", post.Parameters["FunctionName"]); + Assert.Equal(bigParameters, post.Parameters["ParametersJson"]); + Assert.Equal(Enumerable.Range(0, 5).Select(i => $"t{i}").Order(), + post.Lines.Select(l => JsonDocument.Parse(l).RootElement.GetProperty("tenant").GetString()!).Order()); + Assert.Equal(5, h.Svc.Tasks.Count); + Assert.Equal("Completed", (await h.Store.GetRunByNameAsync("WithPost"))!.PostExecStatus); + } + + [Fact] + public async Task AFailingTask_IsRecorded_AndTheRunStillAggregatesAndFinishesWithErrors() + { + await using var h = await NewAsync(); + h.Svc.Body = t => t["TenantFilter"].ToString() == "t1" ? throw new InvalidOperationException("boom") : "{}"; + Assert.True(await Start(h, "Partial", Batch(3), "Agg")); + + Assert.True(await h.DriveUntilFinished("Partial")); + var run = (await h.Store.GetRunByNameAsync("Partial"))!; + Assert.Equal("CompletedWithErrors", run.Status); + Assert.Equal(1, run.Failed); + Assert.Single(h.Svc.PostExecs); + var failed = Assert.Single(await h.Store.GetTasksAsync(run.RunKey, 'D'), t => t.Status == "Failed"); + Assert.Equal("boom", failed.LastError); + } + + [Fact] + public async Task APostExecutionThatKeepsFailing_IsRetried_ThenTheRunFinishes() + { + await using var h = await NewAsync(); + h.Svc.PostExecBody = () => throw new InvalidOperationException("aggregate down"); + Assert.True(await Start(h, "PostFails", Batch(2), "Agg")); + + // A failed aggregation is handed back to pending, and the pump skips a run whose counts have not + // moved for a while, so the retries are spaced out; drive with that backoff out of the way. + Assert.True(await h.DriveUntil(async () => + { + h.Pump.ForgetBackoff(); + return await h.Store.GetRunByNameAsync("PostFails") is { IsFinished: true }; + })); + Assert.Equal(new CraftSettings().Orchestrator.MaxRetries, h.Svc.PostExecs.Count); + Assert.Equal("Failed", (await h.Store.GetRunByNameAsync("PostFails"))!.PostExecStatus); + } + + [Fact] + public async Task ARunNameStillGoing_IsNotStartedAgain_AndItsBatchFileIsStillDeleted() + { + await using var h = await NewAsync(); + Assert.True(await Start(h, "Recurring", Batch(2))); + + var file = Path.Combine(Path.GetTempPath(), $"craft-test-{Guid.NewGuid():N}.jsonl"); + await File.WriteAllTextAsync(file, "{\"Name\":\"Job\",\"TenantFilter\":\"x\"}\n"); + Assert.False(await h.Svc.StartFromBatchAsync("Recurring", "", 4, null, null, CancellationToken.None, batchFilePath: file)); + Assert.False(File.Exists(file)); + Assert.Single(await ReadyNames(h)); + + Assert.True(await h.DriveUntilFinished("Recurring")); + Assert.True(await Start(h, "Recurring", Batch(1))); + } + + [Fact] + public async Task AChildQueuedByATask_HoldsTheParentsPostExecution_UntilTheChildFinishes() + { + await using var h = await NewAsync(); + var childGate = new TaskCompletionSource(); + h.Svc.Body = t => + { + if (t["TenantFilter"].ToString() == "p0") + { + var link = h.Svc.RegisterPendingChild("Parent", "Child"); + Assert.NotNull(link); + h.Svc.StartFromBatchAsync("Child", Batch(2, "c"), 4, null, null, CancellationToken.None, + parentRunKey: link!.Value.ParentRunKey, childKey: link.Value.ChildKey).GetAwaiter().GetResult(); + } + if (t["TenantFilter"].ToString()!.StartsWith('c')) childGate.Task.GetAwaiter().GetResult(); + return "{}"; + }; + Assert.True(await Start(h, "Parent", Batch(1, "p"), "Agg")); + + Assert.True(await h.DriveUntil(async () => await h.Store.GetRunByNameAsync("Child") != null && h.Svc.Tasks.Count >= 2)); + await h.DriveUntil(() => Task.FromResult(false), 300); + Assert.Empty(h.Svc.PostExecs); + Assert.False((await h.Store.GetRunByNameAsync("Parent"))!.IsFinished); + + childGate.SetResult(); + Assert.True(await h.DriveUntilFinished("Parent")); + Assert.True((await h.Store.GetRunByNameAsync("Child"))!.IsFinished); + Assert.Single(h.Svc.PostExecs); + } + + [Fact] + public async Task AChildThatNeverStarts_ReleasesItsParent() + { + await using var h = await NewAsync(); + h.Svc.Body = _ => + { + var link = h.Svc.RegisterPendingChild("Parent", "Stillborn")!.Value; + h.Svc.AbandonPendingChildAsync(link.ParentRunKey, link.ChildKey).GetAwaiter().GetResult(); + return "{}"; + }; + Assert.True(await Start(h, "Parent", Batch(1), "Agg")); + + Assert.True(await h.DriveUntilFinished("Parent")); + Assert.Single(h.Svc.PostExecs); + } + + [Fact] + public async Task ARunCannotBeItsOwnChild_NorAChildOfAFinishedRun() + { + await using var h = await NewAsync(); + Assert.True(await Start(h, "Solo", Batch(1))); + Assert.Null(h.Svc.RegisterPendingChild("Solo", "Solo")); + Assert.True(await h.DriveUntilFinished("Solo")); + Assert.Null(h.Svc.RegisterPendingChild("Solo", "FollowUp")); + } + + [Fact] + public async Task CancellingARun_CancelsWhatIsPending_AndTheRunStillFinalizes() + { + await using var h = await NewAsync(); + Assert.True(await Start(h, "Doomed", Batch(20))); + + var (found, cancelled) = await h.Svc.CancelRunAsync("Doomed"); + Assert.True(found); + Assert.Equal(20, cancelled); + var run = (await h.Store.GetRunByNameAsync("Doomed"))!; + Assert.True(run.IsFinished); + Assert.Equal(20, run.Cancelled); + Assert.Empty(h.Svc.Tasks); + } + + [Fact] + public async Task ASequentialRun_RunsEveryStepInOrder_OnOneWorker_PastAFailingStep() + { + await using var h = await NewAsync(); + h.Svc.Body = t => t["TenantFilter"].ToString() == "s2" ? throw new InvalidOperationException("step down") : "{}"; + Assert.True(await Start(h, "Steps", Batch(5, "s"), sequential: true)); + + Assert.True(await h.DriveUntilFinished("Steps")); + Assert.Equal(["s0", "s1", "s2", "s3", "s4"], h.Svc.Tasks.Select(t => t["TenantFilter"].ToString())); + Assert.Equal(1, h.Svc.Checkouts); + Assert.Equal(1, h.Svc.Reclaims); + Assert.Equal(1, (await h.Store.GetRunByNameAsync("Steps"))!.Failed); + } + + [Fact] + public async Task WorkClaimedByAProcessThatDied_IsTakenBackOnceItsLeaseLapses() + { + await using var h = await NewAsync(); + Assert.True(await Start(h, "Orphaned", Batch(3))); + var run = (await h.Store.GetRunByNameAsync("Orphaned"))!; + Assert.Equal(3, (await h.Store.ClaimAsync(run.RunKey, 3, "dead-host", TimeSpan.Zero, false)).Count); + + Assert.True(await h.DriveUntilFinished("Orphaned")); + Assert.Equal(3, h.Svc.Tasks.Count); + Assert.All(await h.Store.GetTasksAsync(run.RunKey, 'D'), t => Assert.Equal(2, t.Attempt)); + } + + [Fact] + public async Task ALowerBandRunsFirst_AndWithinABandTheOlderRun_NotTheAlphabeticallyFirst() + { + await using var h = await NewAsync(); + Assert.True(await Start(h, "Zulu", Batch(1, "z"), priority: 4)); + await Task.Delay(5); + Assert.True(await Start(h, "Alpha", Batch(1, "a"), priority: 4)); + Assert.True(await Start(h, "Background", Batch(1, "b"), priority: 9)); + Assert.True(await Start(h, "Urgent", Batch(1, "u"), priority: 1)); + + Assert.Equal(["Urgent", "Zulu", "Alpha", "Background"], await ReadyNames(h)); + } + + private static async Task> ReadyNames(Harness h) + { + var names = new List(); + await foreach (var e in h.Store.ReadReadyAsync()) names.Add(e.Name); + return names; + } +} diff --git a/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs b/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs index 4b56760..daecbc9 100644 --- a/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs +++ b/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs @@ -2,11 +2,11 @@ using System.Management.Automation; using System.Management.Automation.Runspaces; using System.Reflection; -using System.Runtime.CompilerServices; using Craft.Hosting; using Craft.Orchestration; using Craft.PowerShellHost; using Craft.Services; +using Microsoft.Extensions.Configuration; using Microsoft.Extensions.Logging.Abstractions; namespace Craft.Tests; @@ -173,23 +173,28 @@ public void SelfParent_IsDroppedAtEnqueue() } [Fact] - public void Drain_ReleasesTheGate_WhenStartFails() + public async Task Drain_ReleasesTheParent_WhenTheChildIsNeverCreated() { - // The deadlock-avoidance guarantee: a child whose start attempt throws must stop gating - // its parent. The service here is deliberately missing its storage fields, so - // StartFromBatchAsync fails immediately — the finally in DrainPending must still release. - var svc = (OrchestratorService)RuntimeHelpers.GetUninitializedObject(typeof(OrchestratorService)); - void Set(string field, object value) => - typeof(OrchestratorService).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)! - .SetValue(svc, value); - Set("_logger", NullLogger.Instance); - var activeRuns = new ConcurrentDictionary(); - Set("_activeRuns", activeRuns); - Set("_childRuns", new ConcurrentDictionary>()); - Set("_recoveringChildren", new ConcurrentDictionary()); - var pendingChildRuns = new ConcurrentDictionary(); - Set("_pendingChildRuns", pendingChildRuns); - activeRuns.TryAdd("LineageDrainParent", new OrchestratorRun { Name = "LineageDrainParent", Status = "Running" }); + // The deadlock-avoidance guarantee: a child registered at enqueue that then fails to start (here an + // empty batch) must stop holding its parent, or the parent never reaches its barrier. + var settings = new Craft.Configuration.CraftSettings(); + var config = new ConfigurationBuilder().AddInMemoryCollection([]).Build(); + var repo = new ScriptRepository(NullLogger.Instance, settings); + var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); + var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); + var jobs = new Craft.Orchestration.JobManager(NullLogger.Instance, settings, limiter); + var mem = new MemoryTableStore(); + var store = new Craft.Storage.WorkStore(NullLogger.Instance, settings, mem); + var svc = new OrchestratorService(NullLogger.Instance, null!, limiter, jobs, store, + new Craft.Storage.ResultStore(NullLogger.Instance, settings, mem), config, settings); + var started = DateTime.UtcNow; + var parent = await store.CreateRunAsync(new Craft.Storage.RunHeader + { + RunKey = Craft.Storage.WorkStore.RunKeyFor("LineageDrainParent", started), + Name = "LineageDrainParent", + StartedUtc = started, + TaskScriptName = "Invoke-CraftTask", + }, [new Craft.Storage.WorkStore.NewTask("t0", new())]); var previousService = s_serviceField.GetValue(null); try @@ -198,17 +203,19 @@ void Set(string field, object value) => OrchestratorBridge.QueueOrchestration("LineageDrainChild", "[]", 4, null, null, null, parentRunName: "LineageDrainParent"); - Assert.True(pendingChildRuns.ContainsKey("LineageDrainChild")); + Assert.Equal(2, (await store.GetRunAsync(parent.RunKey))!.Total); OrchestratorBridge.DrainPending(); - Assert.False(pendingChildRuns.ContainsKey("LineageDrainChild")); - Assert.False(activeRuns.ContainsKey("LineageDrainChild")); + var after = (await store.GetRunAsync(parent.RunKey))!; + Assert.Equal(1, after.Done); + Assert.Equal(2, after.Total); + Assert.Null(await store.GetRunByNameAsync("LineageDrainChild")); } finally { // The bridge service is static process state — put back whatever was there so this - // test cannot redirect other tests' drains into the crippled service. + // test cannot redirect other tests' drains into this one. s_serviceField.SetValue(null, previousService); } } diff --git a/tests/Craft.Tests/OrchestratorCancelRunTests.cs b/tests/Craft.Tests/OrchestratorCancelRunTests.cs deleted file mode 100644 index 0a3ad84..0000000 --- a/tests/Craft.Tests/OrchestratorCancelRunTests.cs +++ /dev/null @@ -1,78 +0,0 @@ -using System.Collections.Concurrent; -using System.Reflection; -using Craft.Configuration; -using Craft.Orchestration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// Cancelling a run this node holds must cancel the live tasks, not a copy loaded from storage: the live -/// ones are what the re-drive and completion check read, so a cancelled copy left them Pending and the -/// re-drive kept putting them back on the queue. -/// -public class OrchestratorCancelRunTests -{ - [Fact] - public async Task CancellingALiveRun_CancelsItsLiveTasks_AndDropsItsQueueRows() - { - var settings = new CraftSettings { Orchestrator = { TablePrefix = "cnl" + Guid.NewGuid().ToString("N")[..8] } }; - settings.Orchestrator.BatchStatusWrites = false; - var backing = new RunRemainingCounterTests.ConditionalStore(); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); - var queue = new JobQueueStore(NullLogger.Instance, settings, backing); - await store.InitializeAsync(); - await queue.InitializeAsync(); - var svc = NewService(settings, store, queue); - - var run = new OrchestratorRun - { - Name = "Mailbox_t1", - Status = "Running", - Priority = 4, - StartedUtc = DateTime.UtcNow, - Tasks = [Pending("a"), Pending("b"), Pending("c")] - }; - await store.UpsertRunAsync(run); - await store.UpsertTaskBatchAsync(run.Name, run.Tasks); - await store.InitRemainingAsync(run.Name, run.Tasks.Count); - await queue.EnqueueBatchAsync(run.Name, run.Tasks.Select(t => (t.Id, 4)).ToList(), run.StartedUtc); - run.Tasks[0].Status = "Running"; - await store.UpsertTaskAsync(run.Name, run.Tasks[0]); - Get>(svc, "_activeRuns")[run.Name] = run; - - var (found, cancelled) = await svc.CancelRunAsync(run.Name); - - Assert.True(found); - Assert.Equal(2, cancelled); - Assert.Equal(["Running", "Cancelled", "Cancelled"], run.Tasks.Select(t => t.Status)); - Assert.Empty(await queue.GetQueuedTaskIdsAsync(run.Name)); - } - - private static OrchestratorTaskItem Pending(string id) => new() { Id = id, Status = "Pending" }; - - private static OrchestratorService NewService(CraftSettings settings, OrchestratorTableStore store, JobQueueStore queue) - { - var svc = (OrchestratorService)System.Runtime.CompilerServices.RuntimeHelpers - .GetUninitializedObject(typeof(OrchestratorService)); - Set(svc, "_logger", NullLogger.Instance); - Set(svc, "_store", store); - Set(svc, "_queue", queue); - Set(svc, "_writer", new OrchestratorStatusWriter(store, NullLogger.Instance, settings)); - Set(svc, "_settings", settings); - Set(svc, "_lock", new object()); - Set(svc, "_activeRuns", new ConcurrentDictionary()); - Set(svc, "_cancelledRuns", new ConcurrentDictionary()); - Set(svc, "_finalizingRuns", new ConcurrentDictionary()); - return svc; - } - - private static void Set(object target, string field, object? value) => - typeof(OrchestratorService).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)! - .SetValue(target, value); - - private static T Get(object target, string field) => - (T)typeof(OrchestratorService).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)! - .GetValue(target)!; -} diff --git a/tests/Craft.Tests/OrchestratorChildRunGuardTests.cs b/tests/Craft.Tests/OrchestratorChildRunGuardTests.cs deleted file mode 100644 index f4a7343..0000000 --- a/tests/Craft.Tests/OrchestratorChildRunGuardTests.cs +++ /dev/null @@ -1,147 +0,0 @@ -using System.Collections.Concurrent; -using System.Reflection; -using System.Runtime.CompilerServices; -using Craft.Orchestration; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// A run must never wait on itself to finalize — and must genuinely wait on its children. -/// -/// The bridge registers the queued run under the ambient-or-explicit parent, so a run re-queued -/// from inside its own context (the recurring-run pattern) used to arrive as its own child. The -/// child-run guard then blocked finalization until the "child" left _activeRuns — which only happens -/// at finalization. Observed live as seven runs stuck "Running" for days, every task terminal, -/// Remaining=0, and their PostExecutions (audit log processing) never dispatched. -/// -/// The guard has to stay specific: a REAL child (different name) must block its parent from the -/// moment it is REGISTERED — which happens at enqueue time, while the child exists nowhere but the -/// bridge queue. Registration takes a pending gate that only lifts, once the start attempt has either put -/// the child into _activeRuns (which takes over the blocking) or failed (so the parent must not -/// wait forever). All directions are asserted here. -/// -public class OrchestratorChildRunGuardTests -{ - /// An OrchestratorService with only the fields the child-run guard touches. - private static OrchestratorService NewService() - { - var svc = (OrchestratorService)RuntimeHelpers.GetUninitializedObject(typeof(OrchestratorService)); - Set(svc, "_logger", NullLogger.Instance); - Set(svc, "_activeRuns", new ConcurrentDictionary()); - Set(svc, "_childRuns", new ConcurrentDictionary>()); - Set(svc, "_recoveringChildren", new ConcurrentDictionary()); - Set(svc, "_pendingChildRuns", new ConcurrentDictionary()); - return svc; - } - - private static void Set(object target, string field, object value) => - typeof(OrchestratorService).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)! - .SetValue(target, value); - - private static T Get(object target, string field) => - (T)typeof(OrchestratorService).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)! - .GetValue(target)!; - - private static void Activate(OrchestratorService svc, string name) => - Get>(svc, "_activeRuns") - .TryAdd(name, new OrchestratorRun { Name = name, Status = "Running" }); - - private static bool AllChildRunsComplete(OrchestratorService svc, string runName) => - (bool)typeof(OrchestratorService).GetMethod("AllChildRunsComplete", - BindingFlags.NonPublic | BindingFlags.Instance)!.Invoke(svc, [runName])!; - - [Fact] - public void SelfRegistration_IsRefused() - { - var svc = NewService(); - Activate(svc, "StandardsApply"); - - Assert.False(svc.TryRegisterPendingChildRun("StandardsApply", "StandardsApply")); - - Assert.False(Get>>(svc, "_childRuns") - .ContainsKey("StandardsApply")); - Assert.True(AllChildRunsComplete(svc, "StandardsApply")); - } - - [Fact] - public void InactiveParent_IsRefused() - { - // Runs queued from PostExecution land here: the spawning run has already finalized, so the - // new run keeps its lineage on the row but must not gate anything. - var svc = NewService(); - - Assert.False(svc.TryRegisterPendingChildRun("FinalizedParent", "FollowUpRun")); - Assert.True(AllChildRunsComplete(svc, "FinalizedParent")); - } - - [Fact] - public void PendingChild_BlocksParent_ThroughItsWholeLifecycle() - { - var svc = NewService(); - Activate(svc, "Parent"); - - // Registered at enqueue time — the child exists nowhere but the bridge queue, and the - // parent must already be blocked, because its own last task is what queued the child. - Assert.True(svc.TryRegisterPendingChildRun("Parent", "Child")); - Assert.False(AllChildRunsComplete(svc, "Parent")); - - // The drain starts the child (it enters the live graph) and then lifts the gate: still - // blocked, the live graph has taken over. - Activate(svc, "Child"); - svc.ReleasePendingChildRun("Child"); - Assert.False(AllChildRunsComplete(svc, "Parent")); - - // Child finalizes and is evicted from the live graph — the parent unblocks. - Get>(svc, "_activeRuns").TryRemove("Child", out _); - Assert.True(AllChildRunsComplete(svc, "Parent")); - } - - [Fact] - public void FailedStart_ReleasesTheGate() - { - // A child that never starts (0 tasks, missing task function, storage failure) must stop - // blocking once the start attempt is over — a leaked gate would defer the parent's - // finalize for the process lifetime. - var svc = NewService(); - Activate(svc, "Parent"); - - Assert.True(svc.TryRegisterPendingChildRun("Parent", "StillbornChild")); - Assert.False(AllChildRunsComplete(svc, "Parent")); - - svc.ReleasePendingChildRun("StillbornChild"); - Assert.True(AllChildRunsComplete(svc, "Parent")); - } - - [Fact] - public void DoubleQueuedChild_NeedsBothReleases() - { - // Two queue entries under the same child name (e.g. two parent tasks each queueing the - // same fan-out) hold independent gates: the first release must not lift the second's. - var svc = NewService(); - Activate(svc, "Parent"); - - Assert.True(svc.TryRegisterPendingChildRun("Parent", "SharedChild")); - Assert.True(svc.TryRegisterPendingChildRun("Parent", "SharedChild")); - - svc.ReleasePendingChildRun("SharedChild"); - Assert.False(AllChildRunsComplete(svc, "Parent")); - - svc.ReleasePendingChildRun("SharedChild"); - Assert.True(AllChildRunsComplete(svc, "Parent")); - } - - [Fact] - public void StaleSelfLink_DoesNotBlockFinalize() - { - // A self-link registered before the guard existed (or rebuilt from a pre-fix storage row) - // can still be sitting in the bag. It must not count as an outstanding child. - var svc = NewService(); - Activate(svc, "AuditLogSearchCreationV2"); - Get>>(svc, "_childRuns") - .TryAdd("AuditLogSearchCreationV2", ["AuditLogSearchCreationV2"]); - - Assert.True(AllChildRunsComplete(svc, "AuditLogSearchCreationV2")); - } -} diff --git a/tests/Craft.Tests/OrchestratorFinalizedRunTests.cs b/tests/Craft.Tests/OrchestratorFinalizedRunTests.cs deleted file mode 100644 index f519476..0000000 --- a/tests/Craft.Tests/OrchestratorFinalizedRunTests.cs +++ /dev/null @@ -1,346 +0,0 @@ -using System.Collections.Concurrent; -using System.Reflection; -using System.Runtime.CompilerServices; -using Craft.Configuration; -using Craft.Orchestration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// A finalized run must stay finalized. -/// -/// It did not. A run's queue rows were only dropped at finalize when it had NO post-execution, so a run -/// with one kept its rows from finalize until the post-execution succeeded. The pump re-claimed them in -/// that window, and the resolver — finding the run absent from _activeRuns because it had FINISHED — -/// rehydrated it from storage and put it back in the live graph. From there the next completion -/// re-finalized it and dispatched the aggregation again. -/// -/// Measured on a live 16-tenant instance before the fix, for one 13-task run: -/// 7x "Run MailboxRules_7ngn50... finalized: Completed (13/0/0/13)" -/// 7x "Dispatching PostExecution Push-StoreMailboxRules" -/// and individual tasks re-executed up to 4 times each, because a coalesced terminal write had not -/// landed yet and the rehydrated task still read Pending. -/// -/// Idempotent Push-* consumers absorbed this silently. Push-ScheduledTaskPostExecution does not — it -/// advances a recurring task by ScheduledTime + recurrence and writes it back, so every extra -/// invocation pushes the next run out by another interval. -/// -/// This pins the resolver half: a descriptor belonging to an already-finished run is dropped rather -/// than rehydrated. The guard has to be specific — an unfinished run absent from memory must still be -/// rehydrated, which is what crash recovery depends on — so both directions are asserted. -/// -public class OrchestratorFinalizedRunTests -{ - private sealed class FakeStore : ICraftTableStore - { - private readonly Dictionary> _tables = new(); - - public Task TryReplaceBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) => throw new NotSupportedException(); - - public Task PingAsync(CancellationToken ct = default) => Task.CompletedTask; - - public Task EnsureTableAsync(string table, CancellationToken ct = default) - { - if (!_tables.ContainsKey(table)) _tables[table] = new(); - return Task.CompletedTask; - } - - public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) - { - _tables[table][(row.PartitionKey, row.RowKey)] = row; - return Task.CompletedTask; - } - - public Task UpsertBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) - { - foreach (var r in rows) _tables[table][(r.PartitionKey, r.RowKey)] = r; - return Task.CompletedTask; - } - - public Task GetAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) - => Task.FromResult(_tables[table].TryGetValue((partitionKey, rowKey), out var r) ? r : null); - - public async IAsyncEnumerable QueryPartitionAsync(string table, string partitionKey, - [EnumeratorCancellation] CancellationToken ct = default) - { - foreach (var kv in _tables[table].Where(k => k.Key.Item1 == partitionKey).ToList()) - { - yield return kv.Value; - await Task.Yield(); - } - } - - public async IAsyncEnumerable QueryTableAsync(string table, - [EnumeratorCancellation] CancellationToken ct = default) - { - foreach (var kv in _tables[table].ToList()) - { - yield return kv.Value; - await Task.Yield(); - } - } - - public Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) - { - _tables[table].Remove((partitionKey, rowKey)); - return Task.CompletedTask; - } - - public Task DeletePartitionAsync(string table, string partitionKey, CancellationToken ct = default) - { - foreach (var k in _tables[table].Keys.Where(k => k.Item1 == partitionKey).ToList()) - _tables[table].Remove(k); - return Task.CompletedTask; - } - } - - /// - /// An OrchestratorService with only the fields the resolver reads. Building the real one would drag - /// in the PowerShell runner and the job manager for what is, on this path, a storage read plus a - /// status check. - /// - private static (OrchestratorService Svc, OrchestratorTableStore Store) NewService() - { - var settings = new CraftSettings { Orchestrator = { TablePrefix = "fin" + Guid.NewGuid().ToString("N")[..6] } }; - var store = new OrchestratorTableStore(NullLogger.Instance, settings, new FakeStore()); - - var svc = (OrchestratorService)RuntimeHelpers.GetUninitializedObject(typeof(OrchestratorService)); - Set(svc, "_logger", NullLogger.Instance); - Set(svc, "_store", store); - Set(svc, "_activeRuns", new ConcurrentDictionary()); - Set(svc, "_taskScriptPaths", new ConcurrentDictionary()); - // Field initializers do not run on an uninitialized object; the resolver locks this to read - // task status once it gets past the finished-run guard. - Set(svc, "_lock", new object()); - return (svc, store); - } - - private static void Set(object target, string field, object value) => - typeof(OrchestratorService).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)! - .SetValue(target, value); - - private static async Task ResolveAsync(OrchestratorService svc, string runName, string taskId) - { - var mi = typeof(OrchestratorService).GetMethod("ResolveTaskWorkAsync", - BindingFlags.NonPublic | BindingFlags.Instance)!; - try - { - var task = (Task)mi.Invoke(svc, [new JobDescriptor(runName, taskId, 4), CancellationToken.None])!; - await task; - return task.GetType().GetProperty("Result")!.GetValue(task); - } - catch (TargetInvocationException ex) - { - throw ex.InnerException ?? ex; - } - } - - private static async Task SeedRunAsync(OrchestratorTableStore store, string name, string status) - { - await store.InitializeAsync(); - await store.UpsertRunAsync(new OrchestratorRun - { - Name = name, - Status = status, - Priority = 4, - StartedUtc = DateTime.UtcNow, - // Pending on purpose: this is the state a coalesced terminal write has not caught up with, - // and the state that had the resolver hand back work for a task that had already run. - Tasks = [new OrchestratorTaskItem { Id = "task-0", Status = "Pending" }] - }); - await store.UpsertTaskAsync(name, new OrchestratorTaskItem { Id = "task-0", Status = "Pending" }); - } - - [Theory] - [InlineData("Completed")] - [InlineData("CompletedWithErrors")] - public async Task DescriptorForAFinishedRun_IsDropped_AndTheRunIsNotPutBackInTheLiveGraph(string status) - { - var (svc, store) = NewService(); - await SeedRunAsync(store, "finished-run", status); - ((ConcurrentDictionary)typeof(OrchestratorService) - .GetField("_taskScriptPaths", BindingFlags.NonPublic | BindingFlags.Instance)! - .GetValue(svc)!)["finished-run"] = "Invoke-CraftTask"; - - var work = await ResolveAsync(svc, "finished-run", "task-0"); - - Assert.Null(work); - - // The resurrection is the actual defect: once a finalized run is back in _activeRuns, the next - // completion re-finalizes it and dispatches its post-execution again. - var active = (ConcurrentDictionary)typeof(OrchestratorService) - .GetField("_activeRuns", BindingFlags.NonPublic | BindingFlags.Instance)!.GetValue(svc)!; - Assert.False(active.ContainsKey("finished-run"), - "a finalized run was rehydrated back into the live graph — it will finalize again"); - } - - // ── The finalize-once claim, and the paths that must not strand a run ────────────────────────── - - private static ConcurrentDictionary Claims(OrchestratorService svc) => - (ConcurrentDictionary)typeof(OrchestratorService) - .GetField("_finalizingRuns", BindingFlags.NonPublic | BindingFlags.Instance)!.GetValue(svc)!; - - /// - /// A second finalize is refused. This is the guard that keeps a run's Push-* aggregation to one - /// invocation; without it the observed run ran its aggregation seven times. - /// - [Fact] - public async Task FinalizeRun_RefusesASecondEntry() - { - var (svc, _) = NewService(); - Set(svc, "_finalizingRuns", new ConcurrentDictionary()); - - // Pre-claim, as a first finalize would. The second entry must return before touching anything — - // this instance has no writer or queue, so getting past the guard would throw rather than pass. - Claims(svc)["run"] = true; - - var run = new OrchestratorRun { Name = "run", Status = "Running", StartedUtc = DateTime.UtcNow }; - await InvokeFinalizeAsync(svc, run); - - // Untouched: a refused finalize must not restamp the run's status or completion time. - Assert.Equal("Running", run.Status); - Assert.Null(run.CompletedUtc); - } - - /// - /// The stranding risk the guard introduces: if a claim outlived a failed finalize, nothing would - /// ever finalize that run again — worse than the duplicate it prevents. A throw must release it. - /// - [Fact] - public async Task FinalizeRun_ReleasesTheClaim_WhenItThrows() - { - var (svc, _) = NewService(); - Set(svc, "_finalizingRuns", new ConcurrentDictionary()); - // _writer is left null, so the core finalize throws once past the guard. - - var run = new OrchestratorRun { Name = "run", Status = "Running", StartedUtc = DateTime.UtcNow }; - - await Assert.ThrowsAnyAsync(() => InvokeFinalizeAsync(svc, run)); - - Assert.False(Claims(svc).ContainsKey("run"), - "a failed finalize kept its claim — this run can never finalize again"); - } - - /// - /// Run names recur within one process (CIPPDBCacheOrchestrator, ProcessDeltaQueries fire on a - /// timer). Dispatching a run again is what makes it finalizable again. - /// - [Fact] - public async Task DispatchingARunAgain_ClearsAPreviousFinalizeClaim() - { - var (svc, _) = NewService(); - Set(svc, "_finalizingRuns", new ConcurrentDictionary()); - // No _runStatusTimers any more: the per-run status timer was replaced by a single sweep loop - // (RunStatusSweepLoopAsync), so DispatchPendingTasksAsync no longer creates or tracks a timer. - Claims(svc)["recurring-run"] = true; - - var run = new OrchestratorRun { Name = "recurring-run", Status = "Running", StartedUtc = DateTime.UtcNow }; - - // Only the bookkeeping prologue is exercised; the enqueue that follows needs a live queue. - try - { - var mi = typeof(OrchestratorService).GetMethod("DispatchPendingTasksAsync", - BindingFlags.NonPublic | BindingFlags.Instance)!; - await (Task)mi.Invoke(svc, [run, "Invoke-CraftTask", 4, CancellationToken.None, false])!; - } - catch { /* expected: no queue on this instance */ } - - Assert.False(Claims(svc).ContainsKey("recurring-run"), - "a recurring run kept its previous claim — its next outing would never finalize"); - } - - private static async Task InvokeFinalizeAsync(OrchestratorService svc, OrchestratorRun run) - { - var mi = typeof(OrchestratorService).GetMethod("FinalizeRunAsync", - BindingFlags.NonPublic | BindingFlags.Instance)!; - try - { - await (Task)mi.Invoke(svc, [run])!; - } - catch (TargetInvocationException ex) - { - throw ex.InnerException ?? ex; - } - } - - // ── A task already executing must not be started a second time ──────────────────────────────── - - /// - /// The crash-recovery race, at the resolver. - /// - /// A queue RowKey embeds the enqueue timestamp, so recovery re-dispatching a Pending task writes a - /// SECOND row rather than updating the surviving one. Most duplicates are harmless — by the time the - /// extra row is claimed the task has finished and it is dropped as terminal. But a claim that lands - /// while the task is still RUNNING used to pass the guard and start another copy: on a killed - /// 140-task fanout, a five-minute Intune collection was re-claimed four minutes in and ran twice. - /// - [Fact] - public async Task DescriptorForATaskThatIsAlreadyRunning_IsDropped() - { - var (svc, store) = NewService(); - await store.InitializeAsync(); - - var run = new OrchestratorRun - { - Name = "live-run", - Status = "Running", - Priority = 4, - StartedUtc = DateTime.UtcNow, - // OwnedHere: a worker in THIS process is executing it — the state dispatch leaves behind. - Tasks = [new OrchestratorTaskItem { Id = "task-0", Status = "Running", OwnedHere = true }] - }; - await store.UpsertRunAsync(run); - await store.UpsertTaskAsync("live-run", run.Tasks[0]); - - // In the live graph, mid-flight — the state a duplicate row races. - ((ConcurrentDictionary)typeof(OrchestratorService) - .GetField("_activeRuns", BindingFlags.NonPublic | BindingFlags.Instance)!.GetValue(svc)!)["live-run"] = run; - ((ConcurrentDictionary)typeof(OrchestratorService) - .GetField("_taskScriptPaths", BindingFlags.NonPublic | BindingFlags.Instance)! - .GetValue(svc)!)["live-run"] = "Invoke-CraftTask"; - - Assert.Null(await ResolveAsync(svc, "live-run", "task-0")); - } - - /// - /// The guard must not block genuine recovery. ResumeInterruptedRunsAsync flips interrupted tasks - /// from Running back to Pending before re-dispatching, so a task that really does need re-running - /// arrives here as Pending — and must resolve to work. - /// - [Fact] - public async Task ATaskResetToPendingByRecovery_StillResolvesToWork() - { - var (svc, store) = NewService(); - await SeedRunAsync(store, "recovered-run", "Running"); - ((ConcurrentDictionary)typeof(OrchestratorService) - .GetField("_taskScriptPaths", BindingFlags.NonPublic | BindingFlags.Instance)! - .GetValue(svc)!)["recovered-run"] = "Invoke-CraftTask"; - - Assert.NotNull(await ResolveAsync(svc, "recovered-run", "task-0")); - } - - [Fact] - public async Task DescriptorForAnUnfinishedRun_IsStillRehydrated() - { - // The guard must not swallow crash recovery: a Running run absent from memory is exactly what - // the rehydrate path exists for. - var (svc, store) = NewService(); - await SeedRunAsync(store, "live-run", "Running"); - ((ConcurrentDictionary)typeof(OrchestratorService) - .GetField("_taskScriptPaths", BindingFlags.NonPublic | BindingFlags.Instance)! - .GetValue(svc)!)["live-run"] = "Invoke-CraftTask"; - - var work = await ResolveAsync(svc, "live-run", "task-0"); - - Assert.NotNull(work); - - var active = (ConcurrentDictionary)typeof(OrchestratorService) - .GetField("_activeRuns", BindingFlags.NonPublic | BindingFlags.Instance)!.GetValue(svc)!; - Assert.True(active.ContainsKey("live-run"), - "an unfinished run was not re-established in the live graph — sibling completion tracking breaks"); - } -} diff --git a/tests/Craft.Tests/OrchestratorRedriveGuardTests.cs b/tests/Craft.Tests/OrchestratorRedriveGuardTests.cs deleted file mode 100644 index 57c1221..0000000 --- a/tests/Craft.Tests/OrchestratorRedriveGuardTests.cs +++ /dev/null @@ -1,163 +0,0 @@ -using System.Collections.Concurrent; -using System.Reflection; -using Craft.Configuration; -using Craft.Orchestration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// Pins the re-drive watchdog's bounds: the status sweep starts it for every live run each tick without -/// awaiting it, so it must never run two verifications for one run, never more than a few at once across -/// runs, and never re-drive a candidate that finished while its verification was waiting. -/// -public class OrchestratorRedriveGuardTests -{ - private sealed record Harness(OrchestratorService Svc, JobQueueStore Queue, - RunRemainingCounterTests.ConditionalStore Backing, string IndexTable); - - private static async Task NewHarnessAsync() - { - var settings = new CraftSettings { Orchestrator = { TablePrefix = "rdg" + Guid.NewGuid().ToString("N")[..8] } }; - var backing = new RunRemainingCounterTests.ConditionalStore(); - var queue = new JobQueueStore(NullLogger.Instance, settings, backing); - await queue.InitializeAsync(); - - var svc = (OrchestratorService)System.Runtime.CompilerServices.RuntimeHelpers - .GetUninitializedObject(typeof(OrchestratorService)); - Set(svc, "_logger", NullLogger.Instance); - Set(svc, "_queue", queue); - Set(svc, "_settings", settings); - Set(svc, "_lock", new object()); - Set(svc, "_activeSequentialDrivers", new ConcurrentDictionary()); - Set(svc, "_requeueFailures", new ConcurrentDictionary()); - Set(svc, "_deferrals", NewFieldValue("_deferrals")); - Set(svc, "_redriveBackoff", NewFieldValue("_redriveBackoff")); - Set(svc, "_redriveInFlight", NewFieldValue("_redriveInFlight")); - Set(svc, "_redriveSlots", new SemaphoreSlim(8, 8)); - Set(svc, "_redriveBackoffEnabled", false); - Set(svc, "_redriveBase", TimeSpan.FromSeconds(60)); - var jm = (JobManager)System.Runtime.CompilerServices.RuntimeHelpers.GetUninitializedObject(typeof(JobManager)); - var jobsField = typeof(JobManager).GetField("_jobs", BindingFlags.NonPublic | BindingFlags.Instance)!; - jobsField.SetValue(jm, Activator.CreateInstance(jobsField.FieldType)); - Set(svc, "_jobManager", jm); - return new Harness(svc, queue, backing, $"{settings.Orchestrator.TablePrefix}QueueIndex"); - } - - private static object NewFieldValue(string field) => - Activator.CreateInstance(typeof(OrchestratorService) - .GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)!.FieldType)!; - - private static void Set(object target, string field, object? value) => - typeof(OrchestratorService).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)! - .SetValue(target, value); - - private static Task Redrive(Harness h, OrchestratorRun run) => - (Task)typeof(OrchestratorService).GetMethod("RedrivePendingTasksAsync", BindingFlags.NonPublic | BindingFlags.Instance)! - .Invoke(h.Svc, [run])!; - - private static OrchestratorRun MakeRun(string name, int count) => new() - { - Name = name, - Status = "Running", - Priority = 4, - StartedUtc = DateTime.UtcNow, - Tasks = Enumerable.Range(0, count) - .Select(i => new OrchestratorTaskItem { Id = $"{name}_t{i}", Status = "Pending" }).ToList() - }; - - /// Holds every index read open until , counting how many are held at once. - private sealed class ReadGate - { - private readonly TaskCompletionSource _open = new(TaskCreationOptions.RunContinuationsAsynchronously); - private int _held; - public int Entered; - public int MaxHeld; - - public ReadGate(Harness h) => h.Backing.OnPartitionQuery = async table => - { - if (table != h.IndexTable) return; - Interlocked.Increment(ref Entered); - var held = Interlocked.Increment(ref _held); - int seen; - while (held > (seen = Volatile.Read(ref MaxHeld)) && Interlocked.CompareExchange(ref MaxHeld, held, seen) != seen) { } - await _open.Task; - Interlocked.Decrement(ref _held); - }; - - public void Release() => _open.TrySetResult(); - - public async Task WaitForEnteredAsync(int count) - { - for (var i = 0; i < 200 && Volatile.Read(ref Entered) < count; i++) await Task.Delay(10); - } - } - - [Fact] - public async Task SecondTick_WhileAVerificationIsInFlight_DoesNotStartAnother() - { - var h = await NewHarnessAsync(); - var run = MakeRun("inflight", 3); - await h.Queue.EnqueueBatchAsync(run.Name, run.Tasks.Select(t => (t.Id, 4)).ToList(), DateTime.UtcNow); - var gate = new ReadGate(h); - - var first = Redrive(h, run); - await gate.WaitForEnteredAsync(1); - var second = Redrive(h, run); - - Assert.True(second.IsCompleted); - gate.Release(); - await first; - Assert.Equal(1, gate.Entered); - } - - [Fact] - public async Task ManyRuns_VerifyAtMostEightAtOnce_AndAllEventuallyVerify() - { - var h = await NewHarnessAsync(); - var runs = Enumerable.Range(0, 20).Select(i => MakeRun($"cap{i:D2}", 2)).ToList(); - foreach (var run in runs) - await h.Queue.EnqueueBatchAsync(run.Name, run.Tasks.Select(t => (t.Id, 4)).ToList(), DateTime.UtcNow); - var gate = new ReadGate(h); - - var ticks = runs.Select(r => Redrive(h, r)).ToList(); - await gate.WaitForEnteredAsync(8); - await Task.Delay(100); - - Assert.Equal(8, gate.Entered); - gate.Release(); - await Task.WhenAll(ticks); - Assert.Equal(20, gate.Entered); - Assert.True(gate.MaxHeld <= 8); - } - - [Fact] - public async Task CandidateThatFinishesDuringTheRead_IsNotReDriven() - { - var h = await NewHarnessAsync(); - var run = MakeRun("finished", 1); - var gate = new ReadGate(h); - - var tick = Redrive(h, run); - await gate.WaitForEnteredAsync(1); - run.Tasks[0].Status = "Completed"; - gate.Release(); - await tick; - await Task.Delay(200); - - Assert.Empty(await h.Queue.GetQueuedTaskIdsAsync(run.Name)); - } - - [Fact] - public async Task OrphanStillPendingAfterTheRead_IsReDriven() - { - var h = await NewHarnessAsync(); - var run = MakeRun("orphan", 1); - - await Redrive(h, run); - for (var i = 0; i < 100 && (await h.Queue.GetQueuedTaskIdsAsync(run.Name)).Count == 0; i++) await Task.Delay(20); - - Assert.Equal(["orphan_t0"], await h.Queue.GetQueuedTaskIdsAsync(run.Name)); - } -} diff --git a/tests/Craft.Tests/OrchestratorResultStreamingTests.cs b/tests/Craft.Tests/OrchestratorResultStreamingTests.cs index 59848e8..c129b0a 100644 --- a/tests/Craft.Tests/OrchestratorResultStreamingTests.cs +++ b/tests/Craft.Tests/OrchestratorResultStreamingTests.cs @@ -98,14 +98,21 @@ public Task DeletePartitionAsync(string table, string partitionKey, Cancellation } } - private static (OrchestratorTableStore Store, LazyProbeStore Backing) NewStore() + private static (ResultStore Store, LazyProbeStore Backing) NewStore() { var backing = new LazyProbeStore(); - var store = new OrchestratorTableStore( - NullLogger.Instance, new CraftSettings(), backing); + var store = new ResultStore( + NullLogger.Instance, new CraftSettings(), backing); return (store, backing); } + private static async Task ReadAllAsync(ResultStore store, string run) + { + var all = new List(); + await foreach (var r in store.StreamResultsAsync(run)) all.Add(r); + return all.ToArray(); + } + /// A result whose JSON exceeds the per-property limit and so gets chunked. private static string BigJson(int chars) => "\"" + new string('x', chars - 2) + "\""; @@ -211,7 +218,7 @@ public async Task SingleProperty_MultiChunkSingleRow_AndMultiRowSpill_AllRoundTr await store.StoreResultAsync("run", "b-chunked", chunked); await store.StoreResultAsync("run", "c-spilled", spilled); - var results = await store.GetResultsAsync("run"); + var results = await ReadAllAsync(store, "run"); Assert.Equal(3, results.Length); Assert.Contains(small, results); @@ -233,7 +240,7 @@ public async Task SpilledResult_Reassembles_WhenRowsArriveOutOfOrder() // still complete by chunk count rather than by arrival order. backing.Order = rows => rows.Reverse(); - var results = await store.GetResultsAsync("run"); + var results = await ReadAllAsync(store, "run"); Assert.Equal(2, results.Length); Assert.Contains(spilled, results); @@ -346,7 +353,7 @@ public async Task EmptyRun_WritesAnEmptyFile() // lines yields zero results without needing to special-case it. Assert.Equal(0, count); Assert.Equal("", await File.ReadAllTextAsync(path)); - Assert.Empty(await store.GetResultsAsync("run")); + Assert.Empty(await ReadAllAsync(store, "run")); } finally { diff --git a/tests/Craft.Tests/OrchestratorResultsAzuriteTests.cs b/tests/Craft.Tests/OrchestratorResultsAzuriteTests.cs index a8b5a0c..0174075 100644 --- a/tests/Craft.Tests/OrchestratorResultsAzuriteTests.cs +++ b/tests/Craft.Tests/OrchestratorResultsAzuriteTests.cs @@ -26,7 +26,7 @@ namespace Craft.Tests; /// public class OrchestratorResultsAzuriteTests { - private static async Task TryConnectAsync() + private static async Task TryConnectAsync() { var settings = new CraftSettings(); @@ -54,8 +54,8 @@ public class OrchestratorResultsAzuriteTests return null; } - var store = new OrchestratorTableStore( - NullLogger.Instance, settings, backing); + var store = new ResultStore( + NullLogger.Instance, settings, backing); await store.InitializeAsync(); return store; } @@ -99,7 +99,7 @@ public async Task EveryStorageShape_RoundTrips_AsExactlyOneLine() finally { if (File.Exists(path)) File.Delete(path); - await store.CleanupRunAsync("run"); + await store.DeleteRunAsync("run"); } } @@ -134,7 +134,7 @@ public async Task ManyResults_StreamOneLineEach_WithSpillsInterleaved() finally { if (File.Exists(path)) File.Delete(path); - await store.CleanupRunAsync("run"); + await store.DeleteRunAsync("run"); } } } diff --git a/tests/Craft.Tests/OrchestratorRetentionAzuriteTests.cs b/tests/Craft.Tests/OrchestratorRetentionAzuriteTests.cs deleted file mode 100644 index a86369f..0000000 --- a/tests/Craft.Tests/OrchestratorRetentionAzuriteTests.cs +++ /dev/null @@ -1,162 +0,0 @@ -using Azure; -using Azure.Data.Tables; -using Craft.Configuration; -using Craft.Orchestration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// The retention sweep against real tables (Azurite, or a storage account via -/// CRAFT_TEST_TABLE_CONNECTION). Two things only the real backend can prove: the orphan scan's -/// $select projection is accepted and still yields PartitionKey and Timestamp, and -/// DeletePartitionAsync really empties a partition written through the normal paths. Skipped, not -/// failed, when no backend is reachable — with the same caveat as -/// : a skip looks like a pass. -/// -public class OrchestratorRetentionAzuriteTests -{ - private sealed class Fixture : IAsyncDisposable - { - public required OrchestratorTableStore Store { get; init; } - public required AzureTableStore Backing { get; init; } - public required CraftSettings Settings { get; init; } - public required string Connection { get; init; } - - public static async Task TryConnectAsync() - { - var settings = new CraftSettings(); - var connection = Environment.GetEnvironmentVariable("CRAFT_TEST_TABLE_CONNECTION"); - if (!string.IsNullOrWhiteSpace(connection)) - settings.Auth.UserStorageConnection = connection; - else - { - settings.Storage.AllowDevelopmentStorage = true; - connection = "UseDevelopmentStorage=true"; - } - - // Unique per run so repeated runs cannot see each other's rows. - settings.Orchestrator.TablePrefix = "azrt" + Guid.NewGuid().ToString("N")[..8]; - - var backing = new AzureTableStore(settings); - try - { - using var cts = new CancellationTokenSource(TimeSpan.FromSeconds(3)); - await backing.PingAsync(cts.Token); - } - catch - { - return null; - } - - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); - await store.InitializeAsync(); - return new Fixture { Store = store, Backing = backing, Settings = settings, Connection = connection }; - } - - public string Table(string suffix) => Settings.Orchestrator.TablePrefix + suffix; - - public async Task CountAsync(string suffix, string partition) - { - var n = 0; - await foreach (var _ in Backing.QueryPartitionAsync(Table(suffix), partition)) n++; - return n; - } - - public Task AddRunAsync(string name, string status, DateTime started, DateTime? completed) => - Store.UpsertRunAsync(new OrchestratorRun { Name = name, Status = status, StartedUtc = started, CompletedUtc = completed }); - - public Task AddTasksAsync(string run, int count) => - Store.UpsertTaskBatchAsync(run, Enumerable.Range(1, count) - .Select(i => new OrchestratorTaskItem { Id = $"task-{i}", Status = "Completed" }).ToList()); - - public async ValueTask DisposeAsync() - { - // Drop the per-run tables so repeated local runs do not pile fixtures into the emulator. - var service = new TableServiceClient(Connection); - foreach (var suffix in new[] { "Runs", "Tasks", "Results" }) - { - try { await service.DeleteTableAsync(Table(suffix)); } - catch (RequestFailedException) { /* already gone */ } - } - } - } - - [Fact] - public async Task Sweep_RemovesFinishedRunsPastRetention_AndKeepsEverythingStillWanted() - { - await using var fx = await Fixture.TryConnectAsync(); - if (fx == null) return; - - var old = DateTime.UtcNow.AddDays(-3); - var recent = DateTime.UtcNow.AddHours(-1); - - await fx.AddRunAsync("done-old", "Completed", old, old); - await fx.AddTasksAsync("done-old", 3); - await fx.Store.StoreResultAsync("done-old", "task-1", "{\"ok\":true}"); - - await fx.AddRunAsync("cancelled-old", "Cancelled", old, old); - await fx.AddTasksAsync("cancelled-old", 1); - - await fx.AddRunAsync("failed-recent", "Failed", recent, recent); - await fx.AddTasksAsync("failed-recent", 2); - - // Started long ago, nobody in this process drives it, but its counter row was written just - // now — the heartbeat that says another process (or this one, moments ago) is still at it. - await fx.AddRunAsync("running", "Running", DateTime.UtcNow.AddDays(-10), null); - await fx.AddTasksAsync("running", 2); - await fx.Store.InitRemainingAsync("running", 2); - - // No Run row, but written moments ago: an orphan, just not an old one. - await fx.AddTasksAsync("ghost", 2); - - var result = await fx.Store.CleanupOldRunsAsync(TimeSpan.FromHours(48), new HashSet()); - - Assert.Collection(result.ExpiredRuns.OrderBy(n => n, StringComparer.Ordinal), - n => Assert.Equal("cancelled-old", n), - n => Assert.Equal("done-old", n)); - Assert.Empty(result.AbandonedRuns); - Assert.Equal(0, result.OrphanPartitionsRemoved); - Assert.Equal(4, result.RunsExamined); - - Assert.Null(await fx.Store.GetRunAsync("done-old")); - Assert.Null(await fx.Store.GetRunAsync("cancelled-old")); - Assert.Equal(0, await fx.CountAsync("Tasks", "done-old")); - Assert.Equal(0, await fx.CountAsync("Results", "done-old")); - Assert.Equal(0, await fx.CountAsync("Tasks", "cancelled-old")); - - Assert.NotNull(await fx.Store.GetRunAsync("failed-recent")); - Assert.Equal(2, await fx.CountAsync("Tasks", "failed-recent")); - Assert.Equal(2, (await fx.Store.GetRunAsync("running"))!.Tasks.Count); - Assert.Equal(2, await fx.CountAsync("Tasks", "ghost")); - } - - [Fact] - public async Task Sweep_RemovesOrphanedPartitions_OnceNothingInThemIsRecent() - { - await using var fx = await Fixture.TryConnectAsync(); - if (fx == null) return; - - await fx.AddTasksAsync("ghost", 3); - await fx.Store.StoreResultAsync("ghost", "task-1", "{\"ok\":true}"); - - // A run this process is driving: exempt from the abandoned rule, and its partition is never - // an orphan because its Run row is there. - await fx.AddRunAsync("kept", "Running", DateTime.UtcNow, null); - await fx.AddTasksAsync("kept", 2); - - // Every row was stamped by the service a moment ago. A cutoff in the future is the only way - // to make them "old" without waiting, and it also absorbs any skew between the emulator's - // clock and this process's. - var result = await fx.Store.CleanupOldRunsAsync(TimeSpan.FromMinutes(-5), new HashSet { "kept" }); - - Assert.Equal(2, result.OrphanPartitionsRemoved); - Assert.Empty(result.ExpiredRuns); - Assert.Empty(result.AbandonedRuns); - Assert.Equal(0, await fx.CountAsync("Tasks", "ghost")); - Assert.Equal(0, await fx.CountAsync("Results", "ghost")); - Assert.Equal(2, await fx.CountAsync("Tasks", "kept")); - Assert.NotNull(await fx.Store.GetRunAsync("kept")); - } -} diff --git a/tests/Craft.Tests/OrchestratorRetentionTests.cs b/tests/Craft.Tests/OrchestratorRetentionTests.cs deleted file mode 100644 index 4f833a2..0000000 --- a/tests/Craft.Tests/OrchestratorRetentionTests.cs +++ /dev/null @@ -1,314 +0,0 @@ -using Craft.Configuration; -using Craft.Orchestration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// The retention sweep is the only thing that bounds the orchestrator tables on a host that is not -/// restarted, and every rule in it is a rule about what NOT to delete while Craft is live. These pin -/// those rules against an in-memory backend; proves -/// the same sweep against real tables, where the projection and the partition deletes are real. -/// -public class OrchestratorRetentionTests -{ - private sealed class FakeStore : ICraftTableStore - { - private readonly Dictionary> _tables = new(); - - public Task TryReplaceBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) => throw new NotSupportedException(); - - public Task PingAsync(CancellationToken ct = default) => Task.CompletedTask; - - public Task EnsureTableAsync(string table, CancellationToken ct = default) - { - if (!_tables.ContainsKey(table)) _tables[table] = new(); - return Task.CompletedTask; - } - - public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) - { - _tables[table][(row.PartitionKey, row.RowKey)] = row; - return Task.CompletedTask; - } - - public Task UpsertBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) - { - foreach (var r in rows) _tables[table][(r.PartitionKey, r.RowKey)] = r; - return Task.CompletedTask; - } - - public Task GetAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) - => Task.FromResult(_tables[table].TryGetValue((partitionKey, rowKey), out var r) ? r : null); - - public async IAsyncEnumerable QueryPartitionAsync(string table, string partitionKey, - [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) - { - foreach (var kv in _tables[table].Where(k => k.Key.Item1 == partitionKey).ToList()) - { - yield return kv.Value; - await Task.Yield(); - } - } - - public async IAsyncEnumerable QueryTableAsync(string table, - [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) - { - foreach (var kv in _tables[table].ToList()) - { - yield return kv.Value; - await Task.Yield(); - } - } - - public Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) - { - _tables[table].Remove((partitionKey, rowKey)); - return Task.CompletedTask; - } - - public Task DeletePartitionAsync(string table, string partitionKey, CancellationToken ct = default) - { - foreach (var k in _tables[table].Keys.Where(k => k.Item1 == partitionKey).ToList()) - _tables[table].Remove(k); - return Task.CompletedTask; - } - } - - private static readonly TimeSpan Retention = TimeSpan.FromHours(48); - private static readonly HashSet NothingActive = new(StringComparer.Ordinal); - private static DateTime Old => DateTime.UtcNow.AddDays(-3); - private static DateTime Recent => DateTime.UtcNow.AddHours(-1); - - private sealed record Harness(OrchestratorTableStore Store, FakeStore Backing, CraftSettings Settings) - { - public string Runs => Settings.Orchestrator.TablePrefix + "Runs"; - public string Tasks => Settings.Orchestrator.TablePrefix + "Tasks"; - public string Results => Settings.Orchestrator.TablePrefix + "Results"; - - public Task AddRunAsync(string name, string status, DateTime started, DateTime? completed) => - Store.UpsertRunAsync(new OrchestratorRun { Name = name, Status = status, StartedUtc = started, CompletedUtc = completed }); - - public Task AddTasksAsync(string run, int count) => - Store.UpsertTaskBatchAsync(run, Enumerable.Range(1, count) - .Select(i => new OrchestratorTaskItem { Id = $"task-{i}", Status = "Completed" }).ToList()); - - /// A raw row carrying a Timestamp, the way a real backend returns every row. - public Task AddStampedRowAsync(string table, string partition, string rowKey, DateTime stamp) => - Backing.UpsertAsync(table, new StoreRow(partition, rowKey) { Timestamp = new DateTimeOffset(stamp, TimeSpan.Zero) }); - - public async Task CountAsync(string table, string partition) - { - var n = 0; - await foreach (var _ in Backing.QueryPartitionAsync(table, partition)) n++; - return n; - } - } - - private static async Task NewHarnessAsync() - { - var settings = new CraftSettings(); - var backing = new FakeStore(); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); - await store.InitializeAsync(); - return new Harness(store, backing, settings); - } - - [Fact] - public async Task CancelledRun_PastRetention_IsRemovedWithItsPartitions() - { - var h = await NewHarnessAsync(); - await h.AddRunAsync("cancelled", "Cancelled", Old, Old); - await h.AddTasksAsync("cancelled", 3); - await h.Store.StoreResultAsync("cancelled", "task-1", "{\"ok\":true}"); - - var result = await h.Store.CleanupOldRunsAsync(Retention); - - Assert.Equal("cancelled", Assert.Single(result.ExpiredRuns)); - Assert.Null(await h.Store.GetRunAsync("cancelled")); - Assert.Equal(0, await h.CountAsync(h.Tasks, "cancelled")); - Assert.Equal(0, await h.CountAsync(h.Results, "cancelled")); - } - - [Theory] - [InlineData("Completed")] - [InlineData("CompletedWithErrors")] - [InlineData("Failed")] - [InlineData("Cancelled")] - public async Task EveryTerminalStatus_PastRetention_IsRemoved(string status) - { - var h = await NewHarnessAsync(); - await h.AddRunAsync("run", status, Old, Old); - - var result = await h.Store.CleanupOldRunsAsync(Retention); - - Assert.Equal("run", Assert.Single(result.ExpiredRuns)); - Assert.Equal(1, result.RunsExamined); - Assert.Null(await h.Store.GetRunAsync("run")); - } - - [Fact] - public async Task FinishedRun_InsideRetention_IsKept_WithItsRows() - { - var h = await NewHarnessAsync(); - await h.AddRunAsync("recent", "Completed", Old, Recent); - await h.AddTasksAsync("recent", 2); - - var result = await h.Store.CleanupOldRunsAsync(Retention); - - Assert.Empty(result.ExpiredRuns); - Assert.NotNull(await h.Store.GetRunAsync("recent")); - Assert.Equal(2, await h.CountAsync(h.Tasks, "recent")); - } - - [Fact] - public async Task FinishedRun_WithoutCompletedUtc_IsJudgedByStartedUtc() - { - var h = await NewHarnessAsync(); - await h.AddRunAsync("failed-old", "Failed", Old, null); - await h.AddRunAsync("failed-recent", "Failed", Recent, null); - - var result = await h.Store.CleanupOldRunsAsync(Retention); - - Assert.Equal("failed-old", Assert.Single(result.ExpiredRuns)); - Assert.NotNull(await h.Store.GetRunAsync("failed-recent")); - } - - [Fact] - public async Task ActiveRun_IsKept_HoweverLongAgoItStarted() - { - var h = await NewHarnessAsync(); - await h.AddRunAsync("long", "Running", DateTime.UtcNow.AddDays(-10), null); - await h.AddTasksAsync("long", 2); - - var result = await h.Store.CleanupOldRunsAsync(Retention, new HashSet { "long" }); - - Assert.Empty(result.AbandonedRuns); - Assert.NotNull(await h.Store.GetRunAsync("long")); - Assert.Equal(2, await h.CountAsync(h.Tasks, "long")); - } - - [Fact] - public async Task RunNobodyIsDriving_WithNoWritesWithinRetention_IsAbandoned() - { - var h = await NewHarnessAsync(); - await h.AddRunAsync("stuck", "Pending", Old, null); - await h.AddTasksAsync("stuck", 2); - - var result = await h.Store.CleanupOldRunsAsync(Retention, NothingActive); - - Assert.Equal("stuck", Assert.Single(result.AbandonedRuns)); - Assert.Empty(result.ExpiredRuns); - Assert.Null(await h.Store.GetRunAsync("stuck")); - Assert.Equal(0, await h.CountAsync(h.Tasks, "stuck")); - } - - [Fact] - public async Task RunNobodyIsDriving_WithAFreshHeartbeat_IsKept() - { - // Started long ago, but a task completed an hour ago: the counter row is the heartbeat. - var h = await NewHarnessAsync(); - await h.AddRunAsync("slow", "Running", DateTime.UtcNow.AddDays(-10), null); - await h.AddTasksAsync("slow", 2); - await h.AddStampedRowAsync(h.Tasks, "slow", "!!run-counter", Recent); - - var result = await h.Store.CleanupOldRunsAsync(Retention, NothingActive); - - Assert.Empty(result.AbandonedRuns); - Assert.NotNull(await h.Store.GetRunAsync("slow")); - Assert.Equal(3, await h.CountAsync(h.Tasks, "slow")); - } - - [Fact] - public async Task UnknownStatus_IsTreatedAsNotFinished() - { - var h = await NewHarnessAsync(); - await h.AddRunAsync("odd-stale", "Suspended", Old, null); - await h.AddRunAsync("odd-fresh", "Suspended", Old, null); - await h.AddStampedRowAsync(h.Tasks, "odd-fresh", "!!run-counter", Recent); - - var result = await h.Store.CleanupOldRunsAsync(Retention, NothingActive); - - Assert.Equal("odd-stale", Assert.Single(result.AbandonedRuns)); - Assert.Empty(result.ExpiredRuns); - Assert.NotNull(await h.Store.GetRunAsync("odd-fresh")); - } - - [Fact] - public async Task OrphanedPartitions_WithOnlyOldRows_AreRemoved() - { - var h = await NewHarnessAsync(); - await h.AddStampedRowAsync(h.Tasks, "ghost", "task-1", Old); - await h.AddStampedRowAsync(h.Tasks, "ghost", "task-2", Old); - await h.AddStampedRowAsync(h.Results, "ghost", "task-1", Old); - - var result = await h.Store.CleanupOldRunsAsync(Retention); - - Assert.Equal(2, result.OrphanPartitionsRemoved); - Assert.Equal(0, await h.CountAsync(h.Tasks, "ghost")); - Assert.Equal(0, await h.CountAsync(h.Results, "ghost")); - } - - [Fact] - public async Task OrphanedPartition_WithAFreshRow_IsKeptWhole() - { - var h = await NewHarnessAsync(); - await h.AddStampedRowAsync(h.Tasks, "ghost", "task-1", Old); - await h.AddStampedRowAsync(h.Tasks, "ghost", "task-2", Old); - await h.AddStampedRowAsync(h.Tasks, "ghost", "task-3", Recent); - - var result = await h.Store.CleanupOldRunsAsync(Retention); - - Assert.Equal(0, result.OrphanPartitionsRemoved); - Assert.Equal(3, await h.CountAsync(h.Tasks, "ghost")); - } - - [Fact] - public async Task PartitionOfARetainedRun_IsNotAnOrphan_EvenWhenItsRowsAreOld() - { - var h = await NewHarnessAsync(); - await h.AddRunAsync("resumed", "Completed", Old, Recent); - await h.AddStampedRowAsync(h.Tasks, "resumed", "task-1", Old); - await h.AddStampedRowAsync(h.Results, "resumed", "task-1", Old); - - var result = await h.Store.CleanupOldRunsAsync(Retention); - - Assert.Equal(0, result.OrphanPartitionsRemoved); - Assert.Equal(1, await h.CountAsync(h.Tasks, "resumed")); - Assert.Equal(1, await h.CountAsync(h.Results, "resumed")); - } - - [Fact] - public async Task RowsWithoutATimestamp_AreNeverJudgedOrphans() - { - // A backend that cannot say how old a row is gets the benefit of the doubt. - var h = await NewHarnessAsync(); - await h.AddTasksAsync("ghost", 2); - - var result = await h.Store.CleanupOldRunsAsync(Retention); - - Assert.Equal(0, result.OrphanPartitionsRemoved); - Assert.Equal(2, await h.CountAsync(h.Tasks, "ghost")); - } - - [Fact] - public async Task Sweep_ReportsEverythingItExamined() - { - var h = await NewHarnessAsync(); - await h.AddRunAsync("done", "Completed", Old, Old); - await h.AddRunAsync("live", "Running", Recent, null); - await h.AddRunAsync("stuck", "Pending", Old, null); - await h.AddStampedRowAsync(h.Results, "ghost", "r", Old); - - var result = await h.Store.CleanupOldRunsAsync(Retention, new HashSet { "live" }); - - Assert.Equal(3, result.RunsExamined); - Assert.Equal("done", Assert.Single(result.ExpiredRuns)); - Assert.Equal("stuck", Assert.Single(result.AbandonedRuns)); - Assert.Equal(1, result.OrphanPartitionsRemoved); - Assert.NotNull(await h.Store.GetRunAsync("live")); - } -} diff --git a/tests/Craft.Tests/OrchestratorRunPersistenceTests.cs b/tests/Craft.Tests/OrchestratorRunPersistenceTests.cs deleted file mode 100644 index f3ca354..0000000 --- a/tests/Craft.Tests/OrchestratorRunPersistenceTests.cs +++ /dev/null @@ -1,249 +0,0 @@ -using System.Runtime.CompilerServices; -using Craft.Configuration; -using Craft.Orchestration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// The run row is what a restart rebuilds a run's identity from. Anything on -/// that is not written here is silently null after recovery, and the -/// feature that depends on it stops working without any error. -/// -/// Reference and ParentRunName were exactly that: present on the model, never persisted. A resumed run -/// came back unreferenceable (QueueStatusBridge could not look it up) and orphaned (its finalize never -/// re-checked the parent, and the parent had no record of it). -/// -public class OrchestratorRunPersistenceTests -{ - private sealed class FakeStore : ICraftTableStore - { - - // Claims are not exercised by this fake. Fail loudly rather than pretend the guard held — - // a silent 'true' here would look exactly like a successful claim. - public Task TryReplaceBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) => throw new NotSupportedException(); - private readonly Dictionary> _tables = new(); - - public Task PingAsync(CancellationToken ct = default) => Task.CompletedTask; - - public Task EnsureTableAsync(string table, CancellationToken ct = default) - { - if (!_tables.ContainsKey(table)) _tables[table] = new(); - return Task.CompletedTask; - } - - public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) - { - _tables[table][(row.PartitionKey, row.RowKey)] = row; - return Task.CompletedTask; - } - - public Task UpsertBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) - { - foreach (var r in rows) _tables[table][(r.PartitionKey, r.RowKey)] = r; - return Task.CompletedTask; - } - - public Task GetAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) - => Task.FromResult(_tables[table].TryGetValue((partitionKey, rowKey), out var r) ? r : null); - - public async IAsyncEnumerable QueryPartitionAsync(string table, string partitionKey, - [EnumeratorCancellation] CancellationToken ct = default) - { - foreach (var kv in _tables[table].Where(k => k.Key.Item1 == partitionKey).ToList()) - { - yield return kv.Value; - await Task.Yield(); - } - } - - public async IAsyncEnumerable QueryTableAsync(string table, - [EnumeratorCancellation] CancellationToken ct = default) - { - foreach (var kv in _tables[table].ToList()) - { - yield return kv.Value; - await Task.Yield(); - } - } - - public Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) - { - _tables[table].Remove((partitionKey, rowKey)); - return Task.CompletedTask; - } - - public Task DeletePartitionAsync(string table, string partitionKey, CancellationToken ct = default) - { - foreach (var k in _tables[table].Keys.Where(k => k.Item1 == partitionKey).ToList()) - _tables[table].Remove(k); - return Task.CompletedTask; - } - } - - private static OrchestratorTableStore NewStore() => - new(NullLogger.Instance, new CraftSettings(), new FakeStore()); - - [Fact] - public async Task Reference_And_ParentRunName_SurviveARestart() - { - var store = NewStore(); - await store.InitializeAsync(); - - await store.UpsertRunAsync(new OrchestratorRun - { - Name = "ChildRun", - Reference = "user-request-4711", - ParentRunName = "ParentRun", - Status = "Running", - Priority = 3, - StartedUtc = DateTime.UtcNow, - }); - - var recovered = await store.GetRunAsync("ChildRun"); - - Assert.NotNull(recovered); - Assert.Equal("user-request-4711", recovered!.Reference); - Assert.Equal("ParentRun", recovered.ParentRunName); - } - - /// - /// PostExecAttemptCount is what bounds the retry of a failed post-execution across restarts. If it - /// did not survive the restart it would read back as 0 every time, and the bound would never be - /// reached — a post-execution that crashes the host would be retried on every start forever. - /// - [Fact] - public async Task PostExecAttemptCount_SurvivesARestart() - { - var store = NewStore(); - await store.InitializeAsync(); - - await store.UpsertRunAsync(new OrchestratorRun - { - Name = "AggregatedRun", - Status = "Completed", - PostExecFunctionName = "StoreMailboxPermissions", - PostExecStatus = "Failed", - PostExecAttemptCount = 2, - StartedUtc = DateTime.UtcNow, - }); - - var recovered = await store.GetRunAsync("AggregatedRun"); - - Assert.Equal("Failed", recovered!.PostExecStatus); - Assert.Equal(2, recovered.PostExecAttemptCount); - } - - /// - /// Rows written before the counter existed have no such property. They must read as 0 — an - /// already-exhausted reading would abandon in-flight aggregations on the upgrade restart. - /// - [Fact] - public async Task RunWrittenWithoutTheCounter_ReadsAsZeroAttempts() - { - var store = NewStore(); - await store.InitializeAsync(); - - await store.UpsertRunAsync(new OrchestratorRun - { - Name = "Legacy", - Status = "Completed", - PostExecStatus = "Pending", - StartedUtc = DateTime.UtcNow, - }); - - Assert.Equal(0, (await store.GetRunAsync("Legacy"))!.PostExecAttemptCount); - } - - /// Runs without either field keep round-tripping as null — no backfill required. - [Fact] - public async Task RunsWithoutReferenceOrParent_RoundTripAsNull() - { - var store = NewStore(); - await store.InitializeAsync(); - - await store.UpsertRunAsync(new OrchestratorRun - { - Name = "Plain", - Status = "Running", - Priority = 5, - StartedUtc = DateTime.UtcNow, - }); - - var recovered = await store.GetRunAsync("Plain"); - - Assert.Null(recovered!.Reference); - Assert.Null(recovered.ParentRunName); - } - - /// - /// The summary scan is what startup uses to rebuild parent→child links. It must report parentage - /// without loading task rows — using GetRunAsync per run would pull every task of every run. - /// - [Fact] - public async Task ListRunSummaries_ReportsParentage_WithoutLoadingTasks() - { - var store = NewStore(); - await store.InitializeAsync(); - - var childTasks = Enumerable.Range(0, 50) - .Select(i => new OrchestratorTaskItem { Id = $"t{i}", Status = "Pending" }).ToList(); - - await store.UpsertRunAsync(new OrchestratorRun - { - Name = "Parent", - Status = "Running", - StartedUtc = DateTime.UtcNow, - }); - await store.UpsertRunAsync(new OrchestratorRun - { - Name = "Child", - ParentRunName = "Parent", - Status = "Running", - StartedUtc = DateTime.UtcNow, - Tasks = childTasks, - }); - await store.UpsertRunAsync(new OrchestratorRun - { - Name = "DoneChild", - ParentRunName = "Parent", - Status = "Completed", - StartedUtc = DateTime.UtcNow, - }); - await store.UpsertTaskBatchAsync("Child", childTasks); - - var summaries = await store.ListRunSummariesAsync(); - - Assert.Equal(3, summaries.Count); - Assert.Null(summaries.Single(s => s.Name == "Parent").ParentRunName); - Assert.Equal("Parent", summaries.Single(s => s.Name == "Child").ParentRunName); - Assert.Equal("Running", summaries.Single(s => s.Name == "Child").Status); - - // Terminal children are filtered out of the rebuild by status — they cannot block a parent. - Assert.Equal("Completed", summaries.Single(s => s.Name == "DoneChild").Status); - } - - /// - /// Reference is looked up case-insensitively by QueueStatusBridge, so it has to come back with its - /// original casing intact rather than being normalised on the way through storage. - /// - [Fact] - public async Task Reference_PreservesCasing() - { - var store = NewStore(); - await store.InitializeAsync(); - - await store.UpsertRunAsync(new OrchestratorRun - { - Name = "Run", - Reference = "Tenant-ABC_Sync", - Status = "Running", - StartedUtc = DateTime.UtcNow, - }); - - Assert.Equal("Tenant-ABC_Sync", (await store.GetRunAsync("Run"))!.Reference); - } -} diff --git a/tests/Craft.Tests/OrchestratorSequentialTests.cs b/tests/Craft.Tests/OrchestratorSequentialTests.cs deleted file mode 100644 index 22ac2e7..0000000 --- a/tests/Craft.Tests/OrchestratorSequentialTests.cs +++ /dev/null @@ -1,529 +0,0 @@ -using System.Collections.Concurrent; -using System.Reflection; -using Craft.Configuration; -using Craft.Orchestration; -using Craft.PowerShellHost; -using Craft.Services; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// Pins SEQUENTIAL orchestrator mode. -/// -/// A run marked runs its tasks ONE AT A TIME, in ascending -/// (payload) order, ON A SINGLE PINNED WORKER: one entry row -/// is dispatched, that one claim drives the WHOLE run inline (checkout one worker → run every step on it → -/// reclaim once), and the not-yet-reached steps never get their own queue row. The default (false) is the -/// existing fan-out: every task enqueued up front and drained in parallel by the pool. -/// -/// These fix the contract at every seam it touches: -/// - persistence — the flag and the per-task order survive a round trip AND a status-write rewrite -/// (Replace mode erases any column the write omits), so a resumed run keeps its order; -/// - dispatch — a sequential run enqueues only its single entry row; fan-out enqueues all; -/// - driver — every step runs, in Sequence order, on ONE worker; a failing step does not strand the -/// rest (best-effort); a cancelled run marks the remaining steps Cancelled; a duplicate -/// entry row is a no-op while a driver is already active; -/// - re-drive — the watchdog leaves a run alone while its driver is active (or its entry job is still -/// queued/running), yet still restarts the driver if the entry row is lost. -/// -/// The driver's loop logic is exercised through , a subclass that overrides the -/// three PowerShell seams (checkout / run-step / reclaim), so these tests need no worker pool. The actual -/// pinning to a live worker is proven separately by live validation against the dev backend. -/// -public class OrchestratorSequentialTests -{ - private const string TaskFunc = "Invoke-CraftTask"; - - // ─── seam subclass ────────────────────────────────────────────────────────────────────────────── - // Overrides only the three PowerShell interactions. Everything else — ordering, best-effort, marker - // writes, completion, cancellation — runs the real OrchestratorService code. - - private sealed class SeqDriver : OrchestratorService - { - // OrchestratorService has only a parameterized constructor; a subclass must chain to it to compile. - // Never actually invoked — the harness builds this via GetUninitializedObject — so the nulls are safe. - private SeqDriver() : base(null!, null!, null!, null!, null!, null!, null!, null!) { } - - public List Order = new(); // step ids in the order the driver ran them - public HashSet FailIds = new(); // ids whose step throws (best-effort test) - public int Checkouts; // must be exactly 1 for a run that starts - public int Reclaims; // must pair 1:1 with Checkouts - public string StepOutput = string.Empty; // what a step "returns" (PostExecution capture test) - public Action? OnStep; // test hook, runs synchronously inside a step - - internal override PowerShellWorker? CheckoutSequentialWorker(CancellationToken ct) - { - Checkouts++; - return null; // step execution is faked, so the worker handle is never dereferenced - } - - internal override void ReclaimSequentialWorker(PowerShellWorker? worker, bool faulted) => Reclaims++; - - internal override Task RunSequentialStepAsync( - OrchestratorRun run, OrchestratorTaskItem task, string taskPath, PowerShellWorker? worker) - { - Order.Add(task.Id); - OnStep?.Invoke(task); - if (FailIds.Contains(task.Id)) - throw new InvalidOperationException("boom " + task.Id); - return Task.FromResult(StepOutput); - } - } - - // ─── harness ──────────────────────────────────────────────────────────────────────────────────── - - private sealed record Harness(SeqDriver Svc, OrchestratorTableStore Store, JobQueueStore Queue, JobManager Jobs); - - private static async Task NewHarnessAsync() - { - var settings = new CraftSettings - { - Orchestrator = { TablePrefix = "seq" + Guid.NewGuid().ToString("N")[..8] } - }; - // Direct, synchronous status writes — no batching barrier or drain loop to stand up in a unit test. - settings.Orchestrator.BatchStatusWrites = false; - - var backing = new RunRemainingCounterTests.ConditionalStore(); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); - var queue = new JobQueueStore(NullLogger.Instance, settings, backing); - await store.InitializeAsync(); - await queue.InitializeAsync(); - var writer = new OrchestratorStatusWriter(store, NullLogger.Instance, settings); - - // Build the service without its constructor (which drags in the PowerShell runner and the worker - // pool) and set only the fields the dispatch / driver / re-drive paths read — the same shape the - // other orchestrator unit tests use. - var svc = (SeqDriver)System.Runtime.CompilerServices.RuntimeHelpers.GetUninitializedObject(typeof(SeqDriver)); - svc.Order = new(); - svc.FailIds = new(); - Set(svc, "_logger", NullLogger.Instance); - Set(svc, "_store", store); - Set(svc, "_queue", queue); - Set(svc, "_writer", writer); - Set(svc, "_lock", new object()); - Set(svc, "_activeRuns", new ConcurrentDictionary()); - Set(svc, "_taskScriptPaths", new ConcurrentDictionary()); - Set(svc, "_finalizingRuns", new ConcurrentDictionary()); - Set(svc, "_finalizeDeferrals", new ConcurrentDictionary()); - Set(svc, "_childRuns", new ConcurrentDictionary>()); - Set(svc, "_cancelledRuns", new ConcurrentDictionary()); - Set(svc, "_activeSequentialDrivers", new ConcurrentDictionary()); - Set(svc, "_requeueFailures", new ConcurrentDictionary()); - Set(svc, "_deferrals", NewFieldDict(svc, "_deferrals")); - Set(svc, "_redriveBackoff", NewFieldDict(svc, "_redriveBackoff")); - Set(svc, "_redriveInFlight", NewFieldDict(svc, "_redriveInFlight")); - Set(svc, "_redriveSlots", new SemaphoreSlim(8, 8)); - Set(svc, "_shedParameters", false); - Set(svc, "_redriveBackoffEnabled", false); // pin the sequential logic, not the backoff timing - Set(svc, "_redriveBase", TimeSpan.FromSeconds(60)); - Set(svc, "_settings", settings); - - // A JobManager with an empty job map, so IsQueuedOrRunning answers false (nothing dispatched here). - var jm = (JobManager)System.Runtime.CompilerServices.RuntimeHelpers.GetUninitializedObject(typeof(JobManager)); - var jobsField = typeof(JobManager).GetField("_jobs", BindingFlags.NonPublic | BindingFlags.Instance)!; - jobsField.SetValue(jm, Activator.CreateInstance(jobsField.FieldType)); - Set(svc, "_jobManager", jm); - - return new Harness(svc, store, queue, jm); - } - - private static object NewFieldDict(object svc, string field) - { - var t = typeof(OrchestratorService).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)!.FieldType; - return Activator.CreateInstance(t)!; - } - - private static void Set(object target, string field, object? value) => - typeof(OrchestratorService).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)! - .SetValue(target, value); - - private static T Get(object target, string field) => - (T)typeof(OrchestratorService).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)! - .GetValue(target)!; - - private static Task Invoke(OrchestratorService svc, string method, params object[] args) => - (Task)typeof(OrchestratorService).GetMethod(method, BindingFlags.NonPublic | BindingFlags.Instance)! - .Invoke(svc, args)!; - - private static Task Dispatch(Harness h, OrchestratorRun run) => - Invoke(h.Svc, "DispatchPendingTasksAsync", run, TaskFunc, run.Priority, CancellationToken.None, false); - private static Task Redrive(Harness h, OrchestratorRun run) => Invoke(h.Svc, "RedrivePendingTasksAsync", run); - - /// Run the sequential driver to completion. Pre-seeds _finalizingRuns so the background - /// finalize the last step schedules cannot mutate the run under the test's assertions. - private static async Task DriveAsync(Harness h, OrchestratorRun run, string taskPath = TaskFunc, - CancellationToken ct = default) - { - Get>(h.Svc, "_finalizingRuns")[run.Name] = true; - var mi = typeof(OrchestratorService).GetMethod("BuildSequentialRunWork", - BindingFlags.NonPublic | BindingFlags.Instance)!; - var work = (Func)mi.Invoke(h.Svc, [run, taskPath])!; - try { await work(ct); } - catch (TargetInvocationException ex) { throw ex.InnerException ?? ex; } - } - - private static OrchestratorRun MakeRun(string name, int count, bool sequential) - { - var run = new OrchestratorRun - { - Name = name, - Status = "Running", - Priority = 4, - Sequential = sequential, - StartedUtc = DateTime.UtcNow, - TaskScriptName = TaskFunc - }; - for (var i = 0; i < count; i++) - run.Tasks.Add(new OrchestratorTaskItem - { - Id = $"{name}_t{i}", - Status = "Pending", - Sequence = i, - Parameters = new Dictionary { ["FunctionName"] = "Push-Noop", ["idx"] = i } - }); - return run; - } - - private static async Task> QueuedAsync(Harness h, string run) => - (await h.Queue.GetQueuedTaskIdsAsync(run)).OrderBy(x => x, StringComparer.Ordinal).ToList(); - - /// Re-drive re-enqueues via a fire-and-forget Task.Run, so poll for the row to land. - private static async Task> WaitQueuedAsync(Harness h, string run, int expected) - { - for (var i = 0; i < 100; i++) - { - var ids = await h.Queue.GetQueuedTaskIdsAsync(run); - if (ids.Count >= expected) return ids.OrderBy(x => x, StringComparer.Ordinal).ToList(); - await Task.Delay(20); - } - return await QueuedAsync(h, run); - } - - private static string St(OrchestratorRun run, int i) => run.Tasks.Single(t => t.Sequence == i).Status; - - // ─── persistence ──────────────────────────────────────────────────────────────────────────────── - - [Fact] - public async Task Sequential_And_Sequence_SurviveCreationRoundTrip() - { - var h = await NewHarnessAsync(); - var run = MakeRun("persist-seq", 3, sequential: true); - await h.Store.UpsertRunAsync(run); - await h.Store.UpsertTaskBatchAsync(run.Name, run.Tasks); - - var loaded = await h.Store.GetRunAsync(run.Name); - - Assert.NotNull(loaded); - Assert.True(loaded!.Sequential); - Assert.Equal([0, 1, 2], - loaded.Tasks.OrderBy(t => t.Sequence).Select(t => t.Sequence).ToArray()); - } - - [Fact] - public async Task FanOutRun_RoundTripsSequentialFalse() - { - var h = await NewHarnessAsync(); - var run = MakeRun("persist-fanout", 2, sequential: false); - await h.Store.UpsertRunAsync(run); - await h.Store.UpsertTaskBatchAsync(run.Name, run.Tasks); - - var loaded = await h.Store.GetRunAsync(run.Name); - - Assert.False(loaded!.Sequential); - } - - [Fact] - public async Task Sequence_SurvivesStatusWriteRewrite() - { - // The status writer rewrites a task row on every transition with Replace semantics, so Sequence - // must be part of that write or a resumed run would read every task as Sequence 0 and lose order. - var h = await NewHarnessAsync(); - var run = MakeRun("persist-statuswrite", 2, sequential: true); - await h.Store.UpsertRunAsync(run); - await h.Store.UpsertTaskBatchAsync(run.Name, run.Tasks); - - // Simulate a terminal status write (as the coalescing writer does) for the second task. - var t1 = run.Tasks[1]; - await h.Store.WriteTaskStatusBatchAsync( - [ - new TaskStatusWrite(run.Name, t1.Id, "Completed", "{}", 0, null, DateTime.UtcNow, null, t1.Sequence) - ]); - - var loaded = await h.Store.GetRunAsync(run.Name); - var reloaded = loaded!.Tasks.Single(t => t.Id == t1.Id); - Assert.Equal(1, reloaded.Sequence); // not erased to 0 by the Replace write - Assert.Equal("Completed", reloaded.Status); - } - - // ─── dispatch gate ────────────────────────────────────────────────────────────────────────────── - - [Fact] - public async Task Sequential_Dispatch_EnqueuesOnlyTheEntryRow() - { - var h = await NewHarnessAsync(); - var run = MakeRun("disp-seq", 5, sequential: true); - - await Dispatch(h, run); - - var queued = await QueuedAsync(h, run.Name); - Assert.Equal(["disp-seq_t0"], queued); // only Sequence 0, despite five pending tasks - } - - [Fact] - public async Task FanOut_Dispatch_EnqueuesEveryTask() - { - var h = await NewHarnessAsync(); - var run = MakeRun("disp-fanout", 5, sequential: false); - - await Dispatch(h, run); - - var queued = await QueuedAsync(h, run.Name); - Assert.Equal(5, queued.Count); - } - - [Fact] - public async Task Sequential_Dispatch_OnResume_EnqueuesEntryOnly_NeverASecondRow() - { - // Resume shape: tasks 0,1 done; task 2 is the reached task and its queue row SURVIVED the restart; - // task 3 has not been reached. Dispatch must not enqueue task 3 (a second row would let a second - // driver start) — it must leave the already-queued entry alone. - var h = await NewHarnessAsync(); - var run = MakeRun("disp-resume", 4, sequential: true); - run.Tasks[0].Status = "Completed"; - run.Tasks[1].Status = "Completed"; - await h.Queue.EnqueueBatchAsync(run.Name, [(run.Tasks[2].Id, 4)], DateTime.UtcNow); - - await Dispatch(h, run); - - var queued = await QueuedAsync(h, run.Name); - Assert.Equal(["disp-resume_t2"], queued); // task 3 NOT enqueued - } - - [Fact] - public async Task Sequential_Dispatch_OnResume_ReEnqueuesCurrentStep_WhenItsRowIsGone() - { - // The reached step's queue row was lost. Dispatch re-enqueues exactly it, to restart the driver. - var h = await NewHarnessAsync(); - var run = MakeRun("disp-resume-gone", 3, sequential: true); - run.Tasks[0].Status = "Completed"; - - await Dispatch(h, run); - - var queued = await QueuedAsync(h, run.Name); - Assert.Equal(["disp-resume-gone_t1"], queued); - } - - // ─── driver ───────────────────────────────────────────────────────────────────────────────────── - - [Fact] - public async Task Driver_RunsEveryStepInSequenceOrder_OnExactlyOneWorker() - { - var h = await NewHarnessAsync(); - var run = MakeRun("drv-order", 4, sequential: true); - run.Tasks.Reverse(); // insertion order 3,2,1,0 — only the Sequence sort can recover 0,1,2,3 - - await DriveAsync(h, run); - - Assert.Equal(["drv-order_t0", "drv-order_t1", "drv-order_t2", "drv-order_t3"], h.Svc.Order); - Assert.All(run.Tasks, t => Assert.Equal("Completed", t.Status)); - Assert.Equal(1, h.Svc.Checkouts); // one worker for the whole run - Assert.Equal(1, h.Svc.Reclaims); // reclaimed once, at the end - } - - [Fact] - public async Task Driver_BestEffort_ContinuesPastAFailingStep() - { - var h = await NewHarnessAsync(); - var run = MakeRun("drv-besteffort", 4, sequential: true); - h.Svc.FailIds.Add("drv-besteffort_t1"); // the second step throws - - await DriveAsync(h, run); - - // Every step is still attempted, in order, and the failure does not strand the rest. - Assert.Equal(["drv-besteffort_t0", "drv-besteffort_t1", "drv-besteffort_t2", "drv-besteffort_t3"], - h.Svc.Order); - Assert.Equal("Completed", St(run, 0)); - Assert.Equal("Failed", St(run, 1)); - Assert.Equal("Completed", St(run, 2)); - Assert.Equal("Completed", St(run, 3)); - Assert.Equal(1, h.Svc.Checkouts); // still one worker — a step failure does not re-grab a worker - Assert.Equal(1, h.Svc.Reclaims); - } - - [Fact] - public async Task Driver_Cancellation_MarksRemainingStepsCancelled_AndStops() - { - var h = await NewHarnessAsync(); - var run = MakeRun("drv-cancel", 4, sequential: true); - var cancelled = Get>(h.Svc, "_cancelledRuns"); - h.Svc.OnStep = t => { if (t.Sequence == 1) cancelled[run.Name] = true; }; // cancel while step 1 runs - - await DriveAsync(h, run); - - // Steps 0 and 1 completed; the driver noticed the cancellation before step 2 and stopped there. - Assert.Equal(["drv-cancel_t0", "drv-cancel_t1"], h.Svc.Order); - Assert.Equal("Completed", St(run, 0)); - Assert.Equal("Completed", St(run, 1)); - Assert.Equal("Cancelled", St(run, 2)); - Assert.Equal("Cancelled", St(run, 3)); - Assert.Equal(1, h.Svc.Reclaims); // the pinned worker is still reclaimed on the way out - } - - [Fact] - public async Task Driver_DuplicateEntry_IsANoOp_WhileAnotherDriverIsActive() - { - // A duplicate entry row for a run that already has an active driver must not start a second one. - var h = await NewHarnessAsync(); - var run = MakeRun("drv-dup", 3, sequential: true); - Get>(h.Svc, "_activeSequentialDrivers")[run.Name] = true; - - await DriveAsync(h, run); - - Assert.Empty(h.Svc.Order); // ran nothing - Assert.Equal(0, h.Svc.Checkouts); // never grabbed a worker - Assert.All(run.Tasks, t => Assert.Equal("Pending", t.Status)); // left for the real driver - } - - [Fact] - public async Task Driver_ClearsItsRegistration_WhenDone() - { - var h = await NewHarnessAsync(); - var run = MakeRun("drv-cleanup", 2, sequential: true); - - await DriveAsync(h, run); - - // The run must not stay registered, or its re-drive would be suppressed forever. - Assert.False(Get>(h.Svc, "_activeSequentialDrivers") - .ContainsKey(run.Name)); - } - - [Fact] - public async Task Driver_PostExecutionRun_CapturesAndStoresStepOutput() - { - var h = await NewHarnessAsync(); - var run = MakeRun("drv-postexec", 2, sequential: true); - run.PostExecFunctionName = "Push-Aggregate"; // capture path - h.Svc.StepOutput = "{\"ok\":true}"; - - await DriveAsync(h, run); - - var results = await h.Store.GetResultsAsync(run.Name); - Assert.Equal(2, results.Length); - Assert.All(results, r => Assert.Contains("ok", r)); - } - - [Fact] - public async Task ResolveTaskWork_ForSequentialRun_ReturnsAndRunsTheDriver() - { - // Routing: a claimed entry row for a sequential run resolves to the pinned driver (not the parallel - // per-task work), and invoking it drives the whole run on one worker. - var h = await NewHarnessAsync(); - var run = MakeRun("resolve-seq", 3, sequential: true); - Get>(h.Svc, "_activeRuns")[run.Name] = run; - Get>(h.Svc, "_taskScriptPaths")[run.Name] = TaskFunc; - Get>(h.Svc, "_finalizingRuns")[run.Name] = true; - - var mi = typeof(OrchestratorService).GetMethod("ResolveTaskWorkAsync", - BindingFlags.NonPublic | BindingFlags.Instance)!; - var task = (Task)mi.Invoke(h.Svc, [new JobDescriptor(run.Name, run.Tasks[0].Id, 4), CancellationToken.None])!; - await task; - var work = (Func?)task.GetType().GetProperty("Result")!.GetValue(task); - - Assert.NotNull(work); - await work!(CancellationToken.None); - - Assert.Equal(["resolve-seq_t0", "resolve-seq_t1", "resolve-seq_t2"], h.Svc.Order); - Assert.Equal(1, h.Svc.Checkouts); - } - - // ─── re-drive restriction ─────────────────────────────────────────────────────────────────────── - - [Fact] - public async Task Redrive_Sequential_DoesNothing_WhileADriverIsActive() - { - // The driver runs every step inline; the not-yet-reached steps deliberately have no queue row. The - // watchdog must not treat them as orphaned and enqueue them — that would spawn a second driver. - var h = await NewHarnessAsync(); - var run = MakeRun("rd-driver", 3, sequential: true); - Get>(h.Svc, "_activeSequentialDrivers")[run.Name] = true; - - await Redrive(h, run); - await Task.Delay(100); // give any (erroneous) fire-and-forget requeue time to land - - Assert.Empty(await QueuedAsync(h, run.Name)); - } - - [Fact] - public async Task Redrive_Sequential_DoesNothing_WhileTheEntryJobIsStillQueued() - { - // No driver registered yet, but the entry job is still Queued/Running in the JobManager (the window - // between the pump claiming the entry row and the driver registering). The watchdog must still leave - // the run alone. - var h = await NewHarnessAsync(); - var run = MakeRun("rd-entryjob", 3, sequential: true); - var jobs = Get>(h.Jobs, "_jobs"); - jobs[$"{run.Name}-{run.Tasks[0].Id}"] = new JobRecord - { - Id = $"{run.Name}-{run.Tasks[0].Id}", - Name = TaskFunc, - RunName = run.Name, - Priority = 4, - Status = "Running", - QueuedUtc = DateTime.UtcNow - }; - - await Redrive(h, run); - await Task.Delay(100); - - Assert.Empty(await QueuedAsync(h, run.Name)); - } - - private static T Get(JobManager jm, string field) => - (T)typeof(JobManager).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)!.GetValue(jm)!; - - [Fact] - public async Task Redrive_Sequential_NoOp_WhenTheEntryStepStillHasItsRow() - { - // Nothing running, entry step (Sequence 0) is queued and waiting; later steps have no row. The entry - // is dispatchable (row present) so not orphaned, and the later steps must not be touched. - var h = await NewHarnessAsync(); - var run = MakeRun("rd-waiting", 3, sequential: true); - await h.Queue.EnqueueBatchAsync(run.Name, [(run.Tasks[0].Id, 4)], DateTime.UtcNow); - - await Redrive(h, run); - await Task.Delay(100); - - Assert.Equal(["rd-waiting_t0"], await QueuedAsync(h, run.Name)); // unchanged; t1/t2 not enqueued - } - - [Fact] - public async Task Redrive_Sequential_ReDrivesOnlyTheCurrentStep_WhenStalled() - { - // Driver gone and the entry row lost: current step (Sequence 0) has no queue row and no driver is - // active. The watchdog re-enqueues exactly the current step — and none of the not-yet-reached ones — - // so a fresh driver resumes the run. - var h = await NewHarnessAsync(); - var run = MakeRun("rd-stalled", 3, sequential: true); - - await Redrive(h, run); - - var queued = await WaitQueuedAsync(h, run.Name, 1); - Assert.Equal(["rd-stalled_t0"], queued); // only the current; t1/t2 stay unqueued - } - - [Fact] - public async Task Redrive_FanOut_ReEnqueuesAllOrphanedTasks() - { - // The default fan-out behaviour is unchanged: every orphaned (Pending, no queue row) task is - // re-driven, not just the first. - var h = await NewHarnessAsync(); - var run = MakeRun("rd-fanout", 3, sequential: false); - - await Redrive(h, run); - - var queued = await WaitQueuedAsync(h, run.Name, 3); - Assert.Equal(3, queued.Count); - } -} diff --git a/tests/Craft.Tests/OrchestratorStaleRunningTests.cs b/tests/Craft.Tests/OrchestratorStaleRunningTests.cs deleted file mode 100644 index bfddb21..0000000 --- a/tests/Craft.Tests/OrchestratorStaleRunningTests.cs +++ /dev/null @@ -1,209 +0,0 @@ -using System.Collections.Concurrent; -using System.Reflection; -using Craft.Configuration; -using Craft.Orchestration; -using Craft.PowerShellHost; -using Craft.Services; -using Craft.Storage; -using Microsoft.Extensions.Configuration; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// A "Running" status this process did not write must not be trusted as "a worker here has it". -/// -/// The resolver drops a descriptor whose task reads Running, which is right for a duplicate queue row -/// claimed while THIS process is executing the task. But Running also reaches the live graph from -/// storage: the durable pre-invoke marker another process wrote before it died. Two production paths -/// put it there, both seen on a hosted instance across an App Service container swap (old and new -/// containers overlap for tens of seconds to minutes, sharing one storage account): -/// -/// A. The old container creates or advances a run after the new container's startup recovery has -/// already passed. It dies holding claims whose tasks are marked Running. The new container's pump -/// claims one of the run's rows, ResolveTaskWorkAsync rehydrates the run from storage with -/// those markers, and once the dead claims' leases lapse and those rows are claimed, the resolver -/// drops each one as "already running" — Skipped, row deleted. Nothing ever re-drives a Running -/// task, and the run can never finalize (observed: 505/508 for 31 hours). -/// -/// B. The pump starts claiming at host start, before ResumeInterruptedRunsAsync (which waits -/// for the worker pool). Its rehydrated copy goes into _activeRuns first; recovery then -/// loads its OWN copy, flips Running→Pending on that copy and in storage, releases the dead claims -/// and calls DispatchPendingTasksAsync, whose _activeRuns.TryAdd loses to the pump's copy. -/// The live graph keeps the stale Running; the released row is claimed and dropped exactly as in A. -/// -/// Holding the queue claim is the ownership proof: only this process's pump enqueues descriptor jobs, -/// and it only does so for a row it has claimed. So a Running task that no worker in this process -/// started is an interrupted task, and is handled the way recovery handles one — attempt counted -/// (poison bound kept) and run. -/// -public class OrchestratorStaleRunningTests -{ - private const string TaskFunc = "Invoke-CraftTask"; - - private sealed record Harness(OrchestratorService Svc, OrchestratorTableStore Store, JobQueueStore Queue, - ConcurrentDictionary Active); - - private static async Task NewHarnessAsync() - { - var settings = new CraftSettings { Orchestrator = { TablePrefix = "stale" + Guid.NewGuid().ToString("N")[..8] } }; - settings.Orchestrator.BatchStatusWrites = false; - var config = new ConfigurationBuilder().AddInMemoryCollection([]).Build(); - - var backing = new RunRemainingCounterTests.ConditionalStore(); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); - var queue = new JobQueueStore(NullLogger.Instance, settings, backing); - await store.InitializeAsync(); - await queue.InitializeAsync(); - var writer = new OrchestratorStatusWriter(store, NullLogger.Instance, settings); - - var repo = new ScriptRepository(NullLogger.Instance, settings); - typeof(ScriptRepository).GetField("_moduleFunctionNames", BindingFlags.NonPublic | BindingFlags.Instance)! - .SetValue(repo, new HashSet([TaskFunc], StringComparer.OrdinalIgnoreCase)); - var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); - var runner = new PowerShellRunnerService(NullLogger.Instance, pool, repo, settings); - - var svc = (OrchestratorService)System.Runtime.CompilerServices.RuntimeHelpers - .GetUninitializedObject(typeof(OrchestratorService)); - var active = new ConcurrentDictionary(); - Set(svc, "_logger", NullLogger.Instance); - Set(svc, "_store", store); - Set(svc, "_queue", queue); - Set(svc, "_writer", writer); - Set(svc, "_psRunner", runner); - Set(svc, "_settings", settings); - Set(svc, "_lock", new object()); - Set(svc, "_activeRuns", active); - Set(svc, "_taskScriptPaths", new ConcurrentDictionary()); - Set(svc, "_finalizingRuns", new ConcurrentDictionary()); - Set(svc, "_finalizeDeferrals", new ConcurrentDictionary()); - Set(svc, "_childRuns", new ConcurrentDictionary>()); - Set(svc, "_recoveringChildren", new ConcurrentDictionary()); - Set(svc, "_pendingChildRuns", new ConcurrentDictionary()); - Set(svc, "_cancelledRuns", new ConcurrentDictionary()); - Set(svc, "_activeSequentialDrivers", new ConcurrentDictionary()); - Set(svc, "_requeueFailures", new ConcurrentDictionary()); - Set(svc, "_deferrals", NewFieldValue(svc, "_deferrals")); - Set(svc, "_redriveBackoff", NewFieldValue(svc, "_redriveBackoff")); - Set(svc, "_redriveInFlight", NewFieldValue(svc, "_redriveInFlight")); - Set(svc, "_redriveSlots", new SemaphoreSlim(8, 8)); - Set(svc, "_shedParameters", false); - var jm = (JobManager)System.Runtime.CompilerServices.RuntimeHelpers.GetUninitializedObject(typeof(JobManager)); - var jobsField = typeof(JobManager).GetField("_jobs", BindingFlags.NonPublic | BindingFlags.Instance)!; - jobsField.SetValue(jm, Activator.CreateInstance(jobsField.FieldType)); - Set(svc, "_jobManager", jm); - return new Harness(svc, store, queue, active); - } - - private static object NewFieldValue(object svc, string field) => - Activator.CreateInstance(typeof(OrchestratorService) - .GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)!.FieldType)!; - - private static void Set(object target, string field, object? value) => - typeof(OrchestratorService).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)! - .SetValue(target, value); - - private static async Task ResolveAsync(OrchestratorService svc, string run, string taskId) - { - var mi = typeof(OrchestratorService).GetMethod("ResolveTaskWorkAsync", BindingFlags.NonPublic | BindingFlags.Instance)!; - try - { - var task = (Task)mi.Invoke(svc, [new JobDescriptor(run, taskId, 4), CancellationToken.None])!; - await task; - return task.GetType().GetProperty("Result")!.GetValue(task); - } - catch (TargetInvocationException ex) { throw ex.InnerException ?? ex; } - } - - /// Storage as another process left it: its in-flight task carries the durable Running marker. - private static async Task SeedAsync(Harness h, string name, int staleAttempts = 0) - { - var tasks = new List - { - new() { Id = "t-stale", Status = "Running", AttemptCount = staleAttempts, Parameters = new() { ["FunctionName"] = "Push-Noop" } }, - new() { Id = "t-next", Status = "Pending", Parameters = new() { ["FunctionName"] = "Push-Noop" } }, - new() { Id = "t-other", Status = "Pending", Parameters = new() { ["FunctionName"] = "Push-Noop" } }, - }; - await h.Store.UpsertRunAsync(new OrchestratorRun - { - Name = name, - Status = "Running", - Priority = 4, - StartedUtc = DateTime.UtcNow, - TaskScriptName = TaskFunc, - Tasks = tasks - }); - foreach (var t in tasks) await h.Store.UpsertTaskAsync(name, t); - } - - // ── Mode A ───────────────────────────────────────────────────────────────────────────────────── - - [Fact] - public async Task ModeA_RunningMarkerFromADeadProcess_IsRunWhenThisProcessClaimsItsRow() - { - var h = await NewHarnessAsync(); - await SeedAsync(h, "run-a"); - - // The dead process's lease lapsed; this process claimed the row. Previously: null (dropped forever). - var work = await ResolveAsync(h.Svc, "run-a", "t-stale"); - - Assert.NotNull(work); - var task = h.Active["run-a"].Tasks.Single(t => t.Id == "t-stale"); - Assert.Equal(1, task.AttemptCount); // the interrupted attempt is counted, as recovery counts it - } - - [Fact] - public async Task ModeA_PoisonBoundHolds_AThirdInterruptedAttemptFailsTheTask() - { - var h = await NewHarnessAsync(); - await SeedAsync(h, "run-poison", staleAttempts: 2); - - Assert.Null(await ResolveAsync(h.Svc, "run-poison", "t-stale")); - - var task = h.Active["run-poison"].Tasks.Single(t => t.Id == "t-stale"); - Assert.Equal("Failed", task.Status); - } - - // ── Mode B ───────────────────────────────────────────────────────────────────────────────────── - - [Fact] - public async Task ModeB_PumpClaimBeforeRecovery_LeavesAStaleRunningInTheLiveGraph_WhichMustStillRun() - { - var h = await NewHarnessAsync(); - await SeedAsync(h, "run-b"); - - // 1. Host start: the pump claims a sibling row before recovery has run. The rehydrated copy — - // carrying the dead process's Running marker — becomes the live graph. - Assert.NotNull(await ResolveAsync(h.Svc, "run-b", "t-next")); - var live = h.Active["run-b"]; - - // 2. Recovery runs on its own copy: flips t-stale to Pending in storage, releases the claims and - // re-dispatches — but its TryAdd loses to the pump's copy. - await h.Svc.ResumeInterruptedRunsAsync(CancellationToken.None); - Assert.Same(live, h.Active["run-b"]); - - // 3. The released row is claimed. Previously: dropped as "already running", never run again. - Assert.NotNull(await ResolveAsync(h.Svc, "run-b", "t-stale")); - } - - // ── The guard's real job is kept ─────────────────────────────────────────────────────────────── - - [Fact] - public async Task ADuplicateRowForATaskThisProcessIsExecuting_IsStillDropped() - { - var h = await NewHarnessAsync(); - await SeedAsync(h, "run-dup"); - await h.Store.UpsertTaskAsync("run-dup", - new OrchestratorTaskItem { Id = "t-stale", Status = "Pending", Parameters = new() { ["FunctionName"] = "Push-Noop" } }); - - // Start the task the way dispatch does, up to the point it is running on a worker here. - var work = (Func)(await ResolveAsync(h.Svc, "run-dup", "t-stale"))!; - var task = h.Active["run-dup"].Tasks.Single(t => t.Id == "t-stale"); - _ = Task.Run(() => work(CancellationToken.None)); // marks Running, then blocks on worker checkout - for (var i = 0; i < 200 && task.Status != "Running"; i++) await Task.Delay(10); - Assert.Equal("Running", task.Status); - - // A second row for the same task, claimed mid-flight, must not start another copy. - Assert.Null(await ResolveAsync(h.Svc, "run-dup", "t-stale")); - } -} diff --git a/tests/Craft.Tests/OrchestratorTaskLargeParametersAzuriteTests.cs b/tests/Craft.Tests/OrchestratorTaskLargeParametersAzuriteTests.cs index b44a86c..37fd671 100644 --- a/tests/Craft.Tests/OrchestratorTaskLargeParametersAzuriteTests.cs +++ b/tests/Craft.Tests/OrchestratorTaskLargeParametersAzuriteTests.cs @@ -1,19 +1,15 @@ using System.Text.Json; using Craft.Configuration; -using Craft.Orchestration; using Craft.Storage; using Microsoft.Extensions.Logging.Abstractions; namespace Craft.Tests; /// -/// End-to-end guard for the failure this whole change exists for: a scheduled task whose Parameters -/// embed a whole policy template serialize to more than Azure Table's 64 KiB-per-property limit, so the -/// orchestrator's Tasks row used to 400 with PropertyValueTooLarge, get dropped, and the task could never -/// be rehydrated at dispatch ("Parameters could not be rehydrated at dispatch — the Tasks-table row is -/// missing"). With large-entity splitting in the backing store, the task row persists and both read paths -/// the orchestrator uses — GetRunAsync (partition scan) and GetTaskParametersAsync (point read, the -/// dispatch rehydrate) — return the Parameters byte-for-byte. +/// End-to-end guard against a real table backend: a scheduled task whose parameters embed a whole policy +/// template is far over Azure Table's 64 KiB-per-property limit, and so can be a run's PostExecution +/// parameters. The payload row must persist and rehydrate byte-for-byte, and the run must still move through +/// claim, finish and its barrier: those are real entity-group transactions here, which cannot split a row. /// /// Azurite by default, a real account via CRAFT_TEST_TABLE_CONNECTION; skipped, not failed, when neither /// is reachable. @@ -21,7 +17,7 @@ namespace Craft.Tests; [Collection(LargeAllocationSerialTests.Name)] public class OrchestratorTaskLargeParametersAzuriteTests { - private static async Task TryConnectAsync() + private static async Task TryConnectAsync() { var settings = new CraftSettings(); var connection = Environment.GetEnvironmentVariable("CRAFT_TEST_TABLE_CONNECTION"); @@ -43,7 +39,7 @@ public class OrchestratorTaskLargeParametersAzuriteTests return null; } - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); + var store = new WorkStore(NullLogger.Instance, settings, backing); await store.InitializeAsync(); return store; } @@ -56,54 +52,49 @@ public class OrchestratorTaskLargeParametersAzuriteTests }; [Theory] - [InlineData(80_000)] // > 64 KiB per-property: the exact reported failure (column split) + [InlineData(80_000)] // > 64 KiB per-property: column split [InlineData(1_200_000)] // > 1 MiB entity: forces a cross-row split too - public async Task ATaskWhoseParametersExceedTheLimit_PersistsAndRehydrates(int settingsChars) + public async Task ARunWithParametersOverTheLimit_PersistsRehydratesAndCompletes(int settingsChars) { var store = await TryConnectAsync(); if (store == null) return; - const string run = "UserTaskOrchestrator_contoso.com"; var big = new string('T', settingsChars); - var parameters = new Dictionary + var started = DateTime.UtcNow; + var header = new RunHeader { - ["Tenant"] = "contoso.com", - ["Settings"] = big, + RunKey = WorkStore.RunKeyFor("UserTaskOrchestrator_contoso.com", started), + Name = "UserTaskOrchestrator_contoso.com", + Priority = 2, + StartedUtc = started, + TaskScriptName = "Invoke-CraftTask", + PostExecFunctionName = "Agg", + PostExecParametersJson = JsonSerializer.Serialize(new { blob = big }), }; try { - await store.UpsertRunAsync(new OrchestratorRun - { - Name = run, - Status = "Running", - Priority = 2, - StartedUtc = DateTime.UtcNow, - TaskScriptName = "ExecScheduledCommand", - Tasks = [new OrchestratorTaskItem { Id = "task-0", Status = "Pending" }] - }); + await store.CreateRunAsync(header, + [new WorkStore.NewTask("task-0", new() { ["Tenant"] = "contoso.com", ["Settings"] = big })]); - // The exact write path Start-UserTasksOrchestrator uses to enqueue a task's payload. - await store.UpsertTaskBatchAsync(run, new List - { - new() { Id = "task-0", Status = "Pending", Parameters = parameters } - }); - - // Dispatch rehydrate: a point read of the one task's Parameters. - var rehydrated = await store.GetTaskParametersAsync(run, "task-0"); - Assert.NotNull(rehydrated); + var claim = Assert.Single(await store.ClaimAsync(header.RunKey, 10, "w", TimeSpan.FromMinutes(5), true)); + var rehydrated = await store.GetPayloadAsync(header.RunKey, claim.Seq); Assert.Equal("contoso.com", AsString(rehydrated!["Tenant"])); Assert.Equal(big, AsString(rehydrated["Settings"])); - // Whole-run read: the partition scan reassembles the same task. - var loaded = await store.GetRunAsync(run); - Assert.NotNull(loaded); - var task = Assert.Single(loaded!.Tasks); - Assert.Equal(big, AsString(task.Parameters["Settings"])); + var barrier = await store.FinishAsync(header.RunKey, [new WorkStore.Finish(claim.Seq, "Completed", Owner: "w")]); + Assert.True(barrier!.ReachedBarrier); + Assert.Equal(header.PostExecParametersJson, await store.GetPostExecParametersAsync(header.RunKey)); + + var aggregate = Assert.Single(await store.ClaimAsync(header.RunKey, 10, "w", TimeSpan.FromMinutes(5), true)); + Assert.Equal(WorkStore.AggregateSeq, aggregate.Seq); + var done = await store.FinishAsync(header.RunKey, [new WorkStore.Finish(aggregate.Seq, "Completed", Owner: "w")]); + Assert.True(done!.Completed); + Assert.Equal("Completed", done.Header.Status); } finally { - await store.CleanupRunAsync(run); + await store.DeleteRunAsync(header.RunKey); } } } diff --git a/tests/Craft.Tests/OrchestratorTaskPathRehydrationTests.cs b/tests/Craft.Tests/OrchestratorTaskPathRehydrationTests.cs deleted file mode 100644 index 0684b09..0000000 --- a/tests/Craft.Tests/OrchestratorTaskPathRehydrationTests.cs +++ /dev/null @@ -1,123 +0,0 @@ -using System.Collections.Concurrent; -using System.Reflection; -using Craft.Configuration; -using Craft.Orchestration; -using Craft.PowerShellHost; -using Craft.Services; -using Craft.Storage; -using Microsoft.Extensions.Configuration; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// The resolver must rebuild a task's script from the run record when the in-memory path cache misses, -/// instead of dropping the task. -/// -/// The cache (_taskScriptPaths) is written only by DispatchPendingTasksAsync. The JobQueuePump, a -/// BackgroundService, starts claiming persisted queue rows at host start — before the scheduler has -/// even waited for the worker pool, let alone run ResumeInterruptedRunsAsync, which is what dispatches -/// (and so caches the path for) a resumed run. In that window every claimed row resolved to a cache -/// miss and the task was dropped: the resolver returned null, the JobManager marked the job Skipped, -/// and the pump deleted the queue row — permanently, for a task still Pending in a run the pump would -/// never see re-queued. -/// -/// The run record persists TaskScriptName, and the ScriptRepository is fully loaded before the pump's -/// first claim, so the path is rebuildable from storage exactly as the resume path rebuilds it. These -/// tests pin that: a cache miss on a live run rehydrates and caches the path; only a run with no -/// resolvable script at all is still dropped. -/// -public class OrchestratorTaskPathRehydrationTests -{ - private static (OrchestratorService Svc, OrchestratorTableStore Store, ConcurrentDictionary Paths) - NewService(params string[] knownModuleFunctions) - { - var settings = new CraftSettings - { - Orchestrator = { TablePrefix = "rehyd" + Guid.NewGuid().ToString("N")[..6] } - }; - var config = new ConfigurationBuilder().AddInMemoryCollection([]).Build(); - - var repo = new ScriptRepository(NullLogger.Instance, settings); - // IsModuleFunction short-circuits on a non-null _moduleFunctionNames, so FindScript resolves - // these names without a ScriptBlock or a module on disk. - typeof(ScriptRepository).GetField("_moduleFunctionNames", BindingFlags.NonPublic | BindingFlags.Instance)! - .SetValue(repo, new HashSet(knownModuleFunctions, StringComparer.OrdinalIgnoreCase)); - - var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); - var runner = new PowerShellRunnerService(NullLogger.Instance, pool, repo, settings); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, - new RunRemainingCounterTests.ConditionalStore()); - var paths = new ConcurrentDictionary(); - - var svc = (OrchestratorService)System.Runtime.CompilerServices.RuntimeHelpers - .GetUninitializedObject(typeof(OrchestratorService)); - Set(svc, "_logger", NullLogger.Instance); - Set(svc, "_store", store); - Set(svc, "_psRunner", runner); - Set(svc, "_activeRuns", new ConcurrentDictionary()); - Set(svc, "_taskScriptPaths", paths); // deliberately EMPTY — the pump-before-resume window - Set(svc, "_lock", new object()); - return (svc, store, paths); - } - - private static void Set(object target, string field, object value) => - typeof(OrchestratorService).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)! - .SetValue(target, value); - - private static async Task ResolveAsync(OrchestratorService svc, string runName, string taskId) - { - var mi = typeof(OrchestratorService).GetMethod("ResolveTaskWorkAsync", - BindingFlags.NonPublic | BindingFlags.Instance)!; - try - { - var task = (Task)mi.Invoke(svc, [new JobDescriptor(runName, taskId, 4), CancellationToken.None])!; - await task; - return task.GetType().GetProperty("Result")!.GetValue(task); - } - catch (TargetInvocationException ex) - { - throw ex.InnerException ?? ex; - } - } - - private static async Task SeedRunAsync(OrchestratorTableStore store, string name, string taskScriptName) - { - await store.InitializeAsync(); - await store.UpsertRunAsync(new OrchestratorRun - { - Name = name, - Status = "Running", - Priority = 4, - StartedUtc = DateTime.UtcNow, - TaskScriptName = taskScriptName, - Tasks = [new OrchestratorTaskItem { Id = "task-0", Status = "Pending" }] - }); - await store.UpsertTaskAsync(name, new OrchestratorTaskItem { Id = "task-0", Status = "Pending" }); - } - - [Fact] - public async Task CacheMissOnALiveRun_RehydratesThePathFromTaskScriptName_AndCachesIt() - { - var (svc, store, paths) = NewService("Invoke-CraftTask"); - await SeedRunAsync(store, "AuditLogProcessV2-contoso.com", "Invoke-CraftTask"); - - var work = await ResolveAsync(svc, "AuditLogProcessV2-contoso.com", "task-0"); - - Assert.NotNull(work); // previously: dropped, because _taskScriptPaths had no entry yet - Assert.Equal("Invoke-CraftTask", paths["AuditLogProcessV2-contoso.com"]); - } - - [Fact] - public async Task ARunWithNoResolvableTaskScript_IsStillDropped() - { - // Empty TaskScriptName, and the naming-convention fallback (Invoke-Task) resolves to nothing - // because the repo knows no such function. This is the only case that should still drop. - var (svc, store, _) = NewService(/* no known functions */); - await SeedRunAsync(store, "mysteryrun", taskScriptName: ""); - - var work = await ResolveAsync(svc, "mysteryrun", "task-0"); - - Assert.Null(work); - } -} diff --git a/tests/Craft.Tests/RunRemainingCounterTests.cs b/tests/Craft.Tests/RunRemainingCounterTests.cs deleted file mode 100644 index 3c22edf..0000000 --- a/tests/Craft.Tests/RunRemainingCounterTests.cs +++ /dev/null @@ -1,353 +0,0 @@ -using Craft.Configuration; -using Craft.Orchestration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// The remaining-task counter that makes run completion a fact in storage rather than a property of one -/// process's memory. -/// -/// Completion is decided today by CheckRunCompletion walking run.Tasks in memory, which is -/// why ResolveTaskWorkAsync documents object identity with the live run graph as a requirement -/// rather than an optimization — a worker handed a deserialized copy would mutate something the graph -/// never sees and the run would never finalize. A counter in storage removes that requirement. -/// -/// Its one hard property is being decremented exactly once per task. The status writer retries runs -/// whose batch failed, so anything that decrements separately from the task's own terminal write gets -/// replayed and double-counts, stranding a run that has actually finished. -/// -public class RunRemainingCounterTests -{ - /// - /// A fake that models the two things the counter depends on: ETag concurrency, and transactions that - /// are all-or-nothing within a partition. - /// - internal sealed class ConditionalStore : ICraftTableStore - { - private readonly System.Collections.Concurrent.ConcurrentDictionary> _tables = new(); - private long _etag; - - public int ConditionalWrites { get; private set; } - public int RejectedWrites { get; private set; } - - /// Set to run just before a conditional write lands — lets a test interleave a competitor. - public Action? OnBeforeConditionalWrite { get; set; } - - /// Set to run at the start of any table query — lets a test model unreachable storage. - public Action? OnBeforeQuery { get; set; } - - /// Awaited at the start of a partition query, with the table name — lets a test hold a read open. - public Func? OnPartitionQuery { get; set; } - - private System.Collections.Concurrent.ConcurrentDictionary<(string, string), StoreRow> Table(string t) => - _tables.GetOrAdd(t, _ => new()); - - private StoreRow Stamp(StoreRow row) => new(row.PartitionKey, row.RowKey) - { - ETag = $"W/\"{Interlocked.Increment(ref _etag)}\"", - Properties = new Dictionary(row.Properties), - }; - - - /// - /// Azure Table returns rows ordered by partition key then row key, and the queue's whole - /// priority scheme rests on that. A fake handing back insertion order would let an ordering bug - /// pass, so model the real contract. - /// - /// Copies, like GetAsync: yielding the stored instances would let a caller that mutates what it - /// read — which claiming does, by design — change storage without ever writing. - /// - private List Ordered(string table) => Table(table).Values - .OrderBy(r => r.PartitionKey, StringComparer.Ordinal) - .ThenBy(r => r.RowKey, StringComparer.Ordinal) - .Select(r => new StoreRow(r.PartitionKey, r.RowKey) - { - ETag = r.ETag, - Properties = new Dictionary(r.Properties), - }) - .ToList(); - - public Task PingAsync(CancellationToken ct = default) => Task.CompletedTask; - public Task EnsureTableAsync(string table, CancellationToken ct = default) { Table(table); return Task.CompletedTask; } - - public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) - { - Table(table)[(row.PartitionKey, row.RowKey)] = Stamp(row); - return Task.CompletedTask; - } - - public Task UpsertBatchAsync(string table, string partitionKey, IReadOnlyList rows, CancellationToken ct = default) - { - foreach (var r in rows) Table(table)[(r.PartitionKey, r.RowKey)] = Stamp(r); - return Task.CompletedTask; - } - - public Task TryReplaceBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) - { - OnBeforeConditionalWrite?.Invoke(); - - var t = Table(table); - // All-or-nothing: verify every guard BEFORE mutating anything. - foreach (var r in rows) - { - if (!t.TryGetValue((r.PartitionKey, r.RowKey), out var cur) || cur.ETag != r.ETag) - { - RejectedWrites++; - return Task.FromResult(false); - } - } - - foreach (var r in rows) t[(r.PartitionKey, r.RowKey)] = Stamp(r); - ConditionalWrites++; - return Task.FromResult(true); - } - - /// Point reads served, by table. - public System.Collections.Concurrent.ConcurrentDictionary Gets { get; } = new(); - - public Task GetAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) - { - Gets.AddOrUpdate(table, 1, (_, n) => n + 1); - // Hand back a copy: a caller mutating what it read must not mutate the store in place. - if (!Table(table).TryGetValue((partitionKey, rowKey), out var r)) return Task.FromResult(null); - return Task.FromResult(new StoreRow(r.PartitionKey, r.RowKey) - { - ETag = r.ETag, - Properties = new Dictionary(r.Properties), - }); - } - - public async IAsyncEnumerable QueryPartitionAsync(string table, string partitionKey, - [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) - { - if (OnPartitionQuery != null) await OnPartitionQuery(table); - foreach (var r in Ordered(table).Where(r => r.PartitionKey == partitionKey)) - { - yield return r; - await Task.Yield(); - } - } - - public async IAsyncEnumerable QueryTableAsync(string table, - [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) - { - OnBeforeQuery?.Invoke(); - foreach (var r in Ordered(table)) { yield return r; await Task.Yield(); } - } - - public Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) - { - Table(table).TryRemove((partitionKey, rowKey), out _); - return Task.CompletedTask; - } - - public Task DeletePartitionAsync(string table, string partitionKey, CancellationToken ct = default) - { - foreach (var k in Table(table).Keys.Where(k => k.Item1 == partitionKey).ToList()) Table(table).TryRemove(k, out _); - return Task.CompletedTask; - } - } - - private const string Run = "StandardsApply"; - - private static (OrchestratorTableStore Store, ConditionalStore Backing) NewStore() - { - var backing = new ConditionalStore(); - var settings = new CraftSettings(); - return (new OrchestratorTableStore(NullLogger.Instance, settings, backing), backing); - } - - private static OrchestratorTaskItem Task_(string id, string status = "Completed") => - new() { Id = id, Status = status, CompletedUtc = DateTime.UtcNow }; - - private static async Task SeededAsync(ConditionalStore backing, int taskCount) - { - var settings = new CraftSettings(); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); - await store.InitializeAsync(); - - var tasks = Enumerable.Range(0, taskCount).Select(i => Task_($"task-{i}", "Pending")).ToList(); - await store.UpsertTaskBatchAsync(Run, tasks); - await store.InitRemainingAsync(Run, taskCount); - return store; - } - - /// - /// The counter shares the task partition so it can ride the same transaction. Nothing that reads - /// tasks may mistake it for one — a phantom task never completes, and the run never finalizes. - /// - [Fact] - public async Task CounterRowIsNotReturnedAsATask() - { - var backing = new ConditionalStore(); - var store = await SeededAsync(backing, 4); - await store.UpsertRunAsync(new OrchestratorRun { Name = Run, Priority = 4 }); - - var run = await store.GetRunAsync(Run); - - Assert.NotNull(run); - Assert.Equal(4, run!.Tasks.Count); - Assert.DoesNotContain(run.Tasks, t => t.Id.Contains("counter", StringComparison.OrdinalIgnoreCase)); - } - - /// - /// The live path. Terminal transitions reach storage through the coalescing status writer in - /// batches, so the counter has to fall by the number of terminal rows in a batch rather than by one - /// per call — routing each completion through the single-task primitive would be three round-trips - /// per task, roughly 22,000 for the run that prompted this work. - /// - /// Non-terminal rows in the same batch (a "Running" marker) must not count. - /// - [Fact] - public async Task BatchedTerminalWritesDecrementByTheirTerminalCount() - { - var backing = new ConditionalStore(); - var store = await SeededAsync(backing, 10); - - var writes = new List - { - new(Run, "task-0", "Completed", null, 1, null, DateTime.UtcNow, null), - new(Run, "task-1", "Failed", null, 1, "boom", DateTime.UtcNow, null), - new(Run, "task-2", "Cancelled", null, 1, null, DateTime.UtcNow, null), - new(Run, "task-3", "Running", null, 1, null, null, null), // not terminal — must not count - }; - - var failed = await store.WriteTaskStatusBatchAsync(writes); - - Assert.Empty(failed); - Assert.Equal(7, await store.GetRemainingAsync(Run)); - } - - [Fact] - public async Task MissingCounterReportsNullRatherThanGuessing() - { - var (store, _) = NewStore(); - await store.InitializeAsync(); - - Assert.Null(await store.GetRemainingAsync("never-seeded")); - Assert.Null(await store.DecrementRemainingAsync("never-seeded", 1)); - } - - // ─── Reconciliation (the lost-decrement repair) ─── - - /// - /// A decrement that exhausts its retries is never re-sent — the terminal rows landed, the counter - /// didn't move, and from then on it overstates the run's outstanding work forever. Production - /// symptom: "complete in memory but storage shows N outstanding - deferring finalize" on every 60s - /// tick for the life of the process. Reconcile recounts the rows and repairs the counter. - /// - [Fact] - public async Task ReconcileRepairsALostDecrement() - { - var backing = new ConditionalStore(); - var store = await SeededAsync(backing, 3); - - // Terminal rows written WITHOUT the counter moving — exactly what a lost decrement leaves. - await store.UpsertTaskAsync(Run, Task_("task-0")); - await store.UpsertTaskAsync(Run, Task_("task-1", "Failed")); - Assert.Equal(3, await store.GetRemainingAsync(Run)); - - Assert.Equal(1, await store.ReconcileRemainingAsync(Run)); - Assert.Equal(1, await store.GetRemainingAsync(Run)); - } - - [Fact] - public async Task ReconcileWithoutDriftChangesNothing() - { - var backing = new ConditionalStore(); - var store = await SeededAsync(backing, 2); - - var writesBefore = backing.ConditionalWrites; - - Assert.Equal(2, await store.ReconcileRemainingAsync(Run)); - Assert.Equal(writesBefore, backing.ConditionalWrites); - } - - [Fact] - public async Task ReconcileWithoutACounterRowIsANoOp() - { - var (store, _) = NewStore(); - await store.InitializeAsync(); - - Assert.Null(await store.ReconcileRemainingAsync("never-seeded")); - } - - /// - /// A decrement landing between reconcile's recount and its write must win. The recount is stale the - /// moment a competitor moves the counter, so the ETag guard has to reject the repair — reporting - /// null sends the caller back around rather than letting an old count overwrite fresh progress. - /// - [Fact] - public async Task ReconcileLosingARaceDoesNotClobberTheCompetitor() - { - var backing = new ConditionalStore(); - var store = await SeededAsync(backing, 3); - - // Manufacture drift so reconcile attempts a write at all. - await store.UpsertTaskAsync(Run, Task_("task-0")); - - backing.OnBeforeConditionalWrite = () => - { - backing.OnBeforeConditionalWrite = null; - store.DecrementRemainingAsync(Run, 1).GetAwaiter().GetResult(); - }; - - Assert.Null(await store.ReconcileRemainingAsync(Run)); - // The competitor's decrement survived: 3 seeded − 1 decremented-by-competitor. - Assert.Equal(2, await store.GetRemainingAsync(Run)); - } - - // ─── Status-guarded cancel (the cancel-a-run write) ─── - - [Fact] - public async Task CancellingAPendingTaskDecrementsTheCounter() - { - var backing = new ConditionalStore(); - var store = await SeededAsync(backing, 3); - await store.UpsertRunAsync(new OrchestratorRun { Name = Run, Priority = 4 }); - - var result = await store.CancelPendingTaskAsync(Run, Task_("task-0", "Cancelled")); - - Assert.True(result.Cancelled); - Assert.Equal(2, await store.GetRemainingAsync(Run)); - Assert.Equal("Cancelled", (await store.GetRunAsync(Run))?.Tasks.Single(t => t.Id == "task-0").Status); - } - - /// - /// The guard that makes cancel safe against dispatch. A task that moved Pending → Running between - /// the caller's read and this write must be left alone: clobbering it would have the task's real - /// completion decrement the counter a second time, and the run would finalize with work - /// outstanding. - /// - [Fact] - public async Task CancellingATaskThatStartedRunningIsRefused() - { - var backing = new ConditionalStore(); - var store = await SeededAsync(backing, 3); - await store.UpsertRunAsync(new OrchestratorRun { Name = Run, Priority = 4 }); - await store.UpsertTaskAsync(Run, Task_("task-0", "Running")); - - var result = await store.CancelPendingTaskAsync(Run, Task_("task-0", "Cancelled")); - - Assert.False(result.Cancelled); - Assert.Equal("Running", result.CurrentStatus); - Assert.Equal(3, await store.GetRemainingAsync(Run)); - Assert.Equal("Running", (await store.GetRunAsync(Run))?.Tasks.Single(t => t.Id == "task-0").Status); - } - - [Fact] - public async Task CancellingOnAPreCounterRunStillWritesTheStatus() - { - var (store, _) = NewStore(); - await store.InitializeAsync(); - await store.UpsertTaskBatchAsync("old-run", [Task_("task-0", "Pending")]); - - var result = await store.CancelPendingTaskAsync("old-run", Task_("task-0", "Cancelled")); - - Assert.True(result.Cancelled); - Assert.Null(await store.GetRemainingAsync("old-run")); - } -} diff --git a/tests/Craft.Tests/StartupClaimGateTests.cs b/tests/Craft.Tests/StartupClaimGateTests.cs deleted file mode 100644 index 7f9a9ec..0000000 --- a/tests/Craft.Tests/StartupClaimGateTests.cs +++ /dev/null @@ -1,140 +0,0 @@ -using System.Reflection; -using System.Runtime.CompilerServices; -using Craft.Configuration; -using Craft.Endpoints; -using Craft.Orchestration; -using Craft.PowerShellHost; -using Craft.Services; -using Craft.Storage; -using Microsoft.Extensions.Configuration; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// The pump claims nothing until startup recovery is done, and every way startup can end opens the gate. -/// -/// A claim taken before recovery rehydrates its run from storage — the previous process's Running markers -/// included — into the live graph. Recovery then resets those markers on its own copy, which loses the -/// _activeRuns race, so the live graph keeps the stale Running (see OrchestratorStaleRunningTests, -/// mode B). Gating the first claim on recovery removes the race; the resolver's ownership guard stays as -/// defense in depth. -/// -public class StartupClaimGateTests -{ - /// An orchestrator carrying only the gate. Field initializers do not run on an uninitialized - /// object, so the gate is installed by hand; every other field stays null. - private static OrchestratorService GatedOrchestrator() - { - var svc = (OrchestratorService)RuntimeHelpers.GetUninitializedObject(typeof(OrchestratorService)); - typeof(OrchestratorService).GetField("_recoveryDone", BindingFlags.NonPublic | BindingFlags.Instance)! - .SetValue(svc, new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously)); - typeof(OrchestratorService).GetField("_logger", BindingFlags.NonPublic | BindingFlags.Instance)! - .SetValue(svc, NullLogger.Instance); - return svc; - } - - private static (JobQueuePump Pump, JobManager Jobs) NewPump(OrchestratorService orchestrator) - { - var settings = new CraftSettings(); - settings.Worker.BgPoolSize = 4; - var config = new ConfigurationBuilder().AddInMemoryCollection(new Dictionary - { - ["JobQueuePollIntervalMs"] = "100", - }).Build(); - - var queue = new JobQueueStore(NullLogger.Instance, settings, new RunRemainingCounterTests.ConditionalStore()); - queue.InitializeAsync().GetAwaiter().GetResult(); - queue.EnqueueBatchAsync("StandardsApply", - Enumerable.Range(0, 20).Select(i => ($"task-{i:D3}", 4)).ToList(), DateTime.UtcNow).GetAwaiter().GetResult(); - - var repo = new ScriptRepository(NullLogger.Instance, settings); - var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); - var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); - var jobs = new JobManager(NullLogger.Instance, settings, limiter); - return (new JobQueuePump(NullLogger.Instance, queue, jobs, config, settings, orchestrator), jobs); - } - - private static async Task WaitUntil(Func condition, int timeoutMs = 5000) - { - var deadline = Environment.TickCount64 + timeoutMs; - while (Environment.TickCount64 < deadline) - { - if (condition()) return true; - await Task.Delay(20); - } - return condition(); - } - - [Fact] - public async Task PumpClaimsNothingBeforeRecovery_AndClaimsNormallyAfter() - { - var orchestrator = GatedOrchestrator(); - var (pump, jobs) = NewPump(orchestrator); - - await pump.StartAsync(CancellationToken.None); - try - { - await Task.Delay(500); // five poll intervals with rows waiting - Assert.Equal(0, jobs.QueuedCount); - - orchestrator.MarkRecoveryDone(); - - Assert.True(await WaitUntil(() => jobs.QueuedCount > 0), "the pump did not claim once recovery was done"); - } - finally - { - await Task.WhenAny(pump.StopAsync(CancellationToken.None), Task.Delay(3000)); - } - } - - private static SchedulerService NewScheduler(OrchestratorService orchestrator, bool poolReady) - { - var settings = new CraftSettings(); - var config = new ConfigurationBuilder().AddInMemoryCollection([]).Build(); - var repo = new ScriptRepository(NullLogger.Instance, settings); - var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); - if (poolReady) pool.Initialize(enableHttp: false, enableBg: false); // signals ready, builds nothing - var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); - var runner = new PowerShellRunnerService(NullLogger.Instance, pool, repo, settings); - var health = new StorageHealthMonitor(new RunRemainingCounterTests.ConditionalStore(), NullLogger.Instance); - return new SchedulerService(NullLogger.Instance, runner, limiter, orchestrator, - new JobManager(NullLogger.Instance, settings, limiter), settings, pool, health, - NativeScheduledTasks.Empty, null!); - } - - [Fact] - public async Task ARecoveryThatThrows_StillOpensTheGate() - { - // The gated orchestrator has no store, so ResumeInterruptedRunsAsync throws on its first line. - var orchestrator = GatedOrchestrator(); - var scheduler = NewScheduler(orchestrator, poolReady: true); - using var cts = new CancellationTokenSource(); - - await scheduler.StartAsync(cts.Token); - try - { - Assert.True(await Task.WhenAny(orchestrator.RecoveryDone, Task.Delay(10_000)) == orchestrator.RecoveryDone, - "a failed recovery left the claim gate shut — the pump would never claim"); - } - finally - { - cts.Cancel(); - try { await scheduler.StopAsync(CancellationToken.None); } catch { /* the gutted orchestrator may fault the loop */ } - } - } - - [Fact] - public async Task ShutdownBeforeTheWorkerPoolIsReady_StillOpensTheGate() - { - var orchestrator = GatedOrchestrator(); - var scheduler = NewScheduler(orchestrator, poolReady: false); - using var cts = new CancellationTokenSource(); - - await scheduler.StartAsync(cts.Token); - cts.Cancel(); - try { await scheduler.StopAsync(CancellationToken.None); } catch { /* cancellation */ } - - Assert.True(await Task.WhenAny(orchestrator.RecoveryDone, Task.Delay(5_000)) == orchestrator.RecoveryDone); - } -} diff --git a/tests/Craft.Tests/StatusWriterDurabilityTests.cs b/tests/Craft.Tests/StatusWriterDurabilityTests.cs deleted file mode 100644 index 3b8b246..0000000 --- a/tests/Craft.Tests/StatusWriterDurabilityTests.cs +++ /dev/null @@ -1,499 +0,0 @@ -using System.Runtime.CompilerServices; -using Craft.Configuration; -using Craft.Orchestration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// The status writer sits on the critical path between a task being dispatched and that task checking -/// out a worker. Production wedged twice in one morning — 101 and 75 minutes — with all 8 limiter slots -/// held by tasks blocked on its barrier, every one of the 8 BG workers idle, 1,919 jobs queued and the -/// heap at 24% of its cap. The worker-health snapshot read Jobs.Running=8 / BgPool.BusyCount=0, which is -/// only reachable if the jobs never got as far as PowerShell. -/// -/// These tests pin the two properties that failure needed: -/// LIVENESS — no wait here is unbounded, and the drain loop cannot die. -/// DURABILITY — nothing bounded is ever dropped. A write that does not persist is retried; a task that -/// cannot be marked Running is deferred, never failed, and stays Pending in storage. -/// -public class StatusWriterDurabilityTests -{ - /// An in-memory store whose writes can be made to hang or fail on demand. - private sealed class ControllableStore : ICraftTableStore - { - - // Claims are not exercised by this fake. Fail loudly rather than pretend the guard held — - // a silent 'true' here would look exactly like a successful claim. - public Task TryReplaceBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) => throw new NotSupportedException(); - private readonly Dictionary> _tables = new(); - private readonly object _sync = new(); - - public ManualResetEventSlim BatchGate { get; } = new(initialState: true); - public volatile bool FailBatches; - - public int BatchCalls; - public int MaxConcurrentBatches; - private int _inFlight; - - /// Single-row upserts. Counted separately from batches because a flush that writes N rows - /// one at a time costs N round-trips no matter how fast each one is — the cost the batch path exists - /// to avoid. - public int SingleUpsertCalls; - - public List Rows(string table) - { - lock (_sync) return _tables.TryGetValue(table, out var t) ? t.Values.ToList() : new List(); - } - - public Task PingAsync(CancellationToken ct = default) => Task.CompletedTask; - - public Task EnsureTableAsync(string table, CancellationToken ct = default) - { - lock (_sync) { if (!_tables.ContainsKey(table)) _tables[table] = new(); } - return Task.CompletedTask; - } - - public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) - { - // Counted, not timed: the single-vs-batch distinction the tests assert on is the NUMBER of - // round-trips, so this needs no simulated latency. - Interlocked.Increment(ref SingleUpsertCalls); - lock (_sync) - { - if (!_tables.ContainsKey(table)) _tables[table] = new(); - _tables[table][(row.PartitionKey, row.RowKey)] = row; - } - return Task.CompletedTask; - } - - public async Task UpsertBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) - { - await Task.Yield(); // a real storage call never completes synchronously - Interlocked.Increment(ref BatchCalls); - var now = Interlocked.Increment(ref _inFlight); - InterlockedMax(ref MaxConcurrentBatches, now); - try - { - // Block here to model a stalled storage call, honouring cancellation so a flush timeout works. - while (!BatchGate.IsSet) - { - ct.ThrowIfCancellationRequested(); - await Task.Delay(5, ct); - } - if (FailBatches) throw new InvalidOperationException("storage unavailable"); - - lock (_sync) - { - if (!_tables.ContainsKey(table)) _tables[table] = new(); - foreach (var r in rows) _tables[table][(r.PartitionKey, r.RowKey)] = r; - } - } - finally - { - Interlocked.Decrement(ref _inFlight); - } - } - - public Task GetAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) - { - lock (_sync) - return Task.FromResult(_tables.TryGetValue(table, out var t) - && t.TryGetValue((partitionKey, rowKey), out var r) ? r : null); - } - - public async IAsyncEnumerable QueryPartitionAsync(string table, string partitionKey, - [EnumeratorCancellation] CancellationToken ct = default) - { - List snapshot; - lock (_sync) - snapshot = _tables.TryGetValue(table, out var t) - ? t.Where(k => k.Key.Item1 == partitionKey).Select(k => k.Value).ToList() - : new List(); - foreach (var r in snapshot) { yield return r; await Task.Yield(); } - } - - public async IAsyncEnumerable QueryTableAsync(string table, - [EnumeratorCancellation] CancellationToken ct = default) - { - List snapshot; - lock (_sync) snapshot = _tables.TryGetValue(table, out var t) ? t.Values.ToList() : new List(); - foreach (var r in snapshot) { yield return r; await Task.Yield(); } - } - - public Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) - { - lock (_sync) { if (_tables.TryGetValue(table, out var t)) t.Remove((partitionKey, rowKey)); } - return Task.CompletedTask; - } - - public Task DeletePartitionAsync(string table, string partitionKey, CancellationToken ct = default) - { - lock (_sync) - { - if (!_tables.TryGetValue(table, out var t)) return Task.CompletedTask; - foreach (var k in t.Keys.Where(k => k.Item1 == partitionKey).ToList()) t.Remove(k); - } - return Task.CompletedTask; - } - - private static void InterlockedMax(ref int target, int value) - { - int cur; - while (value > (cur = Volatile.Read(ref target))) - if (Interlocked.CompareExchange(ref target, value, cur) == cur) return; - } - } - - private static (OrchestratorStatusWriter Writer, ControllableStore Backing) NewWriter( - int barrierTimeoutSec = 2, int flushTimeoutSec = 1, int concurrency = 8) - { - var settings = new CraftSettings(); - settings.Orchestrator.RunningBarrierTimeoutSeconds = barrierTimeoutSec; - settings.Orchestrator.StatusFlushTimeoutSeconds = flushTimeoutSec; - settings.Orchestrator.StatusFlushConcurrency = concurrency; - - var backing = new ControllableStore(); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); - var writer = new OrchestratorStatusWriter(store, NullLogger.Instance, settings); - return (writer, backing); - } - - private static OrchestratorTaskItem Task1(string id = "t1") => - new() { Id = id, Status = "Running", Parameters = new Dictionary { ["TenantFilter"] = "x.com" } }; - - // ── R2: COALESCED SMALL-RESULT WRITES ───────────────────────────────────────────────────────────── - - /// - /// A small result rides the writer and is durable after a flush; one too large for a single table - /// property is refused, so the caller keeps the chunked StoreResultAsync path. - /// - [Fact] - public async Task SmallResult_Coalesces_AndIsDurable_WhileLargeResultIsRefused() - { - var settings = new CraftSettings(); - var backing = new ControllableStore(); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); - using var writer = new OrchestratorStatusWriter(store, NullLogger.Instance, settings); - - Assert.True(writer.TryQueueResult("run", "t1", "{\"ok\":true}")); - // 30 001 chars is one over the single-property bound — must fall back to the direct chunked path. - Assert.False(writer.TryQueueResult("run", "t2", new string('x', 30_001))); - - await writer.FlushAsync(); - - Assert.Contains("{\"ok\":true}", await store.GetResultsAsync("run")); - } - - /// With result-batching off, TryQueueResult refuses so the caller writes results directly. - [Fact] - public void TryQueueResult_IsRefused_WhenResultBatchingIsOff() - { - var settings = new CraftSettings(); - settings.Orchestrator.BatchResultWrites = false; - var backing = new ControllableStore(); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); - using var writer = new OrchestratorStatusWriter(store, NullLogger.Instance, settings); - - Assert.False(writer.TryQueueResult("run", "t1", "{\"ok\":true}")); - } - - // ── LIVENESS ──────────────────────────────────────────────────────────────────────────────────── - - /// - /// THE regression guard. A stalled storage write must not hold the caller forever — that wait is what - /// consumed all 8 slots while every worker sat idle. - /// - [Fact] - public async Task MarkRunning_TimesOut_WhenStorageStalls_RatherThanHangingForever() - { - var (writer, backing) = NewWriter(barrierTimeoutSec: 2, flushTimeoutSec: 1); - using var _ = writer; - - backing.BatchGate.Reset(); // storage stalls from here on - await writer.QueueRunWarmup(); // ensure the drain loop is running - - var sw = System.Diagnostics.Stopwatch.StartNew(); - await Assert.ThrowsAsync(() => writer.MarkRunningAsync("run", Task1())); - sw.Stop(); - - backing.BatchGate.Set(); - Assert.True(sw.Elapsed < TimeSpan.FromSeconds(30), - $"MarkRunningAsync took {sw.Elapsed.TotalSeconds:F1}s — the barrier is not bounded"); - } - - /// A stalled flush must not stop LATER work once storage recovers — the loop has to survive. - [Fact] - public async Task DrainLoop_KeepsWorking_AfterAFlushTimesOut() - { - var (writer, backing) = NewWriter(barrierTimeoutSec: 2, flushTimeoutSec: 1); - using var _ = writer; - - backing.BatchGate.Reset(); - await Assert.ThrowsAsync(() => writer.MarkRunningAsync("run", Task1("stalled"))); - - backing.BatchGate.Set(); // storage recovers - - // A brand-new marker must now succeed, proving the loop is still alive. - await writer.MarkRunningAsync("run", Task1("after-recovery")); - } - - /// FlushAsync is the other barrier consumer — no run could finalize while it hung. - [Fact] - public async Task FlushAsync_ReturnsWithinTheBound_WhenStorageStalls() - { - var (writer, backing) = NewWriter(barrierTimeoutSec: 2, flushTimeoutSec: 1); - using var _ = writer; - - backing.BatchGate.Reset(); - writer.QueueTask("run", Task1()); - - var sw = System.Diagnostics.Stopwatch.StartNew(); - await writer.FlushAsync(); // must not throw and must not hang - sw.Stop(); - - backing.BatchGate.Set(); - Assert.True(sw.Elapsed < TimeSpan.FromSeconds(30), - $"FlushAsync took {sw.Elapsed.TotalSeconds:F1}s — it is not bounded"); - } - - // ── DURABILITY ────────────────────────────────────────────────────────────────────────────────── - - /// - /// The write that could not be persisted must be retried, not dropped. Dropping it — the previous - /// behaviour on ANY exception — silently lost terminal task states, leaving finished tasks looking - /// Pending forever and re-run by the next recovery pass. - /// - [Fact] - public async Task WritesThatFail_AreRetried_NotLost() - { - var (writer, backing) = NewWriter(barrierTimeoutSec: 2, flushTimeoutSec: 1); - using var _ = writer; - - backing.FailBatches = true; - writer.QueueTask("run", new OrchestratorTaskItem { Id = "t1", Status = "Completed" }); - - await Task.Delay(400); // several failing flushes - Assert.Empty(backing.Rows("OrchestratorTasks")); // nothing persisted yet - - backing.FailBatches = false; // storage recovers - - var deadline = Environment.TickCount64 + 5000; - while (Environment.TickCount64 < deadline && backing.Rows("OrchestratorTasks").Count == 0) - await Task.Delay(20); - - var rows = backing.Rows("OrchestratorTasks"); - Assert.Single(rows); - Assert.Equal("Completed", rows[0].GetString("Status")); // the state survived the outage - } - - /// A newer state for the same task must not be clobbered by a retry of an older snapshot. - [Fact] - public async Task Retry_DoesNotOverwrite_NewerStateForTheSameTask() - { - var (writer, backing) = NewWriter(barrierTimeoutSec: 2, flushTimeoutSec: 1); - using var _ = writer; - - backing.FailBatches = true; - writer.QueueTask("run", new OrchestratorTaskItem { Id = "t1", Status = "Running" }); - await Task.Delay(200); - - writer.QueueTask("run", new OrchestratorTaskItem { Id = "t1", Status = "Completed" }); - backing.FailBatches = false; - - var deadline = Environment.TickCount64 + 5000; - while (Environment.TickCount64 < deadline && backing.Rows("OrchestratorTasks").Count == 0) - await Task.Delay(20); - - var rows = backing.Rows("OrchestratorTasks"); - Assert.Single(rows); - Assert.Equal("Completed", rows[0].GetString("Status")); - } - - /// Shutdown must still push everything pending to storage. - [Fact] - public async Task Dispose_DrainsPendingWrites() - { - var (writer, backing) = NewWriter(barrierTimeoutSec: 2, flushTimeoutSec: 5); - - writer.QueueTask("run", new OrchestratorTaskItem { Id = "t1", Status = "Completed" }); - writer.Dispose(); // final drain runs here - await Task.Delay(50); - - Assert.Single(backing.Rows("OrchestratorTasks")); - } - - // ── THROUGHPUT SHAPE ──────────────────────────────────────────────────────────────────────────── - - /// - /// Writes group by run because a batch shares a partition key. CIPP's workload is ~600 runs of ONE - /// task each, so sequential groups meant hundreds of round-trips per flush with the whole process - /// waiting. They must overlap. - /// - [Fact] - public async Task ManySingleTaskRuns_AreWrittenConcurrently_NotOneAtATime() - { - var settings = new CraftSettings(); - var backing = new ControllableStore(); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); - - // The shape that broke production: one run per tenant, one task each. - var writes = Enumerable.Range(0, 200) - .Select(i => new TaskStatusWrite($"AuditLogIngestV2-tenant{i:D3}.dk", "t1", "Completed", "{}", 0, null, null, null)) - .ToList(); - - // Hold every batch inside the store, so overlap is FORCED rather than raced for. - // - // Two earlier versions of this assertion measured the machine instead of the code. It first - // asserted the whole thing finished inside two seconds — that failed on a box running a 124-task - // orchestration and passed on the same commit once idle. Replacing it with "peak concurrency - // reaches the cap of 8" was no better: under a fully saturated CPU the observed peak was 3, - // because starvation delays the continuations ENTERING the counted region, so it narrows the - // window rather than widening it. - // - // With the gate closed every batch blocks after being counted, so the first 8 occupy the - // semaphore and stay there no matter how slow the scheduler is. The only thing left to wait for - // is progress, and the bound below is generous enough that a slow machine takes longer rather - // than failing. - backing.BatchGate.Reset(); - var pending = store.WriteTaskStatusBatchAsync(writes, maxConcurrency: 8); - - for (var i = 0; i < 400 && Volatile.Read(ref backing.MaxConcurrentBatches) < 8; i++) - await Task.Delay(25); - - // Exactly 8: fewer means the per-run writes are not overlapping as designed, more means the - // requested cap is not being honoured. - Assert.Equal(8, Volatile.Read(ref backing.MaxConcurrentBatches)); - - backing.BatchGate.Set(); - var failed = await pending; - - Assert.Empty(failed); - Assert.Equal(200, backing.BatchCalls); - } - - /// One run's failure must not discard the other 199. - [Fact] - public async Task OneFailingRun_DoesNotDiscardTheRest() - { - var settings = new CraftSettings(); - var backing = new FlakyStore(failFor: "AuditLogIngestV2-tenant005.dk"); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); - - var writes = Enumerable.Range(0, 20) - .Select(i => new TaskStatusWrite($"AuditLogIngestV2-tenant{i:D3}.dk", "t1", "Completed", "{}", 0, null, null, null)) - .ToList(); - - var failed = await store.WriteTaskStatusBatchAsync(writes, maxConcurrency: 4); - - Assert.Equal(["AuditLogIngestV2-tenant005.dk"], failed); - Assert.Equal(19, backing.Written); - } - - private sealed class FlakyStore : ICraftTableStore - { - - // Claims are not exercised by this fake. Fail loudly rather than pretend the guard held — - // a silent 'true' here would look exactly like a successful claim. - public Task TryReplaceBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) => throw new NotSupportedException(); - private readonly string _failFor; - public int Written; - public FlakyStore(string failFor) => _failFor = failFor; - - public Task PingAsync(CancellationToken ct = default) => Task.CompletedTask; - public Task EnsureTableAsync(string table, CancellationToken ct = default) => Task.CompletedTask; - public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) => Task.CompletedTask; - - public Task UpsertBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) - { - if (partitionKey == _failFor) throw new InvalidOperationException("partition unavailable"); - Interlocked.Increment(ref Written); - return Task.CompletedTask; - } - - public Task GetAsync(string t, string p, string r, CancellationToken ct = default) => Task.FromResult(null); - public async IAsyncEnumerable QueryPartitionAsync(string t, string p, - [EnumeratorCancellation] CancellationToken ct = default) - { - await Task.CompletedTask; - yield break; - } - - public async IAsyncEnumerable QueryTableAsync(string t, - [EnumeratorCancellation] CancellationToken ct = default) - { - await Task.CompletedTask; - yield break; - } - - public Task DeleteAsync(string t, string p, string r, CancellationToken ct = default) => Task.CompletedTask; - public Task DeletePartitionAsync(string t, string p, CancellationToken ct = default) => Task.CompletedTask; - } - - // ── RUN-ROW WRITE COST ────────────────────────────────────────────────────────────────────────── - - /// - /// Run rows all share the constant "Run" partition key, so a flush carrying N of them can persist - /// them in ceil(N/100) transactions. Writing them one at a time instead costs N round-trips inside a - /// flush that is bounded by StatusFlushTimeoutSeconds — the cost that pushed real flushes past 30s, - /// then past the 90s barrier, deferring every waiting task until it was abandoned as Pending. - /// - /// Guards the write SHAPE, not wall-clock: a timing assertion here would be flaky on a loaded agent. - /// - [Fact] - public async Task RunRows_ArePersistedInBatches_NotOnePerRoundTrip() - { - var (writer, backing) = NewWriter(barrierTimeoutSec: 30, flushTimeoutSec: 20); - using var _ = writer; - - const int runCount = 150; - for (var i = 0; i < runCount; i++) - writer.QueueRun(new OrchestratorRun { Name = $"run-{i:D3}", Status = "Running" }); - - await writer.FlushAsync(); - - Assert.Equal(runCount, backing.Rows("OrchestratorRuns").Count); - - // 150 rows in one partition = 2 transactions of 100 + 50. Allow generous headroom for the - // warmup row and flush-cycle boundaries, but nothing close to one call per row. - Assert.True(backing.SingleUpsertCalls <= 10, - $"run rows were written with {backing.SingleUpsertCalls} single upserts for {runCount} runs — " + - "they share one partition key and should be batched"); - } - - /// - /// A batch transaction fails atomically, so one bad row would take out the other 99. The write path - /// must fall back to per-row writes for that chunk rather than reporting all of them unwritten. - /// - [Fact] - public async Task RunRows_FallBackToIndividualWrites_WhenABatchFails() - { - var (writer, backing) = NewWriter(barrierTimeoutSec: 30, flushTimeoutSec: 20); - using var _ = writer; - - backing.FailBatches = true; // every batch transaction rejects - - for (var i = 0; i < 5; i++) - writer.QueueRun(new OrchestratorRun { Name = $"run-{i}", Status = "Running" }); - - await writer.FlushAsync(); - - // Batching failed, so each row must still have reached storage individually. - Assert.Equal(5, backing.Rows("OrchestratorRuns").Count); - } -} - -internal static class StatusWriterTestExtensions -{ - /// Nudge the drain loop so it is definitely running before a test manipulates storage. - public static Task QueueRunWarmup(this OrchestratorStatusWriter writer) - { - writer.QueueRun(new OrchestratorRun { Name = "warmup", Status = "Running" }); - return Task.Delay(60); - } -} diff --git a/tests/Craft.Tests/TableKeyTests.cs b/tests/Craft.Tests/TableKeyTests.cs index 081fe81..552236b 100644 --- a/tests/Craft.Tests/TableKeyTests.cs +++ b/tests/Craft.Tests/TableKeyTests.cs @@ -91,7 +91,7 @@ public void TwoNamesThatFoldOntoTheSameIdStayDistinct() [Fact] public void QueueRowKeyIsLegalForARepoNamedTask() { - // The end of the chain the bug actually travelled: id → BuildRowKey → upsert → 400. + // The end of the chain the bug actually travelled: id → row key → upsert → 400. var id = IdFor(""" { "FunctionName": "ExecScheduledCommand", @@ -99,8 +99,8 @@ public void QueueRowKeyIsLegalForARepoNamedTask() } """); - var rowKey = JobQueueStore.BuildRowKey(DateTime.UnixEpoch, "UserTaskOrchestrator_No tenant", id); - - Assert.True(TableKeys.IsSafe(rowKey)); + // The id keys the task's result row; the run key is the partition of both tables. + Assert.True(TableKeys.IsSafe(id)); + Assert.True(TableKeys.IsSafe(WorkStore.RunKeyFor("UserTaskOrchestrator_No tenant", DateTime.UnixEpoch))); } } diff --git a/tests/Craft.Tests/WorkStoreTests.cs b/tests/Craft.Tests/WorkStoreTests.cs new file mode 100644 index 0000000..6bff310 --- /dev/null +++ b/tests/Craft.Tests/WorkStoreTests.cs @@ -0,0 +1,265 @@ +using Craft.Configuration; +using Craft.Storage; +using Microsoft.Extensions.Logging.Abstractions; + +namespace Craft.Tests; + +/// +/// The storage model: a task's state is one row in its run's partition, and every transition is one +/// transaction that also moves the run's counts. These pin the transitions and the guarantees that replace +/// the old reconciliation machinery: no task is claimed twice, a lost lease cannot finish someone else's +/// task, the counts reach the barrier exactly once, and a task that keeps dying is failed rather than retried +/// forever. +/// +public class WorkStoreTests +{ + private static readonly TimeSpan Lease = TimeSpan.FromMinutes(30); + + private static (WorkStore Store, MemoryTableStore Mem) New() + { + var mem = new MemoryTableStore(); + return (new WorkStore(NullLogger.Instance, new CraftSettings(), mem), mem); + } + + private static RunHeader Header(string name, int minute = 0, int priority = 4, string? postExec = null) => new() + { + RunKey = WorkStore.RunKeyFor(name, At(minute)), + Name = name, + Priority = priority, + StartedUtc = At(minute), + TaskScriptName = "Invoke-CraftTask", + PostExecFunctionName = postExec, + }; + + private static DateTime At(int minute) => new(2026, 10, 5, 3, minute, 0, DateTimeKind.Utc); + + private static List Tasks(int n) => + Enumerable.Range(0, n).Select(i => new WorkStore.NewTask($"t{i}", new() { ["i"] = i })).ToList(); + + private static async Task CreateAsync(WorkStore s, string name, int tasks, int minute = 0, + int priority = 4, string? postExec = null) => + await s.CreateRunAsync(Header(name, minute, priority, postExec), Tasks(tasks)); + + private static List Done(IEnumerable claims, string owner = "w") => + claims.Select(c => new WorkStore.Finish(c.Seq, "Completed", Owner: owner)).ToList(); + + [Fact] + public async Task ARunsTasksAreClaimedOnce_AndTheLastFinishCompletesTheRun() + { + var (s, _) = New(); + var run = await CreateAsync(s, "R", 3); + + var first = await s.ClaimAsync(run.RunKey, 2, "w", Lease, false); + var second = await s.ClaimAsync(run.RunKey, 2, "w", Lease, false); + Assert.Equal([0, 1], first.Select(c => c.Seq)); + Assert.Equal([2], second.Select(c => c.Seq)); + Assert.Empty(await s.ClaimAsync(run.RunKey, 2, "w", Lease, false)); + + var partial = await s.FinishAsync(run.RunKey, Done(first)); + Assert.False(partial!.Completed); + Assert.Equal(2, partial.Header.Done); + + var last = await s.FinishAsync(run.RunKey, Done(second)); + Assert.True(last!.ReachedBarrier); + Assert.True(last.Completed); + Assert.Equal("Completed", last.Header.Status); + Assert.Empty(await ReadyAsync(s)); + } + + [Fact] + public async Task TheBarrierQueuesTheAggregation_WhichCompletesTheRun() + { + var (s, _) = New(); + var run = await CreateAsync(s, "R", 1, postExec: "StoreThings"); + + var task = await s.ClaimAsync(run.RunKey, 8, "w", Lease, false); + var barrier = await s.FinishAsync(run.RunKey, Done(task)); + Assert.True(barrier!.ReachedBarrier); + Assert.False(barrier.Completed); + Assert.Equal(RunPhase.Aggregate, barrier.Header.Phase); + + var agg = Assert.Single(await s.ClaimAsync(run.RunKey, 8, "w", Lease, false)); + Assert.Equal(WorkStore.AggregateSeq, agg.Seq); + var done = await s.FinishAsync(run.RunKey, Done([agg])); + Assert.True(done!.Completed); + Assert.Equal("Completed", done.Header.PostExecStatus); + Assert.Equal(1, done.Header.Done); + } + + [Fact] + public async Task AFinishFromAWorkerThatLostItsLease_IsIgnored() + { + var (s, _) = New(); + var run = await CreateAsync(s, "R", 1); + var claim = await s.ClaimAsync(run.RunKey, 1, "old", TimeSpan.FromMilliseconds(1), false); + await Task.Delay(20); + Assert.Single(await s.ClaimAsync(run.RunKey, 1, "new", Lease, reclaimExpired: true)); + + var stale = await s.FinishAsync(run.RunKey, Done(claim, "old")); + + Assert.Equal(0, stale!.Applied); + Assert.Equal(0, (await s.GetRunAsync(run.RunKey))!.Done); + } + + [Fact] + public async Task ATaskThatKeepsDying_IsFailedAfterThreeAttempts() + { + var (s, _) = New(); + var run = await CreateAsync(s, "R", 1); + + for (var i = 0; i < 3; i++) + { + Assert.Single(await s.ClaimAsync(run.RunKey, 1, $"w{i}", TimeSpan.FromMilliseconds(1), reclaimExpired: true)); + await Task.Delay(20); + } + Assert.Empty(await s.ClaimAsync(run.RunKey, 1, "w4", Lease, reclaimExpired: true)); + + var header = (await s.GetRunAsync(run.RunKey))!; + Assert.Equal(1, header.Failed); + Assert.Equal("CompletedWithErrors", header.Status); + } + + [Fact] + public async Task TwoClaimersRacingForTheSameRows_NeverBothWin() + { + var (s, mem) = New(); + var run = await CreateAsync(s, "R", 2); + var raced = false; + mem.BeforeSubmit = async () => + { + if (raced) return; + raced = true; + mem.BeforeSubmit = null; + await s.ClaimAsync(run.RunKey, 2, "rival", Lease, false); + }; + + Assert.Empty(await s.ClaimAsync(run.RunKey, 2, "me", Lease, false)); + Assert.All(await s.GetTasksAsync(run.RunKey, 'R'), t => Assert.Equal("rival", t.Owner)); + } + + [Fact] + public async Task CancellingARun_CancelsPendingTasks_AndRunningOnesFinishTheRun() + { + var (s, _) = New(); + var run = await CreateAsync(s, "R", 3); + var running = await s.ClaimAsync(run.RunKey, 1, "w", Lease, false); + + var (cancelled, _) = await s.CancelPendingAsync(run.RunKey); + Assert.Equal(2, cancelled); + Assert.False((await s.GetRunAsync(run.RunKey))!.IsFinished); + + var last = await s.FinishAsync(run.RunKey, Done(running)); + Assert.True(last!.Completed); + Assert.Equal("CompletedWithErrors", last.Header.Status); + Assert.Equal(2, last.Header.Cancelled); + } + + [Fact] + public async Task AParentWaitsForItsChild_ThenCompletes() + { + var (s, _) = New(); + var parent = await CreateAsync(s, "Parent", 1); + var task = await s.ClaimAsync(parent.RunKey, 1, "w", Lease, false); + + Assert.True(await s.AddChildAsync(parent.RunKey, "Child~1")); + var tasksDone = await s.FinishAsync(parent.RunKey, Done(task)); + Assert.False(tasksDone!.Completed); + + var childDone = await s.FinishAsync(parent.RunKey, [new WorkStore.Finish(0, "Completed", ChildKey: "Child~1")]); + Assert.True(childDone!.Completed); + Assert.Equal(2, childDone.Header.Done); + } + + [Fact] + public async Task ARunQueuedFromAnAggregation_IsNotAChild() + { + var (s, _) = New(); + var parent = await CreateAsync(s, "Parent", 1, postExec: "Agg"); + await s.FinishAsync(parent.RunKey, Done(await s.ClaimAsync(parent.RunKey, 1, "w", Lease, false))); + + Assert.False(await s.AddChildAsync(parent.RunKey, "Child~1")); + } + + [Fact] + public async Task ReadyListsTheBestBandFirst_ThenTheOldestRun() + { + var (s, _) = New(); + await CreateAsync(s, "Zeta", 1, minute: 0); + await CreateAsync(s, "Alpha", 1, minute: 5); + await CreateAsync(s, "Urgent", 1, minute: 9, priority: 1); + + Assert.Equal(["Urgent", "Zeta", "Alpha"], (await ReadyAsync(s)).Select(e => e.Name)); + } + + [Fact] + public async Task AReleasedTaskGoesBackToPending_WithItsAttemptRefunded() + { + var (s, _) = New(); + var run = await CreateAsync(s, "R", 1); + var claim = Assert.Single(await s.ClaimAsync(run.RunKey, 1, "w", Lease, false)); + + Assert.True(await s.ReleaseAsync(run.RunKey, claim.Seq, "w", refundAttempt: true)); + + var again = Assert.Single(await s.ClaimAsync(run.RunKey, 1, "w", Lease, false)); + Assert.Equal(1, again.Attempt); + } + + [Fact] + public async Task RetentionDeletesRunsFinishedBeforeTheCutoff() + { + var (s, mem) = New(); + var run = await CreateAsync(s, "R", 1); + await s.FinishAsync(run.RunKey, Done(await s.ClaimAsync(run.RunKey, 1, "w", Lease, false))); + + Assert.Equal(0, await s.SweepFinishedAsync(TimeSpan.FromHours(1))); + Assert.Equal(1, await s.SweepFinishedAsync(TimeSpan.Zero)); + Assert.Null(await s.GetRunAsync(run.RunKey)); + Assert.Null(await s.GetRunByNameAsync("R")); + Assert.DoesNotContain(mem.All($"{new CraftSettings().Orchestrator.TablePrefix}Work"), r => r.PartitionKey == run.RunKey); + } + + /// + /// The header is rewritten in every finish transaction, and a transaction cannot split a row, so a + /// property over Azure's 64 KiB limit there would fail every finish of the run. Azurite does not enforce + /// the limit, so this is checked on the row itself. + /// + [Fact] + public async Task TheHeaderStaysSmall_HoweverLargeThePostExecutionParameters() + { + var (s, mem) = New(); + var big = new string('x', 200_000); + var header = Header("Big", postExec: "Agg"); + await s.CreateRunAsync(new RunHeader + { + RunKey = header.RunKey, + Name = header.Name, + StartedUtc = header.StartedUtc, + TaskScriptName = header.TaskScriptName, + PostExecFunctionName = "Agg", + PostExecParametersJson = big, + }, Tasks(1)); + + var row = mem.All("OrchestratorWork").Single(r => r.RowKey == WorkStore.HeaderKey); + Assert.All(row.Properties.Values, v => Assert.True((v as string)?.Length is null or < 32_000)); + Assert.Equal(big, await s.GetPostExecParametersAsync(header.RunKey)); + } + + [Fact] + public async Task APayloadRoundTripsToTheTaskThatWasClaimed() + { + var (s, _) = New(); + var run = await CreateAsync(s, "R", 2); + var claim = (await s.ClaimAsync(run.RunKey, 2, "w", Lease, false))[1]; + + var payload = await s.GetPayloadAsync(run.RunKey, claim.Seq); + Assert.Equal("t1", claim.TaskId); + Assert.Equal("1", payload!["i"].ToString()); + } + + private static async Task> ReadyAsync(WorkStore s) + { + var list = new List(); + await foreach (var e in s.ReadReadyAsync()) list.Add(e); + return list; + } +} From e636885ee3b416b050cd4d6ac56a81439419b2c2 Mon Sep 17 00:00:00 2001 From: Zacgoose <107489668+Zacgoose@users.noreply.github.com> Date: Tue, 6 Oct 2026 00:27:21 +0800 Subject: [PATCH 06/24] feat(orchestration): let runs of one name overlap unless the caller opts out - QueueOrchestration/QueueOrchestrationFromFile take allowCollision (default true); false skips the run while another run of that name is unfinished - the stamped operation context carries RunKey, so a child registers under its exact parent; a run name still resolves to the newest outing - cancel, reprioritize and task cancel by name apply to every unfinished run of that name - pump job ids are unique per claim; status counts sum runs sharing a name --- Services/Bridges/OrchestratorBridge.cs | 23 ++- Services/Hosting/OperationContext.cs | 6 + Services/Orchestration/JobManager.cs | 1 + .../Orchestration/JobQueueStatusReader.cs | 10 +- Services/Orchestration/OrchestratorService.cs | 155 +++++++++++------- Services/Orchestration/WorkPump.cs | 2 +- .../PowerShellHost/PowerShellRunnerService.cs | 2 + Services/Storage/WorkStore.cs | 32 +++- .../Craft.Tests/JobQueueStatusReaderTests.cs | 13 ++ .../Craft.Tests/OrchestrationContractTests.cs | 77 ++++++++- .../OrchestratorBatchStreamingTests.cs | 6 +- 11 files changed, 245 insertions(+), 82 deletions(-) diff --git a/Services/Bridges/OrchestratorBridge.cs b/Services/Bridges/OrchestratorBridge.cs index 24988ac..f7e4d68 100644 --- a/Services/Bridges/OrchestratorBridge.cs +++ b/Services/Bridges/OrchestratorBridge.cs @@ -23,9 +23,11 @@ public static class OrchestratorBridge public static void Initialize(OrchestratorService service) => s_service = service; + /// True (the default) lets runs of one name stack up; false skips this run + /// while another run of the same name is unfinished. public static void QueueOrchestration(string name, string batchJson, int priority, string? postExecFunctionName = null, string? postExecParametersJson = null, - string? reference = null, string? parentRunName = null, bool sequential = false) + string? reference = null, string? parentRunName = null, bool sequential = false, bool allowCollision = true) { // Sanitized here as well as at run creation so the child-run registration below // records the SAME name the service ends up creating — a raw name with a table-illegal @@ -35,7 +37,8 @@ public static void QueueOrchestration(string name, string batchJson, int priorit var child = RegisterPendingChild(parentRunName, name); s_pending.Enqueue(new PendingOrchestration(name, batchJson, priority, postExecFunctionName, postExecParametersJson, parentRunName, reference, - Sequential: sequential, ParentRunKey: child?.ParentRunKey, ChildKey: child?.ChildKey)); + Sequential: sequential, ParentRunKey: child?.ParentRunKey, ChildKey: child?.ChildKey, + AllowCollision: allowCollision)); } /// @@ -49,20 +52,24 @@ public static void QueueOrchestration(string name, string batchJson, int priorit /// /// The file is owned by the orchestrator from this point: it is deleted once parsed. /// + /// True (the default) lets runs of one name stack up; false skips this run + /// while another run of the same name is unfinished. public static void QueueOrchestrationFromFile(string name, string batchFilePath, int priority, string? postExecFunctionName = null, string? postExecParametersJson = null, - string? reference = null, string? parentRunName = null, bool sequential = false) + string? reference = null, string? parentRunName = null, bool sequential = false, bool allowCollision = true) { name = TableKeys.Sanitize(name); parentRunName = ResolveParentRunName(name, parentRunName); var child = RegisterPendingChild(parentRunName, name); s_pending.Enqueue(new PendingOrchestration(name, string.Empty, priority, postExecFunctionName, postExecParametersJson, parentRunName, reference, batchFilePath, - Sequential: sequential, ParentRunKey: child?.ParentRunKey, ChildKey: child?.ChildKey)); + Sequential: sequential, ParentRunKey: child?.ParentRunKey, ChildKey: child?.ChildKey, + AllowCollision: allowCollision)); } /// - /// Resolve the parent run of a queued orchestration. The explicit argument wins — PowerShell + /// Resolve the parent run of a queued orchestration: a run key (exact — runs of one name can overlap) or a + /// run name (the newest outing). The explicit argument wins — PowerShell /// callers MUST pass it (read from the stamped $global:CraftOperationContext), because the /// ambient fallback cannot work for them: the pipeline runs on the runspace's reused thread, /// whose frozen ExecutionContext never sees the per-invocation AsyncLocal (see @@ -75,7 +82,7 @@ public static void QueueOrchestrationFromFile(string name, string batchFilePath, private static string? ResolveParentRunName(string name, string? parentRunName) { if (string.IsNullOrEmpty(parentRunName)) - parentRunName = OperationContext.Current?.RunName; + parentRunName = OperationContext.Current?.RunKey ?? OperationContext.Current?.RunName; if (string.IsNullOrEmpty(parentRunName)) return null; // Sanitized like the child name: the parent was created under its sanitized name, and the @@ -119,7 +126,7 @@ private static async Task StartAsync(PendingOrchestration p) if (s_service == null) { DiscardUndispatchable(p); return; } created = await s_service.StartFromBatchAsync(p.Name, p.BatchJson, p.Priority, p.PostExecFunctionName, p.PostExecParametersJson, CancellationToken.None, - p.ParentRunName, p.Reference, p.BatchFilePath, p.Sequential, p.ParentRunKey, p.ChildKey); + p.ParentRunName, p.Reference, p.BatchFilePath, p.Sequential, p.ParentRunKey, p.ChildKey, p.AllowCollision); } catch (Exception ex) { @@ -160,7 +167,7 @@ private static void DiscardUndispatchable(PendingOrchestration p) public record PendingOrchestration(string Name, string BatchJson, int Priority, string? PostExecFunctionName, string? PostExecParametersJson, string? ParentRunName, string? Reference = null, string? BatchFilePath = null, bool Sequential = false, - string? ParentRunKey = null, string? ChildKey = null) + string? ParentRunKey = null, string? ChildKey = null, bool AllowCollision = true) { public bool PendingChildRegistered => ChildKey != null; } diff --git a/Services/Hosting/OperationContext.cs b/Services/Hosting/OperationContext.cs index cfe22b6..05ba74e 100644 --- a/Services/Hosting/OperationContext.cs +++ b/Services/Hosting/OperationContext.cs @@ -58,6 +58,12 @@ public sealed class Invocation /// public string? RunName { get; init; } + /// + /// Storage key of the enclosing run. Runs of one name can overlap, so this — not + /// — is what identifies the parent exactly when a task queues a child. + /// + public string? RunKey { get; init; } + /// /// Queue priority of the enclosing run, exposed so nested enqueues can inherit it. /// PowerShell cannot read this statically — the pipeline thread never sees the AsyncLocal diff --git a/Services/Orchestration/JobManager.cs b/Services/Orchestration/JobManager.cs index 12f5b36..8957f45 100644 --- a/Services/Orchestration/JobManager.cs +++ b/Services/Orchestration/JobManager.cs @@ -355,6 +355,7 @@ private async Task RunJobAsync(QueuedJob job, CancellationToken ct) var parentInvocation = new OperationContext.Invocation(job.Record.Name) { RunName = job.Record.RunName, + RunKey = job.Descriptor?.RunKey, Priority = job.Descriptor != null ? job.Record.Priority : job.InheritPriority, Category = "Job" }; diff --git a/Services/Orchestration/JobQueueStatusReader.cs b/Services/Orchestration/JobQueueStatusReader.cs index a6914f2..2c81328 100644 --- a/Services/Orchestration/JobQueueStatusReader.cs +++ b/Services/Orchestration/JobQueueStatusReader.cs @@ -89,6 +89,7 @@ private TimeSpan EffectiveTtl(TimeSpan? maxAge) private async Task BuildSnapshotAsync(CancellationToken ct) { + // Local jobs are counted per run name; runs sharing a name take them oldest first. var local = _jobs.GetJobs().Where(j => j.RunName != null && j.Status is "Queued" or "Running") .GroupBy(j => j.RunName!).ToDictionary(g => g.Key, g => g.Count(), StringComparer.Ordinal); @@ -101,10 +102,14 @@ private async Task BuildSnapshotAsync(CancellationToken ct) { var outstanding = Math.Max(0, e.Total - e.Done); var claimed = Math.Min(outstanding, local.GetValueOrDefault(e.Name)); + local[e.Name] = local.GetValueOrDefault(e.Name) - claimed; total += outstanding; unclaimed += outstanding - claimed; if (outstanding > claimed && (oldest == null || e.StartedUtc < oldest)) oldest = e.StartedUtc; - byRun[e.Name] = new RunQueueInfo(outstanding - claimed, claimed, e.Band, e.StartedUtc, e.Total, e.Done, e.Reference); + byRun[e.Name] = byRun.TryGetValue(e.Name, out var same) + ? new RunQueueInfo(same.Unclaimed + outstanding - claimed, same.Claimed + claimed, Math.Min(same.MinPriority, e.Band), + same.OldestQueuedUtc, same.Total + e.Total, same.Done + e.Done, same.Reference ?? e.Reference) + : new RunQueueInfo(outstanding - claimed, claimed, e.Band, e.StartedUtc, e.Total, e.Done, e.Reference); if (head.Count < HeadRows && runsListed < HeadRuns && outstanding > claimed) { @@ -149,11 +154,12 @@ public async Task> GetJobDetailsAsync(string? runName = null, st var now = DateTime.UtcNow; var merged = new List(local); + var held = _jobs.GetJobs().Where(j => j.Status is "Queued" or "Running").Select(j => j.Name).ToHashSet(StringComparer.Ordinal); foreach (var row in snap.Rows) { if (!string.IsNullOrEmpty(runName) && !string.Equals(row.RunName, runName, StringComparison.OrdinalIgnoreCase)) continue; var id = $"{row.RunName}-{row.TaskId}"; - if (_jobs.IsQueuedOrRunning(id)) continue; + if (held.Contains(id)) continue; merged.Add(new JobDetail { Id = id, diff --git a/Services/Orchestration/OrchestratorService.cs b/Services/Orchestration/OrchestratorService.cs index 5010776..df95d51 100644 --- a/Services/Orchestration/OrchestratorService.cs +++ b/Services/Orchestration/OrchestratorService.cs @@ -152,54 +152,53 @@ public async Task StartPlannerRunAsync(string command, int priority, Cancellatio /// /// Start a run from a pre-built batch (OrchestratorBridge). The batch is a JSON Lines file /// (, deleted on every path) or a JSON array string. Returns whether a run - /// was created: false when the batch is empty or a run of this name is still going. + /// was created: false when the batch is empty, or when is false and a + /// run of this name is still going. By default runs of one name stack up side by side. /// public async Task StartFromBatchAsync(string name, string batchJson, int priority, string? postExecFunctionName, string? postExecParametersJson, CancellationToken ct, string? parentRunName = null, string? reference = null, string? batchFilePath = null, - bool sequential = false, string? parentRunKey = null, string? childKey = null) + bool sequential = false, string? parentRunKey = null, string? childKey = null, bool allowCollision = true) { + var gated = false; try { name = TableKeys.Sanitize(name); - if (!_activePlanners.TryAdd(name, true) || await IsActiveAsync(name, ct)) + if (!allowCollision) { - _activePlanners.TryRemove(name, out _); - _logger.LogInformation("[Orchestrator] Run {Name} already active, skipping", name); - return false; - } - - try - { - var tasks = !string.IsNullOrEmpty(batchFilePath) - ? ParseTasksFromJsonLinesFile(batchFilePath, name) - : ParseTasksFromJson(batchJson, name); - if (tasks.Count == 0) - { - _logger.LogWarning("[Orchestrator] Batch for {Name} produced 0 tasks", name); - return false; - } - - var genericTaskFunc = _settings.Orchestrator.GenericTaskFunction; - if (string.IsNullOrEmpty(genericTaskFunc) || Script(genericTaskFunc) == null) + gated = _activePlanners.TryAdd(name, true); + if (!gated || await IsActiveAsync(name, ct)) { - _logger.LogError("[Orchestrator] Cannot start {Name}: task function {Func} not found", name, genericTaskFunc); + _logger.LogInformation("[Orchestrator] Run {Name} already active, skipping", name); return false; } + } - if (string.IsNullOrEmpty(postExecFunctionName)) postExecFunctionName = null; - if (string.IsNullOrEmpty(postExecParametersJson)) postExecParametersJson = null; - await CreateAsync(name, tasks, priority, genericTaskFunc, postExecFunctionName, postExecParametersJson, - reference, parentRunKey, childKey, sequential, ct); - return true; + var tasks = !string.IsNullOrEmpty(batchFilePath) + ? ParseTasksFromJsonLinesFile(batchFilePath, name) + : ParseTasksFromJson(batchJson, name); + if (tasks.Count == 0) + { + _logger.LogWarning("[Orchestrator] Batch for {Name} produced 0 tasks", name); + return false; } - finally + + var genericTaskFunc = _settings.Orchestrator.GenericTaskFunction; + if (string.IsNullOrEmpty(genericTaskFunc) || Script(genericTaskFunc) == null) { - _activePlanners.TryRemove(name, out _); + _logger.LogError("[Orchestrator] Cannot start {Name}: task function {Func} not found", name, genericTaskFunc); + return false; } + + if (string.IsNullOrEmpty(postExecFunctionName)) postExecFunctionName = null; + if (string.IsNullOrEmpty(postExecParametersJson)) postExecParametersJson = null; + await CreateAsync(name, tasks, priority, genericTaskFunc, postExecFunctionName, postExecParametersJson, + reference, parentRunKey, childKey, sequential, ct); + return true; } finally { + if (gated) _activePlanners.TryRemove(name, out _); if (!string.IsNullOrEmpty(batchFilePath)) { try { if (File.Exists(batchFilePath)) File.Delete(batchFilePath); } @@ -209,13 +208,21 @@ await CreateAsync(name, tasks, priority, genericTaskFunc, postExecFunctionName, } private async Task IsActiveAsync(string name, CancellationToken ct) => - await _store.GetRunByNameAsync(name, ct) is { IsFinished: false }; + (await _store.GetActiveRunsAsync(name, ct)).Count > 0; + + /// The runs an operator action names: the run with that key, or every unfinished run of that name. + private async Task> TargetRunsAsync(string keyOrName) + { + keyOrName = TableKeys.Sanitize(keyOrName); + if (keyOrName.Contains('~') && await _store.GetRunAsync(keyOrName) is { IsFinished: false } run) return [run]; + return await _store.GetActiveRunsAsync(keyOrName); + } private async Task CreateAsync(string name, List tasks, int priority, string taskScriptName, string? postExecFunctionName, string? postExecParametersJson, string? reference, string? parentRunKey, string? childKey, bool sequential, CancellationToken ct) { - var started = DateTime.UtcNow; + var started = NextStartTime(); var header = new RunHeader { RunKey = WorkStore.RunKeyFor(name, started), @@ -237,24 +244,39 @@ private async Task CreateAsync(string name, List tasks, in sequential ? " (sequential)" : ""); } + private static long s_lastStartTicks; + + /// Strictly increasing start times, so runs of one name started together still get distinct keys. + private static DateTime NextStartTime() + { + while (true) + { + var last = Interlocked.Read(ref s_lastStartTicks); + var next = Math.Max(DateTime.UtcNow.Ticks, last + 1); + if (Interlocked.CompareExchange(ref s_lastStartTicks, next, last) == last) return new DateTime(next, DateTimeKind.Utc); + } + } + // ── child runs ── /// - /// Make wait for a child that is about to be queued. Called at enqueue time, - /// inside the parent's task, so the parent cannot finish first. Returns the parent's run key and the - /// placeholder key the child completes, or null when the parent is not running tasks (a run queued from an - /// aggregation is not a child) or is the child itself. + /// Make wait for a child that is about to be queued. Called at enqueue time, + /// inside the parent's task, so the parent cannot finish first. is the parent's + /// run key (exact, from the stamped context's RunKey) or, from older callers, its name (the newest outing). + /// Returns the parent's run key and the placeholder key the child completes, or null when the parent is + /// not running tasks (a run queued from an aggregation is not a child) or has the child's name (a run + /// re-queueing itself for its next cycle). /// - internal (string ParentRunKey, string ChildKey)? RegisterPendingChild(string parentRunName, string childRunName) + internal (string ParentRunKey, string ChildKey)? RegisterPendingChild(string parentRun, string childRunName) { - if (parentRunName == childRunName) return null; + if (parentRun == childRunName) return null; return Task.Run(async () => { - var parent = await _store.GetRunByNameAsync(parentRunName); - if (parent is not { Phase: RunPhase.Tasks }) return ((string, string)?)null; + var parent = await _store.ResolveRunAsync(parentRun); + if (parent is not { Phase: RunPhase.Tasks } || parent.Name == childRunName) return ((string, string)?)null; var childKey = $"{childRunName}|{Guid.NewGuid():N}"; if (!await _store.AddChildAsync(parent.RunKey, childKey)) return null; - _logger.LogInformation("[Orchestrator] Registered child run {Child} under parent {Parent}", childRunName, parentRunName); + _logger.LogInformation("[Orchestrator] Registered child run {Child} under parent {Parent}", childRunName, parent.Name); return (parent.RunKey, childKey); }).GetAwaiter().GetResult(); } @@ -512,34 +534,45 @@ private static Dictionary TaskInvocation(DictionaryCancel a run's pending tasks; running ones finish. Returns whether the run exists and how many were cancelled. + /// + /// Cancel the pending tasks of a run (by key) or of every unfinished run of a name; running ones finish. + /// Returns whether any run was found and how many tasks were cancelled. + /// public async Task<(bool found, int cancelledCount)> CancelRunAsync(string name) { - var header = await _store.GetRunByNameAsync(TableKeys.Sanitize(name)); - if (header == null || header.IsFinished) return (false, 0); - - await _store.RequestCancelAsync(header.RunKey); - var (cancelled, _) = await _store.CancelPendingAsync(header.RunKey); - _logger.LogInformation("[Scheduler] Run {Name} cancelled: {Cancelled} pending tasks cancelled", header.Name, cancelled); - return (true, cancelled); + var runs = await TargetRunsAsync(name); + var total = 0; + foreach (var header in runs) + { + await _store.RequestCancelAsync(header.RunKey); + var (cancelled, _) = await _store.CancelPendingAsync(header.RunKey); + total += cancelled; + _logger.LogInformation("[Scheduler] Run {Name} cancelled: {Cancelled} pending tasks cancelled", header.Name, cancelled); + } + return (runs.Count > 0, total); } /// Cancel one task that is still pending in storage. False when it is not pending (or not found). public async Task TryCancelQueuedTaskAsync(string runName, string taskId) { - var header = await _store.GetRunByNameAsync(runName); - if (header == null || header.IsFinished) return false; - var task = (await _store.GetTasksAsync(header.RunKey, 'P')).FirstOrDefault(t => t.TaskId == taskId); - if (task == null) return false; - var outcome = await _store.FinishAsync(header.RunKey, [new WorkStore.Finish(task.Seq, "Cancelled", "Cancelled by user")], 'P'); - return outcome?.Applied > 0; + foreach (var header in await TargetRunsAsync(runName)) + { + var task = (await _store.GetTasksAsync(header.RunKey, 'P')).FirstOrDefault(t => t.TaskId == taskId); + if (task == null) continue; + var outcome = await _store.FinishAsync(header.RunKey, [new WorkStore.Finish(task.Seq, "Cancelled", "Cancelled by user")], 'P'); + if (outcome?.Applied > 0) return true; + } + return false; } - /// Move a run to another priority band. Applies to the whole run: its tasks share one queue position. + /// Move a run (or every unfinished run of a name) to another priority band. Applies to whole runs: + /// a run's tasks share one queue position. public async Task ReprioritizeRunAsync(string runName, int priority) { - var header = await _store.GetRunByNameAsync(runName); - return header is { IsFinished: false } && await _store.SetPriorityAsync(header.RunKey, Math.Clamp(priority, 0, 99)); + var moved = false; + foreach (var header in await TargetRunsAsync(runName)) + moved |= await _store.SetPriorityAsync(header.RunKey, Math.Clamp(priority, 0, 99)); + return moved; } /// Cancel every pending task of every run — the whole backlog. Returns how many were cancelled. @@ -569,8 +602,8 @@ public void Cancelled(JobDescriptor descriptor) // ── lookups ── - public string? GetRunReference(string runName) => - Task.Run(() => _store.GetRunByNameAsync(runName)).GetAwaiter().GetResult()?.Reference; + public string? GetRunReference(string runName) => Task.Run(async () => + (await _store.ResolveRunAsync(runName) ?? await _store.GetRunByNameAsync(runName))?.Reference).GetAwaiter().GetResult(); public string? FindRunByReference(string reference) => Task.Run(async () => { @@ -644,12 +677,14 @@ public async Task RunStatusSweepLoopAsync(CancellationToken ct) private async Task LogRunStatusAsync(CancellationToken ct) { if (!_logger.IsEnabled(LogLevel.Information)) return; + // Running jobs are counted per run name, so runs sharing a name take them oldest first. var running = _jobManager.GetJobs(status: "Running").Where(j => j.RunName != null) .GroupBy(j => j.RunName!).ToDictionary(g => g.Key, g => g.Count()); var now = DateTime.UtcNow; await foreach (var e in _store.ReadReadyAsync(200, ct)) { - var r = running.GetValueOrDefault(e.Name); + var r = Math.Min(Math.Max(0, e.Total - e.Done), running.GetValueOrDefault(e.Name)); + running[e.Name] = running.GetValueOrDefault(e.Name) - r; var p = Math.Max(0, e.Total - e.Done - r); if (_lastStatusLog.TryGetValue(e.RunKey, out var prev) && prev.C == e.Done && prev.R == r && prev.P == p && now - prev.LoggedUtc < StatusHeartbeat) continue; diff --git a/Services/Orchestration/WorkPump.cs b/Services/Orchestration/WorkPump.cs index cf38b61..d2f972d 100644 --- a/Services/Orchestration/WorkPump.cs +++ b/Services/Orchestration/WorkPump.cs @@ -151,7 +151,7 @@ internal async Task RefillAsync(CancellationToken ct) { var name = c.Seq == WorkStore.AggregateSeq ? $"{header.Name}-PostExec" : $"{header.Name}-{c.TaskId}"; var descriptor = new JobDescriptor(header.Name, c.TaskId, header.Priority) { RunKey = c.RunKey, Seq = c.Seq, Attempt = c.Attempt }; - var jobId = _jobs.Enqueue(descriptor, name); + var jobId = _jobs.Enqueue(descriptor, name, id: $"{c.RunKey}|{c.Seq}"); _inFlight[jobId] = (c, now + _lease); } need -= claims.Count; diff --git a/Services/PowerShellHost/PowerShellRunnerService.cs b/Services/PowerShellHost/PowerShellRunnerService.cs index a08a8c1..080e43d 100644 --- a/Services/PowerShellHost/PowerShellRunnerService.cs +++ b/Services/PowerShellHost/PowerShellRunnerService.cs @@ -502,6 +502,7 @@ public async Task ExecuteScript(string functionName, Dictionary? { WorkerId = $"W{worker.Id}", RunName = parentRun, + RunKey = OperationContext.Current?.RunKey, Priority = OperationContext.Current?.Priority, Category = "Job" }; @@ -644,6 +645,7 @@ public async Task ExecuteScriptWithOutput(string functionName, Dictionar { WorkerId = $"W{worker.Id}", RunName = parentRun, + RunKey = OperationContext.Current?.RunKey, Priority = OperationContext.Current?.Priority, Category = "Planner" }; diff --git a/Services/Storage/WorkStore.cs b/Services/Storage/WorkStore.cs index acf068f..96c56e9 100644 --- a/Services/Storage/WorkStore.cs +++ b/Services/Storage/WorkStore.cs @@ -19,7 +19,8 @@ namespace Craft.Storage; /// whose lease lapses, and the next claim takes it back. /// /// Three small tables sit beside it: Ready (one row per run with work, ordered by band then start time, -/// read by the scheduler), Names (latest run per name) and Finished (completion order, for retention). +/// read by the scheduler), Names (latest run per name, and every unfinished run by name, since runs of one +/// name may overlap) and Finished (completion order, for retention). /// All three are hints derived from the Work rows: a stale one costs a read, never a wrong answer. /// public sealed class WorkStore @@ -131,6 +132,7 @@ public async Task CreateRunAsync(RunHeader header, IReadOnlyList return row == null ? null : RunHeader.FromRow(row); } + private const string ActivePartition = "A"; + + /// + /// Every unfinished run with this name, oldest first. Run keys are {name}~{hex ticks}, so one name's + /// outings share a key prefix and sort by start; the Name check drops a different name that merely + /// starts with this one and a tilde. + /// + public async Task> GetActiveRunsAsync(string name, CancellationToken ct = default) + { + var runs = new List(); + var stale = new List(); + await foreach (var row in _store.QueryRowKeyRangeAsync(_names, ActivePartition, $"{name}~", $"{name}~g", null, ct)) + { + if (row.GetString("Name") != name) continue; + if (await GetRunAsync(row.RowKey, ct) is { IsFinished: false } run) runs.Add(run); + else stale.Add(row.RowKey); + } + foreach (var key in stale) await _store.DeleteAsync(_names, ActivePartition, key, ct); + return runs; + } + + /// A run by its key, or else the newest unfinished run with that name. + public async Task ResolveRunAsync(string keyOrName, CancellationToken ct = default) => + (keyOrName.Contains('~') ? await GetRunAsync(keyOrName, ct) : null) + ?? (await GetActiveRunsAsync(keyOrName, ct)).LastOrDefault(); + /// The latest run with this name, active or finished. public async Task GetRunByNameAsync(string name, CancellationToken ct = default) { @@ -517,6 +545,7 @@ private async Task RetireAsync(RunHeader h, CancellationToken ct) { _rate.Forget(h.RunKey); await _store.DeleteAsync(_ready, ReadyPartition(h.Priority), ReadyKey(h), ct); + await _store.DeleteAsync(_names, ActivePartition, h.RunKey, ct); var done = (h.CompletedUtc ?? DateTime.UtcNow).Ticks.ToString("D19", CultureInfo.InvariantCulture); await _store.UpsertAsync(_finished, new StoreRow("F", $"{done}|{h.RunKey}") { Properties = { ["RunKey"] = h.RunKey } }, ct); } @@ -596,6 +625,7 @@ public async Task DeleteRunAsync(string runKey, CancellationToken ct = default) var header = await GetRunAsync(runKey, ct); await _store.DeletePartitionAsync(_work, runKey, ct); await _store.DeletePartitionAsync(_results, runKey, ct); + await _store.DeleteAsync(_names, ActivePartition, runKey, ct); if (header == null) return; var name = await _store.GetAsync(_names, "N", header.Name, ct); if (name?.GetString("RunKey") == runKey) await _store.DeleteAsync(_names, "N", header.Name, ct); diff --git a/tests/Craft.Tests/JobQueueStatusReaderTests.cs b/tests/Craft.Tests/JobQueueStatusReaderTests.cs index 5342bfc..b5d1e0b 100644 --- a/tests/Craft.Tests/JobQueueStatusReaderTests.cs +++ b/tests/Craft.Tests/JobQueueStatusReaderTests.cs @@ -69,6 +69,19 @@ public async Task WorkThisProcessHolds_IsNotCountedAsWaiting() Assert.Equal(3, summary.QueuedDurable); } + [Fact] + public async Task RunsSharingAName_AreSummedUnderThatName() + { + var (reader, store, _) = New(); + await Create(store, "Twin", 3, 0); + await Create(store, "Twin", 4, 1); + + var info = (await reader.GetAsync())!.ByRun["Twin"]; + + Assert.Equal(7, info.Total); + Assert.Equal(7, info.Unclaimed); + } + [Fact] public async Task AFinishedRun_LeavesTheBacklog() { diff --git a/tests/Craft.Tests/OrchestrationContractTests.cs b/tests/Craft.Tests/OrchestrationContractTests.cs index 749894a..2dd2c4a 100644 --- a/tests/Craft.Tests/OrchestrationContractTests.cs +++ b/tests/Craft.Tests/OrchestrationContractTests.cs @@ -13,7 +13,8 @@ namespace Craft.Tests; /// The orchestration contract as CIPP sees it, end to end through the real store, pump and JobManager with /// only the PowerShell calls faked: a batch becomes tasks invoked with TaskJson; their output reaches /// the PostExecution script as one JSON line each, with FunctionName and ParametersJson; a run -/// queued by a task holds its parent until it finishes; a name still running is not started twice. +/// queued by a task holds its parent until it finishes; runs of one name stack up unless the caller asks +/// for no collisions. /// public class OrchestrationContractTests { @@ -93,8 +94,9 @@ private static string Batch(int n, string prefix = "t") => JsonSerializer.Serialize(Enumerable.Range(0, n).Select(i => new { Name = "Job", TenantFilter = $"{prefix}{i}", N = i })); private static Task Start(Harness h, string name, string batch, string? postExec = null, string? postParams = null, - bool sequential = false, int priority = 4) => - h.Svc.StartFromBatchAsync(name, batch, priority, postExec, postParams, CancellationToken.None, sequential: sequential); + bool sequential = false, int priority = 4, bool allowCollision = true) => + h.Svc.StartFromBatchAsync(name, batch, priority, postExec, postParams, CancellationToken.None, sequential: sequential, + allowCollision: allowCollision); [Fact] public async Task EveryTaskRunsOnce_WithItsBatchItemAsTaskJson_AndTheRunCompletes() @@ -163,19 +165,80 @@ public async Task APostExecutionThatKeepsFailing_IsRetried_ThenTheRunFinishes() } [Fact] - public async Task ARunNameStillGoing_IsNotStartedAgain_AndItsBatchFileIsStillDeleted() + public async Task WithoutCollisions_ARunNameStillGoing_IsNotStartedAgain_AndItsBatchFileIsStillDeleted() { await using var h = await NewAsync(); - Assert.True(await Start(h, "Recurring", Batch(2))); + Assert.True(await Start(h, "Recurring", Batch(2), allowCollision: false)); var file = Path.Combine(Path.GetTempPath(), $"craft-test-{Guid.NewGuid():N}.jsonl"); await File.WriteAllTextAsync(file, "{\"Name\":\"Job\",\"TenantFilter\":\"x\"}\n"); - Assert.False(await h.Svc.StartFromBatchAsync("Recurring", "", 4, null, null, CancellationToken.None, batchFilePath: file)); + Assert.False(await h.Svc.StartFromBatchAsync("Recurring", "", 4, null, null, CancellationToken.None, batchFilePath: file, + allowCollision: false)); Assert.False(File.Exists(file)); Assert.Single(await ReadyNames(h)); Assert.True(await h.DriveUntilFinished("Recurring")); - Assert.True(await Start(h, "Recurring", Batch(1))); + Assert.True(await Start(h, "Recurring", Batch(1), allowCollision: false)); + } + + [Fact] + public async Task ByDefault_RunsOfOneNameStackUp_AndEachRunsEveryTask() + { + await using var h = await NewAsync(); + Assert.True(await Start(h, "Stacked", Batch(3), "Agg")); + Assert.True(await Start(h, "Stacked", Batch(3), "Agg")); + Assert.Equal(["Stacked", "Stacked"], await ReadyNames(h)); + + Assert.True(await h.DriveUntil(async () => (await ReadyNames(h)).Count == 0)); + Assert.Equal(6, h.Svc.Tasks.Count); + Assert.Equal(2, h.Svc.PostExecs.Count); + Assert.All(h.Svc.PostExecs, p => Assert.Equal(3, p.Lines.Length)); + // Same task ids in both runs, so the jobs share a display name; each must still be its own job, or + // the pump would track (and renew the lease of) only one of the two claims. + Assert.Equal(2, h.Jobs.GetJobs().Count(j => j.Name == "Stacked-Job_t0")); + } + + [Fact] + public async Task ACollisionFreeStart_IsSkippedWhileAStackedRunOfThatNameIsGoing() + { + await using var h = await NewAsync(); + Assert.True(await Start(h, "Mixed", Batch(1))); + Assert.True(await Start(h, "Mixed", Batch(1))); + Assert.False(await Start(h, "Mixed", Batch(1), allowCollision: false)); + } + + [Fact] + public async Task AChildFindsItsExactParent_ByRunKey_WhenRunsOfThatNameOverlap() + { + await using var h = await NewAsync(); + Assert.True(await Start(h, "Parent", Batch(1, "older"), "Agg")); + Assert.True(await Start(h, "Parent", Batch(1, "newer"), "Agg")); + var outings = await h.Store.GetActiveRunsAsync("Parent"); + var (older, newest) = (outings[0], outings[1]); + + var link = h.Svc.RegisterPendingChild(older.RunKey, "Child"); + Assert.Equal(older.RunKey, link!.Value.ParentRunKey); + Assert.Equal(2, (await h.Store.GetRunAsync(older.RunKey))!.Total); + Assert.Equal(1, (await h.Store.GetRunAsync(newest.RunKey))!.Total); + + // An older caller passing only the name gets the newest outing. + Assert.Equal(newest.RunKey, h.Svc.RegisterPendingChild("Parent", "Other")!.Value.ParentRunKey); + // A run re-queueing itself is never its own child, by key or by name. + Assert.Null(h.Svc.RegisterPendingChild(older.RunKey, "Parent")); + } + + [Fact] + public async Task CancellingByName_CancelsEveryRunOfThatName() + { + await using var h = await NewAsync(); + Assert.True(await Start(h, "Twin", Batch(4))); + Assert.True(await Start(h, "Twin", Batch(4))); + + var (found, cancelled) = await h.Svc.CancelRunAsync("Twin"); + + Assert.True(found); + Assert.Equal(8, cancelled); + Assert.Empty(await h.Store.GetActiveRunsAsync("Twin")); } [Fact] diff --git a/tests/Craft.Tests/OrchestratorBatchStreamingTests.cs b/tests/Craft.Tests/OrchestratorBatchStreamingTests.cs index dd8de03..b5d3965 100644 --- a/tests/Craft.Tests/OrchestratorBatchStreamingTests.cs +++ b/tests/Craft.Tests/OrchestratorBatchStreamingTests.cs @@ -175,8 +175,8 @@ public void MissingFile_YieldsNoTasks() /// /// Cleanup is the caller's, on every path — including the ones that never parse. /// - /// StartFromBatchAsync returns early when a run of the same name is already in progress or already - /// active, and neither return looks at the batch. Those are the common outcome for a duplicate + /// StartFromBatchAsync returns early when collisions are off and a run of the same name is already in + /// progress or active, and neither return looks at the batch. Those are the common outcome for a duplicate /// enqueue, so cleanup living at the parse site would leave the container's temp directory /// accumulating the batches of every run that was skipped rather than started. This pins the /// deletion to the outer method by driving it through those early returns. @@ -201,7 +201,7 @@ public async Task BatchFileIsDeleted_EvenWhenTheRunIsSkippedWithoutParsing() var path = WriteLines(["""{"FunctionName":"A","TenantFilter":"a.com"}"""]); await svc.StartFromBatchAsync("busy-run", string.Empty, 4, null, null, - CancellationToken.None, null, null, path); + CancellationToken.None, null, null, path, allowCollision: false); Assert.False(File.Exists(path), "the batch file outlived a skipped run — every enqueue that is skipped now leaks a temp file"); From d00b4b7bb234e9271379e8645413eaf5573339e7 Mon Sep 17 00:00:00 2001 From: Zacgoose <107489668+Zacgoose@users.noreply.github.com> Date: Tue, 6 Oct 2026 00:32:03 +0800 Subject: [PATCH 07/24] feat(runtime): pass AllowCollision and the parent run key from Start-CraftOrchestrator --- Runtime/CraftRuntime/Start-CraftOrchestrator.ps1 | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/Runtime/CraftRuntime/Start-CraftOrchestrator.ps1 b/Runtime/CraftRuntime/Start-CraftOrchestrator.ps1 index 7e7daac..fd7a370 100644 --- a/Runtime/CraftRuntime/Start-CraftOrchestrator.ps1 +++ b/Runtime/CraftRuntime/Start-CraftOrchestrator.ps1 @@ -30,6 +30,9 @@ function Start-CraftOrchestrator { worker and runs every step on it to completion, without going back to the pool between steps. A step that fails is recorded and the run carries on with the next (best-effort). + - AllowCollision (bool) — optional, default $true: runs of one name stack up side by side. + $false skips this run while another run of the same name is still + going (recurring work that must not pile up). .EXAMPLE # Fan-out (default): every task is queued up front and drained in parallel by the worker pool. @@ -140,11 +143,13 @@ function Start-CraftOrchestrator { # Lineage: pass the enclosing run explicitly. The bridge's own ambient read is null for calls # made from the pipeline thread — which is exactly where this function runs — so without this # a parent run would finalize (and dispatch its PostExecution) before its child runs complete. - $ParentRunName = $OpContext.RunName + # RunKey names the exact run when several runs share a name. + $ParentRunName = $OpContext.RunKey ?? $OpContext.RunName # Sequential mode: PowerShell marshals absent/$false to $false. When set, the orchestrator queues the # batch one task at a time in payload order rather than fanning out. $Sequential = [bool]($InputObject.Sequential) + $AllowCollision = $InputObject.AllowCollision -ne $false Write-Information "Craft: Queuing orchestrator '$OrchestratorName' ($TaskCount tasks, P$Priority$(if ($Sequential) { ', Sequential' })$(if ($PostExecFunctionName) { ", PostExec: $PostExecFunctionName" })$(if ($ParentRunName) { ", Parent: $ParentRunName" }))" [Craft.Services.OrchestratorBridge]::QueueOrchestrationFromFile( @@ -155,7 +160,8 @@ function Start-CraftOrchestrator { $PostExecParametersJson, $InputObject.Reference, $ParentRunName, - $Sequential + $Sequential, + $AllowCollision ) return "Craft-$OrchestratorName" } From a4d72cf8e1c640e987483734b9b3a0049db30b85 Mon Sep 17 00:00:00 2001 From: Zacgoose <107489668+Zacgoose@users.noreply.github.com> Date: Tue, 6 Oct 2026 00:49:41 +0800 Subject: [PATCH 08/24] feat(orchestration): report runs skipped because a run of that name is active - OrchestratorBridge.IsRunActive(name) lets a caller check before queueing - a start skipped at drain time logs a warning - Start-CraftOrchestrator skips with a warning and returns "-Skipped" when AllowCollision is false and the name is active --- .../CraftRuntime/Start-CraftOrchestrator.ps1 | 8 ++- Services/Bridges/OrchestratorBridge.cs | 13 ++++ Services/Orchestration/OrchestratorService.cs | 5 +- .../OrchestratorBridgeLineageTests.cs | 70 ++++++++++++++----- 4 files changed, 76 insertions(+), 20 deletions(-) diff --git a/Runtime/CraftRuntime/Start-CraftOrchestrator.ps1 b/Runtime/CraftRuntime/Start-CraftOrchestrator.ps1 index fd7a370..3f6bc82 100644 --- a/Runtime/CraftRuntime/Start-CraftOrchestrator.ps1 +++ b/Runtime/CraftRuntime/Start-CraftOrchestrator.ps1 @@ -71,6 +71,13 @@ function Start-CraftOrchestrator { $OrchestratorName = $InputObject.OrchestratorName ?? 'UnnamedOrchestrator' + # Collisions off: a run of this name that is still going wins, and this one is skipped up front. + $AllowCollision = $InputObject.AllowCollision -ne $false + if (-not $AllowCollision -and [Craft.Services.OrchestratorBridge]::IsRunActive($OrchestratorName)) { + Write-Warning "Craft: Skipped orchestrator '$OrchestratorName' - a run with this name is still active" + return "Craft-$OrchestratorName-Skipped" + } + # QueueFunction pattern: call the function first to generate batch items if (-not $InputObject.Batch -and $InputObject.QueueFunction) { $QueueFuncName = "Push-$($InputObject.QueueFunction.FunctionName)" @@ -149,7 +156,6 @@ function Start-CraftOrchestrator { # Sequential mode: PowerShell marshals absent/$false to $false. When set, the orchestrator queues the # batch one task at a time in payload order rather than fanning out. $Sequential = [bool]($InputObject.Sequential) - $AllowCollision = $InputObject.AllowCollision -ne $false Write-Information "Craft: Queuing orchestrator '$OrchestratorName' ($TaskCount tasks, P$Priority$(if ($Sequential) { ', Sequential' })$(if ($PostExecFunctionName) { ", PostExec: $PostExecFunctionName" })$(if ($ParentRunName) { ", Parent: $ParentRunName" }))" [Craft.Services.OrchestratorBridge]::QueueOrchestrationFromFile( diff --git a/Services/Bridges/OrchestratorBridge.cs b/Services/Bridges/OrchestratorBridge.cs index f7e4d68..acc4f5d 100644 --- a/Services/Bridges/OrchestratorBridge.cs +++ b/Services/Bridges/OrchestratorBridge.cs @@ -102,6 +102,19 @@ public static void QueueOrchestrationFromFile(string name, string batchFilePath, private static (string ParentRunKey, string ChildKey)? RegisterPendingChild(string? parentRunName, string childName) => string.IsNullOrEmpty(parentRunName) ? null : s_service?.RegisterPendingChild(parentRunName, childName); + /// + /// Whether a run of this name is unfinished or already queued here to start. Lets a caller that does not + /// want overlapping runs (allowCollision: false) skip, and say so, before building the batch. The + /// start itself checks again, so a run that appears in between is still skipped. + /// PS usage: [Craft.Services.OrchestratorBridge]::IsRunActive($name). + /// + public static bool IsRunActive(string name) + { + name = TableKeys.Sanitize(name); + if (s_pending.Any(p => p.Name == name)) return true; + return s_service != null && Task.Run(() => s_service.IsRunActiveAsync(name)).GetAwaiter().GetResult(); + } + /// Synchronous drain — blocks until all pending orchestrations are started. public static void DrainPending() { diff --git a/Services/Orchestration/OrchestratorService.cs b/Services/Orchestration/OrchestratorService.cs index df95d51..9c2b794 100644 --- a/Services/Orchestration/OrchestratorService.cs +++ b/Services/Orchestration/OrchestratorService.cs @@ -169,7 +169,7 @@ public async Task StartFromBatchAsync(string name, string batchJson, int p gated = _activePlanners.TryAdd(name, true); if (!gated || await IsActiveAsync(name, ct)) { - _logger.LogInformation("[Orchestrator] Run {Name} already active, skipping", name); + _logger.LogWarning("[Orchestrator] Run {Name} skipped: a run of that name is still active and collisions are off", name); return false; } } @@ -210,6 +210,9 @@ await CreateAsync(name, tasks, priority, genericTaskFunc, postExecFunctionName, private async Task IsActiveAsync(string name, CancellationToken ct) => (await _store.GetActiveRunsAsync(name, ct)).Count > 0; + /// Whether any run with this name is unfinished. + public Task IsRunActiveAsync(string name, CancellationToken ct = default) => IsActiveAsync(TableKeys.Sanitize(name), ct); + /// The runs an operator action names: the run with that key, or every unfinished run of that name. private async Task> TargetRunsAsync(string keyOrName) { diff --git a/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs b/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs index daecbc9..6dd2b0f 100644 --- a/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs +++ b/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs @@ -177,24 +177,8 @@ public async Task Drain_ReleasesTheParent_WhenTheChildIsNeverCreated() { // The deadlock-avoidance guarantee: a child registered at enqueue that then fails to start (here an // empty batch) must stop holding its parent, or the parent never reaches its barrier. - var settings = new Craft.Configuration.CraftSettings(); - var config = new ConfigurationBuilder().AddInMemoryCollection([]).Build(); - var repo = new ScriptRepository(NullLogger.Instance, settings); - var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); - var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); - var jobs = new Craft.Orchestration.JobManager(NullLogger.Instance, settings, limiter); - var mem = new MemoryTableStore(); - var store = new Craft.Storage.WorkStore(NullLogger.Instance, settings, mem); - var svc = new OrchestratorService(NullLogger.Instance, null!, limiter, jobs, store, - new Craft.Storage.ResultStore(NullLogger.Instance, settings, mem), config, settings); - var started = DateTime.UtcNow; - var parent = await store.CreateRunAsync(new Craft.Storage.RunHeader - { - RunKey = Craft.Storage.WorkStore.RunKeyFor("LineageDrainParent", started), - Name = "LineageDrainParent", - StartedUtc = started, - TaskScriptName = "Invoke-CraftTask", - }, [new Craft.Storage.WorkStore.NewTask("t0", new())]); + var (svc, store) = NewStoreBackedService(); + var parent = await CreateRunAsync(store, "LineageDrainParent"); var previousService = s_serviceField.GetValue(null); try @@ -219,4 +203,54 @@ public async Task Drain_ReleasesTheParent_WhenTheChildIsNeverCreated() s_serviceField.SetValue(null, previousService); } } + + [Fact] + public async Task IsRunActive_SeesUnfinishedRunsAndRunsQueuedToStart_SoACallerCanSkipAndSaySo() + { + var (svc, store) = NewStoreBackedService(); + await CreateRunAsync(store, "LineageActiveRun"); + + var previousService = s_serviceField.GetValue(null); + try + { + OrchestratorBridge.Initialize(svc); + Assert.True(OrchestratorBridge.IsRunActive("LineageActiveRun")); + Assert.False(OrchestratorBridge.IsRunActive("LineageNoSuchRun")); + + OrchestratorBridge.QueueOrchestration("LineageQueuedRun", "[]", 4); + Assert.True(OrchestratorBridge.IsRunActive("LineageQueuedRun")); + Assert.NotNull(TakePending("LineageQueuedRun")); + } + finally + { + s_serviceField.SetValue(null, previousService); + } + } + + private static (OrchestratorService Service, Craft.Storage.WorkStore Store) NewStoreBackedService() + { + var settings = new Craft.Configuration.CraftSettings(); + var config = new ConfigurationBuilder().AddInMemoryCollection([]).Build(); + var repo = new ScriptRepository(NullLogger.Instance, settings); + var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); + var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); + var jobs = new Craft.Orchestration.JobManager(NullLogger.Instance, settings, limiter); + var mem = new MemoryTableStore(); + var store = new Craft.Storage.WorkStore(NullLogger.Instance, settings, mem); + var svc = new OrchestratorService(NullLogger.Instance, null!, limiter, jobs, store, + new Craft.Storage.ResultStore(NullLogger.Instance, settings, mem), config, settings); + return (svc, store); + } + + private static Task CreateRunAsync(Craft.Storage.WorkStore store, string name) + { + var started = DateTime.UtcNow; + return store.CreateRunAsync(new Craft.Storage.RunHeader + { + RunKey = Craft.Storage.WorkStore.RunKeyFor(name, started), + Name = name, + StartedUtc = started, + TaskScriptName = "Invoke-CraftTask", + }, [new Craft.Storage.WorkStore.NewTask("t0", new())]); + } } From 4f902e1f7f3afd83441a67efafb3b17968b515e6 Mon Sep 17 00:00:00 2001 From: Zacgoose <107489668+Zacgoose@users.noreply.github.com> Date: Tue, 6 Oct 2026 01:33:34 +0800 Subject: [PATCH 09/24] feat(orchestration): add MaxConcurrency and sequential StopOnFailure, and stop blocked runs starving the queue - MaxConcurrency (0 = no limit) caps a run's tasks running at once; the pump applies it against the claims it holds, at no storage cost; dropped for sequential runs - StopOnFailure makes a sequential run cancel its remaining steps at the first failure (including a step that keeps crashing); carrying on stays the default - a run whose pending tasks ran out is skipped until its counts move, a claim is released, or its earliest lease is due; other empty runs back off 30 s doubling to 15 min, so runs waiting at the head no longer starve the runs behind them - Ready rows carry the run's mode and are read 1,000 to a page; the running-row scan per claim is bounded - bridge and Start-CraftOrchestrator take maxConcurrency and stopOnFailure - tests: shared harness, mode tests, storage-cost pins, Azurite end-to-end runs, wrapper inheritance pins --- .../CraftRuntime/Start-CraftOrchestrator.ps1 | 10 +- Services/Bridges/OrchestratorBridge.cs | 24 +- Services/Orchestration/OrchestratorService.cs | 54 +++- Services/Orchestration/WorkPump.cs | 98 +++++-- Services/Storage/WorkStore.cs | 71 ++++- tests/Craft.Tests/CountingTableStore.cs | 127 ++++++++ tests/Craft.Tests/MemoryTableStore.cs | 88 ++++-- .../Craft.Tests/OrchestrationAzuriteTests.cs | 146 +++++++++ .../Craft.Tests/OrchestrationContractTests.cs | 94 +----- tests/Craft.Tests/OrchestrationCostTests.cs | 276 ++++++++++++++++++ tests/Craft.Tests/OrchestrationHarness.cs | 145 +++++++++ tests/Craft.Tests/OrchestrationModeTests.cs | 221 ++++++++++++++ .../OrchestratorBridgeLineageTests.cs | 83 ++++++ tests/Craft.Tests/WorkStoreTests.cs | 48 +++ 14 files changed, 1323 insertions(+), 162 deletions(-) create mode 100644 tests/Craft.Tests/CountingTableStore.cs create mode 100644 tests/Craft.Tests/OrchestrationAzuriteTests.cs create mode 100644 tests/Craft.Tests/OrchestrationCostTests.cs create mode 100644 tests/Craft.Tests/OrchestrationHarness.cs create mode 100644 tests/Craft.Tests/OrchestrationModeTests.cs diff --git a/Runtime/CraftRuntime/Start-CraftOrchestrator.ps1 b/Runtime/CraftRuntime/Start-CraftOrchestrator.ps1 index 3f6bc82..72aa4ce 100644 --- a/Runtime/CraftRuntime/Start-CraftOrchestrator.ps1 +++ b/Runtime/CraftRuntime/Start-CraftOrchestrator.ps1 @@ -33,6 +33,10 @@ function Start-CraftOrchestrator { - AllowCollision (bool) — optional, default $true: runs of one name stack up side by side. $false skips this run while another run of the same name is still going (recurring work that must not pile up). + - MaxConcurrency (int) — optional, default 0 (no limit): at most this many of the run's tasks + run at once. Not used with Sequential. + - StopOnFailure (bool) — optional, Sequential only: the first failed step cancels the steps + after it. By default a sequential run carries on past a failure. .EXAMPLE # Fan-out (default): every task is queued up front and drained in parallel by the worker pool. @@ -156,6 +160,8 @@ function Start-CraftOrchestrator { # Sequential mode: PowerShell marshals absent/$false to $false. When set, the orchestrator queues the # batch one task at a time in payload order rather than fanning out. $Sequential = [bool]($InputObject.Sequential) + $MaxConcurrency = [int]($InputObject.MaxConcurrency ?? 0) + $StopOnFailure = [bool]($InputObject.StopOnFailure) Write-Information "Craft: Queuing orchestrator '$OrchestratorName' ($TaskCount tasks, P$Priority$(if ($Sequential) { ', Sequential' })$(if ($PostExecFunctionName) { ", PostExec: $PostExecFunctionName" })$(if ($ParentRunName) { ", Parent: $ParentRunName" }))" [Craft.Services.OrchestratorBridge]::QueueOrchestrationFromFile( @@ -167,7 +173,9 @@ function Start-CraftOrchestrator { $InputObject.Reference, $ParentRunName, $Sequential, - $AllowCollision + $AllowCollision, + $MaxConcurrency, + $StopOnFailure ) return "Craft-$OrchestratorName" } diff --git a/Services/Bridges/OrchestratorBridge.cs b/Services/Bridges/OrchestratorBridge.cs index acc4f5d..128b6ac 100644 --- a/Services/Bridges/OrchestratorBridge.cs +++ b/Services/Bridges/OrchestratorBridge.cs @@ -25,9 +25,14 @@ public static class OrchestratorBridge /// True (the default) lets runs of one name stack up; false skips this run /// while another run of the same name is unfinished. + /// At most this many of the run's tasks run at once; 0 (the default) is no + /// limit. Ignored for a sequential run. + /// Sequential runs only: the first failed step cancels the rest instead of the + /// run carrying on (the default). public static void QueueOrchestration(string name, string batchJson, int priority, string? postExecFunctionName = null, string? postExecParametersJson = null, - string? reference = null, string? parentRunName = null, bool sequential = false, bool allowCollision = true) + string? reference = null, string? parentRunName = null, bool sequential = false, bool allowCollision = true, + int maxConcurrency = 0, bool stopOnFailure = false) { // Sanitized here as well as at run creation so the child-run registration below // records the SAME name the service ends up creating — a raw name with a table-illegal @@ -38,7 +43,7 @@ public static void QueueOrchestration(string name, string batchJson, int priorit s_pending.Enqueue(new PendingOrchestration(name, batchJson, priority, postExecFunctionName, postExecParametersJson, parentRunName, reference, Sequential: sequential, ParentRunKey: child?.ParentRunKey, ChildKey: child?.ChildKey, - AllowCollision: allowCollision)); + AllowCollision: allowCollision, MaxConcurrency: maxConcurrency, StopOnFailure: stopOnFailure)); } /// @@ -54,9 +59,14 @@ public static void QueueOrchestration(string name, string batchJson, int priorit /// /// True (the default) lets runs of one name stack up; false skips this run /// while another run of the same name is unfinished. + /// At most this many of the run's tasks run at once; 0 (the default) is no + /// limit. Ignored for a sequential run. + /// Sequential runs only: the first failed step cancels the rest instead of the + /// run carrying on (the default). public static void QueueOrchestrationFromFile(string name, string batchFilePath, int priority, string? postExecFunctionName = null, string? postExecParametersJson = null, - string? reference = null, string? parentRunName = null, bool sequential = false, bool allowCollision = true) + string? reference = null, string? parentRunName = null, bool sequential = false, bool allowCollision = true, + int maxConcurrency = 0, bool stopOnFailure = false) { name = TableKeys.Sanitize(name); parentRunName = ResolveParentRunName(name, parentRunName); @@ -64,7 +74,7 @@ public static void QueueOrchestrationFromFile(string name, string batchFilePath, s_pending.Enqueue(new PendingOrchestration(name, string.Empty, priority, postExecFunctionName, postExecParametersJson, parentRunName, reference, batchFilePath, Sequential: sequential, ParentRunKey: child?.ParentRunKey, ChildKey: child?.ChildKey, - AllowCollision: allowCollision)); + AllowCollision: allowCollision, MaxConcurrency: maxConcurrency, StopOnFailure: stopOnFailure)); } /// @@ -139,7 +149,8 @@ private static async Task StartAsync(PendingOrchestration p) if (s_service == null) { DiscardUndispatchable(p); return; } created = await s_service.StartFromBatchAsync(p.Name, p.BatchJson, p.Priority, p.PostExecFunctionName, p.PostExecParametersJson, CancellationToken.None, - p.ParentRunName, p.Reference, p.BatchFilePath, p.Sequential, p.ParentRunKey, p.ChildKey, p.AllowCollision); + p.ParentRunName, p.Reference, p.BatchFilePath, p.Sequential, p.ParentRunKey, p.ChildKey, p.AllowCollision, + p.MaxConcurrency, p.StopOnFailure); } catch (Exception ex) { @@ -180,7 +191,8 @@ private static void DiscardUndispatchable(PendingOrchestration p) public record PendingOrchestration(string Name, string BatchJson, int Priority, string? PostExecFunctionName, string? PostExecParametersJson, string? ParentRunName, string? Reference = null, string? BatchFilePath = null, bool Sequential = false, - string? ParentRunKey = null, string? ChildKey = null, bool AllowCollision = true) + string? ParentRunKey = null, string? ChildKey = null, bool AllowCollision = true, int MaxConcurrency = 0, + bool StopOnFailure = false) { public bool PendingChildRegistered => ChildKey != null; } diff --git a/Services/Orchestration/OrchestratorService.cs b/Services/Orchestration/OrchestratorService.cs index 9c2b794..132285f 100644 --- a/Services/Orchestration/OrchestratorService.cs +++ b/Services/Orchestration/OrchestratorService.cs @@ -122,7 +122,7 @@ public async Task StartOrResumeRun(string name, string plannerPath, string taskP } await CreateAsync(name, tasks, priority, Path.GetFileNameWithoutExtension(taskPath), null, null, null, null, null, - false, ct); + new RunMode(false, 0, false), ct); } finally { @@ -154,11 +154,15 @@ public async Task StartPlannerRunAsync(string command, int priority, Cancellatio /// (, deleted on every path) or a JSON array string. Returns whether a run /// was created: false when the batch is empty, or when is false and a /// run of this name is still going. By default runs of one name stack up side by side. + /// above 0 caps how many of the run's tasks run at once; it has no meaning + /// for a sequential run (one step at a time already) and is dropped there. + /// makes a sequential run cancel its remaining steps at the first failure; other runs always carry on. /// public async Task StartFromBatchAsync(string name, string batchJson, int priority, string? postExecFunctionName, string? postExecParametersJson, CancellationToken ct, string? parentRunName = null, string? reference = null, string? batchFilePath = null, - bool sequential = false, string? parentRunKey = null, string? childKey = null, bool allowCollision = true) + bool sequential = false, string? parentRunKey = null, string? childKey = null, bool allowCollision = true, + int maxConcurrency = 0, bool stopOnFailure = false) { var gated = false; try @@ -193,7 +197,7 @@ public async Task StartFromBatchAsync(string name, string batchJson, int p if (string.IsNullOrEmpty(postExecFunctionName)) postExecFunctionName = null; if (string.IsNullOrEmpty(postExecParametersJson)) postExecParametersJson = null; await CreateAsync(name, tasks, priority, genericTaskFunc, postExecFunctionName, postExecParametersJson, - reference, parentRunKey, childKey, sequential, ct); + reference, parentRunKey, childKey, ResolveMode(name, sequential, maxConcurrency, stopOnFailure), ct); return true; } finally @@ -221,9 +225,30 @@ private async Task> TargetRunsAsync(string keyOrName) return await _store.GetActiveRunsAsync(keyOrName); } + /// How a run's tasks are scheduled: one at a time on a pinned worker (sequential), at most N at once, + /// or all at once; and whether a sequential run stops at its first failure. + internal readonly record struct RunMode(bool Sequential, int MaxConcurrency, bool StopOnFailure); + + private RunMode ResolveMode(string name, bool sequential, int maxConcurrency, bool stopOnFailure) + { + maxConcurrency = Math.Max(0, maxConcurrency); + if (sequential && maxConcurrency > 0) + { + _logger.LogWarning("[Orchestrator] Run {Name}: MaxConcurrency {Max} ignored, a sequential run already runs one step at a time", + name, maxConcurrency); + maxConcurrency = 0; + } + if (stopOnFailure && !sequential) + { + _logger.LogWarning("[Orchestrator] Run {Name}: StopOnFailure ignored, it applies to sequential runs only", name); + stopOnFailure = false; + } + return new RunMode(sequential, maxConcurrency, stopOnFailure); + } + private async Task CreateAsync(string name, List tasks, int priority, string taskScriptName, string? postExecFunctionName, string? postExecParametersJson, string? reference, string? parentRunKey, - string? childKey, bool sequential, CancellationToken ct) + string? childKey, RunMode mode, CancellationToken ct) { var started = NextStartTime(); var header = new RunHeader @@ -238,13 +263,16 @@ private async Task CreateAsync(string name, List tasks, in Reference = reference, ParentRunKey = parentRunKey, ParentChildKey = childKey, - Sequential = sequential, + Sequential = mode.Sequential, + MaxConcurrency = mode.MaxConcurrency, + StopOnFailure = mode.StopOnFailure, }; await _store.CreateRunAsync(header, tasks.Select(t => new WorkStore.NewTask(t.Id, t.Parameters)).ToList(), ct); - _logger.LogInformation("[Orchestrator] Run {Name} created with {Count} tasks at P{Priority}{PostExec}{Sequential}", + _logger.LogInformation("[Orchestrator] Run {Name} created with {Count} tasks at P{Priority}{PostExec}{Mode}", name, tasks.Count, header.Priority, postExecFunctionName != null ? $" (PostExec: Push-{postExecFunctionName})" : "", - sequential ? " (sequential)" : ""); + mode.Sequential ? (mode.StopOnFailure ? " (sequential, stop on failure)" : " (sequential)") + : mode.MaxConcurrency > 0 ? $" (max {mode.MaxConcurrency} at once)" : ""); } private static long s_lastStartTicks; @@ -473,7 +501,7 @@ private Func BuildSequentialRunWork(RunHeader header, J if (current.CancelRequested) { await FinishAsync(header, step.Seq, "Cancelled", "Cancelled by user"); - await _store.CancelPendingAsync(header.RunKey, jobCt); + await _store.CancelPendingAsync(header.RunKey, ct: jobCt); break; } if (step.Seq == WorkStore.AggregateSeq) @@ -498,8 +526,14 @@ private Func BuildSequentialRunWork(RunHeader header, J } catch (Exception ex) { - _logger.LogError(ex, "[Scheduler] Sequential task failed: {TaskId} — continuing with the next step", step.TaskId); await FinishAsync(header, step.Seq, "Failed", ex.Message); + if (current.StopOnFailure) + { + _logger.LogError(ex, "[Scheduler] Sequential task failed: {TaskId} — stopping the run", step.TaskId); + await _store.CancelPendingAsync(header.RunKey, WorkStore.StoppedReason(step.TaskId), jobCt); + break; + } + _logger.LogError(ex, "[Scheduler] Sequential task failed: {TaskId} — continuing with the next step", step.TaskId); } if (await _store.ClaimSequentialAsync(header.RunKey, Owner, Lease, continuing: true, jobCt) is not { } next) break; @@ -587,7 +621,7 @@ public async Task ClearQueueAsync(CancellationToken ct = default) foreach (var e in entries) { await _store.RequestCancelAsync(e.RunKey, ct); - total += (await _store.CancelPendingAsync(e.RunKey, ct)).Cancelled; + total += (await _store.CancelPendingAsync(e.RunKey, ct: ct)).Cancelled; } _logger.LogWarning("[JobQueue] Durable queue cleared — {Count} queued task(s) cancelled", total); return total; diff --git a/Services/Orchestration/WorkPump.cs b/Services/Orchestration/WorkPump.cs index d2f972d..d857786 100644 --- a/Services/Orchestration/WorkPump.cs +++ b/Services/Orchestration/WorkPump.cs @@ -25,18 +25,30 @@ public class WorkPump : BackgroundService private readonly TimeSpan _idlePollInterval; private readonly Task? _claimGate; - /// How long a run that had nothing claimable is skipped while its counts stand still, and how often a - /// run's lapsed leases are looked for. + /// + /// Runs the pump has written off for now, so they cost no storage read until something can have changed. + /// + /// A run whose pending tasks ran out (its remaining work is running, or it waits on child runs) can only + /// have claimable work again when its counts move (a task or child finishes, its aggregation falls due), + /// when a claim is released back to pending, or when a claim lapses unrenewed. So it is skipped until its + /// Ready counts change, a release names it, or its earliest claim's lease runs out. Without this, runs + /// waiting at the head of a large queue spend the per-refill read budget over and over and starve every run + /// behind them. + /// + /// A run that came back empty for another reason (a lost race, a busy sequential driver) is skipped for + /// 30 s, doubling each time it is found empty again, up to 15 minutes, until its counts move. + /// + private readonly Dictionary _skip = new(StringComparer.Ordinal); + private readonly System.Collections.Concurrent.ConcurrentQueue _released = new(); private static readonly TimeSpan EmptyBackoff = TimeSpan.FromSeconds(30); - private static readonly TimeSpan ExpiredCheckInterval = TimeSpan.FromSeconds(60); + private static readonly TimeSpan MaxEmptyBackoff = TimeSpan.FromMinutes(15); - /// Runs read from storage per refill. Skipped runs cost nothing, so a head of runs that are all waiting - /// (on children, or on their own running tasks) cannot hide the runs behind it. + /// Runs read from storage per refill. Skipped runs cost nothing. private const int MaxRunsReadPerRefill = 32; - private const int ReadyPageSize = 100; - private readonly Dictionary _emptyUntil = new(StringComparer.Ordinal); - private readonly Dictionary _expiredCheckedAt = new(StringComparer.Ordinal); + /// Ready rows per page: the whole list of a normal instance in one request, and few requests when a + /// large backlog has to be scanned past. + private const int ReadyPageSize = 1000; /// Claims handed to the JobManager, by job id, with when their lease runs out. private readonly Dictionary _inFlight = new(StringComparer.Ordinal); @@ -56,6 +68,7 @@ public WorkPump(ILogger logger, WorkStore store, JobManager jobs, ICon _pollInterval = TimeSpan.FromMilliseconds(Math.Max(100, configuration.GetValue("JobQueuePollIntervalMs", 1000))); _idlePollInterval = TimeSpan.FromMilliseconds(Math.Max(_pollInterval.TotalMilliseconds, configuration.GetValue("JobQueueIdlePollIntervalMs", 10_000))); + _store.Released += _released.Enqueue; } protected override async Task ExecuteAsync(CancellationToken stoppingToken) @@ -75,7 +88,6 @@ protected override async Task ExecuteAsync(CancellationToken stoppingToken) var claimed = 0; try { - Forget(); claimed = await RefillAsync(stoppingToken); await RenewAsync(stoppingToken); } @@ -95,7 +107,10 @@ protected override async Task ExecuteAsync(CancellationToken stoppingToken) } } - internal void ForgetBackoff() => _emptyUntil.Clear(); + internal void ForgetBackoff() => _skip.Clear(); + + /// The pump's notion of now, for its backoff and renewal timing; tests replace it. + internal Func Clock { get; set; } = () => DateTime.UtcNow; /// Stop tracking claims the JobManager is done with. Their finish (or release) was written by the job. private void Forget() @@ -104,20 +119,42 @@ private void Forget() _inFlight.Remove(id); } - /// Claim until the buffer holds a batch. Returns how many tasks were claimed. + /// + /// Claim until the buffer holds a batch. Returns how many tasks were claimed. + /// + /// A run's concurrency limit is applied here, against the claims this process holds: only one Craft + /// instance works the tables, so what it holds is what is running. A claim left by a process that has + /// since died is not running, so it rightly does not count. During an overlapping restart, two processes + /// could briefly run up to the limit each. + /// internal async Task RefillAsync(CancellationToken ct) { + Forget(); + while (_released.TryDequeue(out var releasedRun)) _skip.Remove(releasedRun); if (_jobs.QueuedCount > _lowWater) return 0; var need = _batchSize - _jobs.QueuedCount; var claimed = 0; - var now = DateTime.UtcNow; + var now = Clock(); var read = 0; + Dictionary? held = null; await foreach (var entry in _store.ReadReadyAsync(ReadyPageSize, ct)) { if (need <= 0 || read >= MaxRunsReadPerRefill) break; - if (_emptyUntil.TryGetValue(entry.RunKey, out var skip) && skip.Until > now - && skip.Done == entry.Done && skip.Total == entry.Total) continue; + var progressed = true; + if (_skip.TryGetValue(entry.RunKey, out var skip) && skip.Done == entry.Done && skip.Total == entry.Total) + { + if (skip.Until > now) continue; + progressed = false; + } + + var want = need; + if (entry.MaxConcurrency > 0 && !entry.Sequential) + { + held ??= _inFlight.Values.GroupBy(v => v.Claim.RunKey).ToDictionary(g => g.Key, g => g.Count(), StringComparer.Ordinal); + want = Math.Min(need, entry.MaxConcurrency - held.GetValueOrDefault(entry.RunKey)); + if (want <= 0) continue; + } read++; var header = await _store.GetRunAsync(entry.RunKey, ct); @@ -128,6 +165,7 @@ internal async Task RefillAsync(CancellationToken ct) } IReadOnlyList claims; + var probe = new WorkStore.ClaimProbe(); if (header.Sequential) { var step = await _store.ClaimSequentialAsync(header.RunKey, _owner, _lease, ct: ct); @@ -135,17 +173,26 @@ internal async Task RefillAsync(CancellationToken ct) } else { - var checkExpired = !_expiredCheckedAt.TryGetValue(header.RunKey, out var at) || now - at >= ExpiredCheckInterval; - if (checkExpired) _expiredCheckedAt[header.RunKey] = now; - claims = await _store.ClaimAsync(header.RunKey, need, _owner, _lease, checkExpired, ct); + claims = await _store.ClaimAsync(header.RunKey, want, _owner, _lease, reclaimExpired: true, probe, ct); } - if (claims.Count == 0) + if (probe.PendingExhausted) { - _emptyUntil[header.RunKey] = (now + EmptyBackoff, entry.Done, entry.Total); + // A lapsing claim is the one change that bumps no count; look again when the earliest lease is due. + _skip[header.RunKey] = (entry.Done, entry.Total, probe.EarliestLeaseUntil?.UtcDateTime ?? DateTime.MaxValue, 0); + } + else if (claims.Count == 0) + { + var strikes = progressed ? 0 : skip.Strikes + 1; + var backoff = TimeSpan.FromTicks(Math.Min(MaxEmptyBackoff.Ticks, EmptyBackoff.Ticks << Math.Min(strikes, 10))); + _skip[header.RunKey] = (entry.Done, entry.Total, now + backoff, strikes); continue; } - _emptyUntil.Remove(header.RunKey); + else + { + _skip.Remove(header.RunKey); + } + if (claims.Count == 0) continue; foreach (var c in claims) { @@ -153,17 +200,14 @@ internal async Task RefillAsync(CancellationToken ct) var descriptor = new JobDescriptor(header.Name, c.TaskId, header.Priority) { RunKey = c.RunKey, Seq = c.Seq, Attempt = c.Attempt }; var jobId = _jobs.Enqueue(descriptor, name, id: $"{c.RunKey}|{c.Seq}"); _inFlight[jobId] = (c, now + _lease); + if (held != null) held[c.RunKey] = held.GetValueOrDefault(c.RunKey) + 1; } need -= claims.Count; claimed += claims.Count; } - if (_emptyUntil.Count > 10_000) - foreach (var key in _emptyUntil.Where(kv => kv.Value.Until <= now).Select(kv => kv.Key).ToList()) - { - _emptyUntil.Remove(key); - _expiredCheckedAt.Remove(key); - } + // ponytail: forgetting every mark costs one read per run on the next pass; prune by Ready membership if that ever shows up. + if (_skip.Count > 50_000) _skip.Clear(); return claimed; } @@ -171,7 +215,7 @@ internal async Task RefillAsync(CancellationToken ct) /// Renew claims in their last third, so a long buffer wait or a long task never loses its lease. private async Task RenewAsync(CancellationToken ct) { - var now = DateTime.UtcNow; + var now = Clock(); var due = _inFlight.Where(kv => kv.Value.LeaseUntil - now < _lease / 3).ToList(); if (due.Count == 0) return; diff --git a/Services/Storage/WorkStore.cs b/Services/Storage/WorkStore.cs index 96c56e9..31f6fea 100644 --- a/Services/Storage/WorkStore.cs +++ b/Services/Storage/WorkStore.cs @@ -37,6 +37,7 @@ public sealed class WorkStore private readonly int MaxAttempts; private const int ConflictRetries = 16; + private const int MaxRunningScan = 500; private readonly ICraftTableStore _store; private readonly ILogger _logger; @@ -148,7 +149,8 @@ public Task PublishReadyAsync(RunHeader h, CancellationToken ct = default) => Properties = { ["RunKey"] = h.RunKey, ["Name"] = h.Name, ["Total"] = h.Total, ["Done"] = h.Done, ["Failed"] = h.Failed, - ["Cancelled"] = h.Cancelled, ["Reference"] = h.Reference, + ["Cancelled"] = h.Cancelled, ["Reference"] = h.Reference, ["Sequential"] = h.Sequential ? 1 : 0, + ["MaxConcurrency"] = h.MaxConcurrency, } }, ct); @@ -205,8 +207,10 @@ public async Task> GetActiveRunsAsync(string name, CancellationT catch (JsonException) { return []; } } + /// A run with work, as the scheduler sees it; its mode rides along so the pump can apply a + /// concurrency limit without reading the run. public sealed record ReadyEntry(int Band, string RunKey, string Name, int Total, int Done, DateTime StartedUtc, - string? Reference = null, int Failed = 0, int Cancelled = 0); + string? Reference = null, int Failed = 0, int Cancelled = 0, bool Sequential = false, int MaxConcurrency = 0); /// Runs with work, best band first and oldest first within it. public async IAsyncEnumerable ReadReadyAsync(int pageSize = 32, @@ -219,7 +223,8 @@ public async IAsyncEnumerable ReadReadyAsync(int pageSize = 32, var ticks = long.TryParse(row.RowKey.AsSpan(0, Math.Min(19, row.RowKey.Length)), NumberStyles.None, CultureInfo.InvariantCulture, out var t) ? t : 0; yield return new ReadyEntry(band, key, row.GetString("Name") ?? key, row.GetInt32("Total") ?? 0, row.GetInt32("Done") ?? 0, new DateTime(ticks, DateTimeKind.Utc), row.GetString("Reference"), - row.GetInt32("Failed") ?? 0, row.GetInt32("Cancelled") ?? 0); + row.GetInt32("Failed") ?? 0, row.GetInt32("Cancelled") ?? 0, row.GetInt32("Sequential") == 1, + row.GetInt32("MaxConcurrency") ?? 0); } } @@ -243,6 +248,18 @@ public async Task> GetTasksAsync(string runKey, char? state = null public sealed record ClaimedTask(string RunKey, int Seq, string TaskId, int Attempt); + /// What a claim saw of the run, for the scheduler: whether its pending tasks ran out, and when its + /// earliest live claim lapses (if it is not renewed by then, there is work to take back). + public sealed class ClaimProbe + { + public bool PendingExhausted { get; internal set; } + public DateTimeOffset? EarliestLeaseUntil { get; internal set; } + } + + /// Raised with the run key when a claim is handed back to pending, so a scheduler that had written + /// the run off as drained looks at it again. + public event Action? Released; + /// /// Move up to pending tasks to running under , plus, with /// , running tasks whose lease lapsed. A task claimed for the @@ -250,25 +267,36 @@ public sealed record ClaimedTask(string RunKey, int Seq, string TaskId, int Atte /// race returns empty and the caller moves on. /// public async Task> ClaimAsync(string runKey, int max, string owner, TimeSpan lease, - bool reclaimExpired, CancellationToken ct = default) + bool reclaimExpired, ClaimProbe? probe = null, CancellationToken ct = default) { max = Math.Min(max, MaxPerTransaction); if (max <= 0) return []; var now = DateTimeOffset.UtcNow; var expired = new List(); - if (reclaimExpired) + if (reclaimExpired || probe != null) { - await foreach (var r in Range(runKey, 'R', ct: ct)) + // Bounded so a run left with thousands of lapsed claims costs a fixed read per claim; an earliest + // lease from a partial scan only makes the scheduler look again sooner. + await foreach (var r in Range(runKey, 'R', MaxRunningScan, ct)) + { if (r.GetDateTimeOffset("LeaseUntil") is not { } until || until <= now) { - expired.Add(r); - if (expired.Count >= max) break; + if (reclaimExpired && expired.Count < max) expired.Add(r); + else if (probe != null) probe.EarliestLeaseUntil = now; + } + else if (probe != null && (probe.EarliestLeaseUntil is not { } seen || until < seen)) + { + probe.EarliestLeaseUntil = until; } + if (probe == null && expired.Count >= max) break; + } } + var pending = new List(); if (expired.Count < max) await foreach (var r in Range(runKey, 'P', max - expired.Count, ct: ct)) pending.Add(r); + if (probe != null) probe.PendingExhausted = pending.Count < max - expired.Count; if (pending.Count + expired.Count == 0) return []; var leaseUntil = now.Add(lease); @@ -335,6 +363,11 @@ public async Task> ClaimAsync(string runKey, int max, if (reclaiming && attempt > MaxAttempts) { await FinishAsync(runKey, [new Finish(seq, "Failed", $"Interrupted {attempt - 1} times without completing")], 'R', ct); + if (header.StopOnFailure) + { + await CancelPendingAsync(runKey, StoppedReason(row.GetString("TaskId")), ct); + return null; + } return await ClaimSequentialAsync(runKey, owner, lease, continuing, ct); } @@ -352,6 +385,9 @@ public async Task> ClaimAsync(string runKey, int max, : null; } + /// Why a stop-on-failure run cancelled its remaining steps. + public static string StoppedReason(string? failedTaskId) => $"Not run: step {failedTaskId} failed and the run stops on failure"; + /// Give up a sequential run's driver lease so the next step can be claimed by anyone. public async Task ReleaseDriverAsync(string runKey, string owner, CancellationToken ct = default) { @@ -383,8 +419,10 @@ public async Task ReleaseAsync(string runKey, int seq, string owner, bool if (row == null || row.GetString("Owner") != owner) return false; var attempt = (row.GetInt32("Attempt") ?? 1) - (refundAttempt ? 1 : 0); await _rate.TakeAsync(runKey, 2, ct); - return await _store.TrySubmitAsync(_work, runKey, + var released = await _store.TrySubmitAsync(_work, runKey, [StoreOp.Delete(row), StoreOp.Insert(PendingRow(runKey, seq, row.GetString("TaskId")!, attempt))], ct); + if (released) Released?.Invoke(runKey); + return released; } /// Push the lease out on claims this owner still holds. Returns the claims it no longer holds. @@ -582,7 +620,8 @@ public async Task AddChildAsync(string parentKey, string childKey, Cancell // ── cancel ── /// Cancel every pending task of a run. Running tasks finish; the barrier then fires as usual. - public async Task<(int Cancelled, FinishOutcome? Outcome)> CancelPendingAsync(string runKey, CancellationToken ct = default) + public async Task<(int Cancelled, FinishOutcome? Outcome)> CancelPendingAsync(string runKey, + string reason = "Cancelled by user", CancellationToken ct = default) { var cancelled = 0; FinishOutcome? outcome = null; @@ -590,7 +629,7 @@ public async Task AddChildAsync(string parentKey, string childKey, Cancell { var page = new List(); await foreach (var r in Range(runKey, 'P', MaxPerTransaction, ct: ct)) - if (SeqOf(r.RowKey) != AggregateSeq) page.Add(new Finish(SeqOf(r.RowKey), "Cancelled", "Cancelled by user")); + if (SeqOf(r.RowKey) != AggregateSeq) page.Add(new Finish(SeqOf(r.RowKey), "Cancelled", reason)); if (page.Count == 0) return (cancelled, outcome); var result = await FinishAsync(runKey, page, 'P', ct); if (result == null || result.Applied == 0) return (cancelled, outcome); @@ -707,6 +746,12 @@ public sealed class RunHeader public string? DriverOwner { get; set; } public DateTimeOffset? DriverLease { get; set; } public bool Sequential { get; init; } + + /// At most this many of the run's tasks run at once; 0 is no limit. Not used with . + public int MaxConcurrency { get; init; } + + /// Sequential runs: the first failed step cancels the steps after it instead of carrying on. + public bool StopOnFailure { get; init; } public bool CancelRequested { get; set; } public int Total { get; set; } public int Done { get; set; } @@ -736,6 +781,8 @@ public sealed class RunHeader ["DriverOwner"] = DriverOwner, ["DriverLease"] = DriverLease, ["Sequential"] = Sequential ? 1 : 0, + ["MaxConcurrency"] = MaxConcurrency, + ["StopOnFailure"] = StopOnFailure ? 1 : 0, ["CancelRequested"] = CancelRequested ? 1 : 0, ["Total"] = Total, ["Done"] = Done, @@ -762,6 +809,8 @@ public sealed class RunHeader DriverOwner = r.GetString("DriverOwner"), DriverLease = r.GetDateTimeOffset("DriverLease"), Sequential = r.GetInt32("Sequential") == 1, + MaxConcurrency = r.GetInt32("MaxConcurrency") ?? 0, + StopOnFailure = r.GetInt32("StopOnFailure") == 1, CancelRequested = r.GetInt32("CancelRequested") == 1, Total = r.GetInt32("Total") ?? 0, Done = r.GetInt32("Done") ?? 0, diff --git a/tests/Craft.Tests/CountingTableStore.cs b/tests/Craft.Tests/CountingTableStore.cs new file mode 100644 index 0000000..6a1d285 --- /dev/null +++ b/tests/Craft.Tests/CountingTableStore.cs @@ -0,0 +1,127 @@ +using System.Collections.Concurrent; +using System.Runtime.CompilerServices; +using Craft.Storage; + +namespace Craft.Tests; + +/// +/// Counts what reaches the table backend, per table: point reads, queries, the rows they returned and the +/// pages that would have taken, transactions and writes. Wraps another store and forwards everything. +/// Pages are rows actually consumed divided by the page size asked for (1,000 when none), so a query the +/// caller stops reading early costs only the pages it read, as with the Azure SDK's lazy paging. +/// +internal sealed class CountingTableStore(ICraftTableStore inner) : ICraftTableStore +{ + public sealed class Counts + { + public int PointReads, Queries, Rows, Pages, Submits, Upserts, BatchUpserts, Deletes; + public override string ToString() => + $"reads={PointReads} queries={Queries} rows={Rows} pages={Pages} submits={Submits} upserts={Upserts} batches={BatchUpserts} deletes={Deletes}"; + } + + private readonly ConcurrentDictionary _byTable = new(StringComparer.Ordinal); + + public Counts For(string table) => _byTable.GetOrAdd(table, _ => new Counts()); + public void Reset() => _byTable.Clear(); + + /// Every table's counts summed. + public Counts Total() + { + var t = new Counts(); + foreach (var c in _byTable.Values) + { + t.PointReads += c.PointReads; t.Queries += c.Queries; t.Rows += c.Rows; t.Pages += c.Pages; + t.Submits += c.Submits; t.Upserts += c.Upserts; t.BatchUpserts += c.BatchUpserts; t.Deletes += c.Deletes; + } + return t; + } + + private async IAsyncEnumerable Count(string table, IAsyncEnumerable rows, int pageSize, + [EnumeratorCancellation] CancellationToken ct = default) + { + var c = For(table); + Interlocked.Increment(ref c.Queries); + Interlocked.Increment(ref c.Pages); + var n = 0; + await foreach (var row in rows.WithCancellation(ct)) + { + if (n > 0 && n % pageSize == 0) Interlocked.Increment(ref c.Pages); + n++; + Interlocked.Increment(ref c.Rows); + yield return row; + } + } + + public Task PingAsync(CancellationToken ct = default) => inner.PingAsync(ct); + public Task EnsureTableAsync(string table, CancellationToken ct = default) => inner.EnsureTableAsync(table, ct); + + public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) + { + Interlocked.Increment(ref For(table).Upserts); + return inner.UpsertAsync(table, row, ct); + } + + public Task UpsertBatchAsync(string table, string partitionKey, IReadOnlyList rows, CancellationToken ct = default) + { + Interlocked.Increment(ref For(table).BatchUpserts); + return inner.UpsertBatchAsync(table, partitionKey, rows, ct); + } + + public Task TryReplaceBatchAsync(string table, string partitionKey, IReadOnlyList rows, CancellationToken ct = default) + { + Interlocked.Increment(ref For(table).Submits); + return inner.TryReplaceBatchAsync(table, partitionKey, rows, ct); + } + + public Task GetAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) + { + Interlocked.Increment(ref For(table).PointReads); + return inner.GetAsync(table, partitionKey, rowKey, ct); + } + + public IAsyncEnumerable QueryPartitionAsync(string table, string partitionKey, CancellationToken ct = default) => + Count(table, inner.QueryPartitionAsync(table, partitionKey, ct), 1000, ct); + + public IAsyncEnumerable QueryTableAsync(string table, CancellationToken ct = default) => + Count(table, inner.QueryTableAsync(table, ct), 1000, ct); + + public IAsyncEnumerable QueryTableAsync(string table, string? filter, CancellationToken ct = default) => + Count(table, inner.QueryTableAsync(table, filter, ct), 1000, ct); + + public IAsyncEnumerable QueryTableAsync(string table, string? filter, IReadOnlyList? properties, + CancellationToken ct = default) => + Count(table, inner.QueryTableAsync(table, filter, properties, ct), 1000, ct); + + public IAsyncEnumerable QueryTableAsync(string table, string? filter, int maxPerPage, CancellationToken ct = default) => + Count(table, inner.QueryTableAsync(table, filter, maxPerPage, ct), Math.Max(1, maxPerPage), ct); + + public IAsyncEnumerable QueryRowKeyRangeAsync(string table, string partitionKey, string fromRowKey, string toRowKey, + IReadOnlyList? properties = null, CancellationToken ct = default) => + Count(table, inner.QueryRowKeyRangeAsync(table, partitionKey, fromRowKey, toRowKey, properties, ct), 1000, ct); + + public Task TrySubmitAsync(string table, string partitionKey, IReadOnlyList ops, CancellationToken ct = default) + { + Interlocked.Increment(ref For(table).Submits); + return inner.TrySubmitAsync(table, partitionKey, ops, ct); + } + + public Task DeleteTableAsync(string table, CancellationToken ct = default) => inner.DeleteTableAsync(table, ct); + + public Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) + { + Interlocked.Increment(ref For(table).Deletes); + return inner.DeleteAsync(table, partitionKey, rowKey, ct); + } + + public Task DeleteBatchAsync(string table, string partitionKey, IReadOnlyList rowKeys, CancellationToken ct = default) + { + Interlocked.Increment(ref For(table).Deletes); + return inner.DeleteBatchAsync(table, partitionKey, rowKeys, ct); + } + + public Task DeletePartitionAsync(string table, string partitionKey, CancellationToken ct = default) + { + Interlocked.Increment(ref For(table).Deletes); + return inner.DeletePartitionAsync(table, partitionKey, ct); + } +} diff --git a/tests/Craft.Tests/MemoryTableStore.cs b/tests/Craft.Tests/MemoryTableStore.cs index 1a918e6..e3fe843 100644 --- a/tests/Craft.Tests/MemoryTableStore.cs +++ b/tests/Craft.Tests/MemoryTableStore.cs @@ -6,12 +6,13 @@ namespace Craft.Tests; /// In-memory with the backend properties the orchestration relies on: rows come /// back ordered by PartitionKey then RowKey, every write stamps a new ETag, and /// is all-or-nothing under one lock. Reads hand out copies, so mutating a row -/// read from here never writes it. +/// read from here never writes it. Partition and row-key range reads are index seeks, as on Azure, so a test +/// with a large table pays for the rows it reads rather than for the whole table. /// internal sealed class MemoryTableStore : ICraftTableStore { private readonly object _lock = new(); - private readonly Dictionary> _tables = new(); + private readonly Dictionary _tables = new(); private long _etag; /// Awaited before a transaction is checked — lets a test interleave a competing write. @@ -25,9 +26,31 @@ internal sealed class MemoryTableStore : ICraftTableStore return c != 0 ? c : string.CompareOrdinal(a.Item2, b.Item2); }); - private SortedDictionary<(string, string), StoreRow> Table(string t) + private sealed class Table { - if (!_tables.TryGetValue(t, out var rows)) _tables[t] = rows = new(Order); + public readonly Dictionary<(string, string), StoreRow> Rows = new(); + public readonly SortedSet<(string, string)> Keys = new(Order); + + public void Set((string, string) key, StoreRow row) + { + Rows[key] = row; + Keys.Add(key); + } + + public void Remove((string, string) key) + { + Rows.Remove(key); + Keys.Remove(key); + } + + /// Keys from (inclusive) to (exclusive), in order. + public IEnumerable<(string, string)> Between((string, string) from, (string, string) to) => + Order.Compare(from, to) >= 0 ? [] : Keys.GetViewBetween(from, to).Where(k => Order.Compare(k, to) < 0); + } + + private Table Of(string t) + { + if (!_tables.TryGetValue(t, out var rows)) _tables[t] = rows = new Table(); return rows; } @@ -47,26 +70,30 @@ internal sealed class MemoryTableStore : ICraftTableStore public IReadOnlyList All(string table) { - lock (_lock) return Table(table).Values.Select(Copy).ToList(); + lock (_lock) + { + var t = Of(table); + return t.Keys.Select(k => Copy(t.Rows[k])).ToList(); + } } public Task PingAsync(CancellationToken ct = default) => Task.CompletedTask; public Task EnsureTableAsync(string table, CancellationToken ct = default) { - lock (_lock) Table(table); + lock (_lock) Of(table); return Task.CompletedTask; } public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) { - lock (_lock) Table(table)[(row.PartitionKey, row.RowKey)] = Stamp(row); + lock (_lock) Of(table).Set((row.PartitionKey, row.RowKey), Stamp(row)); return Task.CompletedTask; } public Task UpsertBatchAsync(string table, string partitionKey, IReadOnlyList rows, CancellationToken ct = default) { - lock (_lock) foreach (var r in rows) Table(table)[(r.PartitionKey, r.RowKey)] = Stamp(r); + lock (_lock) foreach (var r in rows) Of(table).Set((r.PartitionKey, r.RowKey), Stamp(r)); return Task.CompletedTask; } @@ -83,10 +110,10 @@ private bool Submit(string table, IReadOnlyList ops) { lock (_lock) { - var t = Table(table); + var t = Of(table); foreach (var op in ops) { - t.TryGetValue((op.Row.PartitionKey, op.Row.RowKey), out var cur); + t.Rows.TryGetValue((op.Row.PartitionKey, op.Row.RowKey), out var cur); var ok = op.Kind switch { StoreOpKind.Insert => cur == null, @@ -100,7 +127,7 @@ private bool Submit(string table, IReadOnlyList ops) { var key = (op.Row.PartitionKey, op.Row.RowKey); if (op.Kind == StoreOpKind.Delete) t.Remove(key); - else t[key] = Stamp(op.Row); + else t.Set(key, Stamp(op.Row)); } Submits++; return true; @@ -110,29 +137,52 @@ private bool Submit(string table, IReadOnlyList ops) public Task GetAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) { lock (_lock) - return Task.FromResult(Table(table).TryGetValue((partitionKey, rowKey), out var r) ? Copy(r) : null); + return Task.FromResult(Of(table).Rows.TryGetValue((partitionKey, rowKey), out var r) ? Copy(r) : null); } - private List Snapshot(string table, Func where) + private List Snapshot(string table, Func> keys) { - lock (_lock) return Table(table).Values.Where(where).Select(Copy).ToList(); + lock (_lock) + { + var t = Of(table); + return keys(t).Select(k => Copy(t.Rows[k])).ToList(); + } } public async IAsyncEnumerable QueryPartitionAsync(string table, string partitionKey, [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) { - foreach (var r in Snapshot(table, r => r.PartitionKey == partitionKey)) { yield return r; await Task.Yield(); } + foreach (var r in Snapshot(table, t => t.Between((partitionKey, ""), (partitionKey + "\0", "")))) + { + yield return r; + await Task.Yield(); + } + } + + public async IAsyncEnumerable QueryRowKeyRangeAsync(string table, string partitionKey, string fromRowKey, + string toRowKey, IReadOnlyList? properties = null, + [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) + { + foreach (var r in Snapshot(table, t => t.Between((partitionKey, fromRowKey), (partitionKey, toRowKey)))) + { + yield return r; + await Task.Yield(); + } } public async IAsyncEnumerable QueryTableAsync(string table, [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) { - foreach (var r in Snapshot(table, _ => true)) { yield return r; await Task.Yield(); } + foreach (var r in Snapshot(table, t => t.Keys)) + { + yield return r; + await Task.Yield(); + } } public Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) { - lock (_lock) Table(table).Remove((partitionKey, rowKey)); + lock (_lock) Of(table).Remove((partitionKey, rowKey)); return Task.CompletedTask; } @@ -140,8 +190,8 @@ public Task DeletePartitionAsync(string table, string partitionKey, Cancellation { lock (_lock) { - var t = Table(table); - foreach (var k in t.Keys.Where(k => k.Item1 == partitionKey).ToList()) t.Remove(k); + var t = Of(table); + foreach (var k in t.Between((partitionKey, ""), (partitionKey + "\0", "")).ToList()) t.Remove(k); } return Task.CompletedTask; } diff --git a/tests/Craft.Tests/OrchestrationAzuriteTests.cs b/tests/Craft.Tests/OrchestrationAzuriteTests.cs new file mode 100644 index 0000000..01b324e --- /dev/null +++ b/tests/Craft.Tests/OrchestrationAzuriteTests.cs @@ -0,0 +1,146 @@ +using System.Text.Json; +using Craft.Configuration; +using Craft.Storage; + +namespace Craft.Tests; + +/// +/// The orchestration end to end on a real table backend: entity-group transactions, ETag guards, row-key range +/// queries and key escaping as Azure applies them, none of which the in-memory store can prove. Azurite by +/// default, a real account via CRAFT_TEST_TABLE_CONNECTION; each test is skipped, not failed, when neither is +/// reachable, and works in its own table prefix. +/// +[Collection(LargeAllocationSerialTests.Name)] +public class OrchestrationAzuriteTests +{ + private static async Task TryCreateAsync(int poolSize = 4) + { + var settings = new CraftSettings(); + var connection = Environment.GetEnvironmentVariable("CRAFT_TEST_TABLE_CONNECTION"); + if (!string.IsNullOrWhiteSpace(connection)) settings.Auth.UserStorageConnection = connection; + else settings.Storage.AllowDevelopmentStorage = true; + + var tables = new AzureTableStore(settings); + try + { + using var cts = new CancellationTokenSource(TimeSpan.FromSeconds(3)); + await tables.PingAsync(cts.Token); + } + catch + { + return null; + } + + var prefix = "azo" + Guid.NewGuid().ToString("N")[..10]; + return await OrchestrationHarness.CreateAsync(poolSize, tables, s => s.Orchestrator.TablePrefix = prefix); + } + + private static string Batch(int n, string prefix) => OrchestrationHarness.Batch(n, prefix); + + [Fact] + public async Task AFanOutWithALargeAggregationPayload_RunsEveryTaskOnce_AndAggregatesThemAll() + { + await using var h = await TryCreateAsync(); + if (h == null) return; + var big = JsonSerializer.Serialize(new { blob = new string('x', 150_000) }); + Assert.True(await h.Start("Az Fan-Out #1", Batch(60, "t"), "Agg", big)); + + // A space and a '#' in the name: the run is keyed by its sanitized form. + Assert.True(await h.DriveUntilFinished(TableKeys.Sanitize("Az Fan-Out #1"), 60_000)); + var post = Assert.Single(h.Svc.PostExecs); + Assert.Equal(60, post.Lines.Length); + Assert.Equal(big, post.Parameters["ParametersJson"]); + Assert.Equal(60, h.Svc.Started.Distinct().Count()); + Assert.Equal(60, h.Svc.Started.Count); + } + + [Fact] + public async Task AChildHoldsItsParentsAggregation_UntilItFinishes() + { + await using var h = await TryCreateAsync(); + if (h == null) return; + var childGate = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + h.Svc.BeforeRun = async t => + { + var id = FakeOrchestrator.IdOf(t); + if (id == "p0") + { + var parent = (await h.Store.GetActiveRunsAsync("AzParent"))[0]; + var link = h.Svc.RegisterPendingChild(parent.RunKey, "AzChild")!.Value; + await h.Svc.StartFromBatchAsync("AzChild", Batch(2, "c"), 4, null, null, CancellationToken.None, + parentRunKey: link.ParentRunKey, childKey: link.ChildKey); + } + if (id.StartsWith('c')) await childGate.Task; + }; + Assert.True(await h.Start("AzParent", Batch(1, "p"), "Agg")); + + Assert.True(await h.DriveUntil(async () => await h.Store.GetRunByNameAsync("AzChild") != null && h.Svc.Started.Count >= 3, 30_000)); + await h.DriveUntil(() => Task.FromResult(false), 500); + Assert.Empty(h.Svc.PostExecs); + + childGate.SetResult(); + Assert.True(await h.DriveUntilFinished("AzParent", 30_000)); + Assert.Single(h.Svc.PostExecs); + } + + [Fact] + public async Task AConcurrencyLimit_HoldsOnARealBackend() + { + await using var h = await TryCreateAsync(poolSize: 6); + if (h == null) return; + h.Svc.HoldMs = 150; + Assert.True(await h.Start("AzCapped", Batch(12, "c"), maxConcurrency: 2)); + + Assert.True(await h.DriveUntilFinished("AzCapped", 60_000)); + Assert.Equal(2, h.Svc.MaxActive); + Assert.Equal(12, h.Svc.Started.Distinct().Count()); + } + + [Fact] + public async Task ASequentialStopOnFailureRun_StopsAndRecordsTheRestCancelled() + { + await using var h = await TryCreateAsync(); + if (h == null) return; + h.Svc.Body = t => FakeOrchestrator.IdOf(t) == "s1" ? throw new InvalidOperationException("down") : "{}"; + Assert.True(await h.Start("AzStop", Batch(4, "s"), "Agg", sequential: true, stopOnFailure: true)); + + Assert.True(await h.DriveUntilFinished("AzStop", 30_000)); + Assert.Equal(["s0", "s1"], h.Svc.Started); + var run = (await h.Store.GetRunByNameAsync("AzStop"))!; + Assert.Equal((1, 2, "CompletedWithErrors"), (run.Failed, run.Cancelled, run.Status)); + Assert.Single(h.Svc.PostExecs); + } + + [Fact] + public async Task StackedRunsOfOneName_AreFoundByName_AndCancelledTogether() + { + await using var h = await TryCreateAsync(); + if (h == null) return; + // A dash and a tilde-free suffix, so the by-name range has neighbours that share its prefix. + Assert.True(await h.Start("AzTwin", Batch(5, "a"))); + Assert.True(await h.Start("AzTwin", Batch(5, "b"))); + Assert.True(await h.Start("AzTwin-Other", Batch(5, "c"))); + Assert.False(await h.Start("AzTwin", Batch(1, "d"), allowCollision: false)); + + Assert.Equal(2, (await h.Store.GetActiveRunsAsync("AzTwin")).Count); + var (found, cancelled) = await h.Svc.CancelRunAsync("AzTwin"); + Assert.True(found); + Assert.Equal(10, cancelled); + Assert.Empty(await h.Store.GetActiveRunsAsync("AzTwin")); + Assert.Single(await h.Store.GetActiveRunsAsync("AzTwin-Other")); + } + + [Fact] + public async Task AClaimLeftByADeadProcess_IsTakenBackAfterItsLease() + { + await using var h = await TryCreateAsync(); + if (h == null) return; + Assert.True(await h.Start("AzOrphan", Batch(3, "o"))); + var run = (await h.Store.GetRunByNameAsync("AzOrphan"))!; + Assert.Equal(3, (await h.Store.ClaimAsync(run.RunKey, 3, "dead-host", TimeSpan.FromMilliseconds(1), false)).Count); + await Task.Delay(50); + + Assert.True(await h.DriveUntilFinished("AzOrphan", 30_000)); + Assert.All(await h.Store.GetTasksAsync(run.RunKey, 'D'), t => Assert.Equal(2, t.Attempt)); + } +} diff --git a/tests/Craft.Tests/OrchestrationContractTests.cs b/tests/Craft.Tests/OrchestrationContractTests.cs index 2dd2c4a..2c7f4af 100644 --- a/tests/Craft.Tests/OrchestrationContractTests.cs +++ b/tests/Craft.Tests/OrchestrationContractTests.cs @@ -1,11 +1,6 @@ -using System.Collections.Concurrent; using System.Text.Json; using Craft.Configuration; -using Craft.Orchestration; -using Craft.PowerShellHost; -using Craft.Storage; using Microsoft.Extensions.Configuration; -using Microsoft.Extensions.Logging.Abstractions; namespace Craft.Tests; @@ -18,85 +13,13 @@ namespace Craft.Tests; /// public class OrchestrationContractTests { - private const string TaskFunc = "Invoke-CraftTask"; - private const string PostExecFunc = "Invoke-CraftPostExecution"; + private static Task NewAsync() => OrchestrationHarness.CreateAsync(); - private sealed class Svc(JobManager jobs, WorkStore store, ResultStore results, IConfiguration config, CraftSettings settings) - : OrchestratorService(NullLogger.Instance, null!, null!, jobs, store, results, config, settings) - { - public readonly ConcurrentQueue> Tasks = new(); - public readonly ConcurrentQueue<(Dictionary Parameters, string[] Lines)> PostExecs = new(); - public Func, string>? Body; - public Func? PostExecBody; - public int Checkouts, Reclaims; - - internal override string? FindScript(string name) => name; + private static string Batch(int n, string prefix = "t") => OrchestrationHarness.Batch(n, prefix); - internal override async Task RunScriptAsync(string path, Dictionary parameters, bool captureOutput, - PowerShellWorker? worker = null) - { - if (path == PostExecFunc) - { - PostExecs.Enqueue((parameters, File.ReadAllLines((string)parameters["ResultsPath"]))); - if (PostExecBody != null) await PostExecBody(); - return string.Empty; - } - var task = JsonSerializer.Deserialize>((string)parameters["TaskJson"])!; - Tasks.Enqueue(task); - var output = Body?.Invoke(task) ?? JsonSerializer.Serialize(new { tenant = task["TenantFilter"].ToString() }); - return captureOutput ? output : string.Empty; - } - - internal override PowerShellWorker? CheckoutSequentialWorker(CancellationToken ct) { Checkouts++; return null; } - internal override void ReclaimSequentialWorker(PowerShellWorker? worker, bool faulted) => Reclaims++; - } - - private sealed record Harness(Svc Svc, WorkStore Store, WorkPump Pump, JobManager Jobs) : IAsyncDisposable - { - public async Task DriveUntil(Func> done, int timeoutMs = 10_000) - { - var deadline = Environment.TickCount64 + timeoutMs; - while (Environment.TickCount64 < deadline) - { - await Pump.RefillAsync(CancellationToken.None); - if (await done()) return true; - await Task.Delay(10); - } - return await done(); - } - - public Task DriveUntilFinished(string name, int timeoutMs = 10_000) => - DriveUntil(async () => await Store.GetRunByNameAsync(name) is { IsFinished: true }, timeoutMs); - - public async ValueTask DisposeAsync() => await Jobs.StopAsync(CancellationToken.None); - } - - private static async Task NewAsync() - { - var settings = new CraftSettings(); - settings.Worker.BgPoolSize = 4; - var config = new ConfigurationBuilder().AddInMemoryCollection([]).Build(); - var repo = new ScriptRepository(NullLogger.Instance, settings); - var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); - var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); - var jobs = new JobManager(NullLogger.Instance, settings, limiter); - var mem = new MemoryTableStore(); - var store = new WorkStore(NullLogger.Instance, settings, mem); - var results = new ResultStore(NullLogger.Instance, settings, mem); - var svc = new Svc(jobs, store, results, config, settings); - await svc.ResumeInterruptedRunsAsync(CancellationToken.None); - var pump = new WorkPump(NullLogger.Instance, store, jobs, config, settings, svc); - _ = Task.Run(() => jobs.StartAsync(CancellationToken.None)); - return new Harness(svc, store, pump, jobs); - } - - private static string Batch(int n, string prefix = "t") => - JsonSerializer.Serialize(Enumerable.Range(0, n).Select(i => new { Name = "Job", TenantFilter = $"{prefix}{i}", N = i })); - - private static Task Start(Harness h, string name, string batch, string? postExec = null, string? postParams = null, - bool sequential = false, int priority = 4, bool allowCollision = true) => - h.Svc.StartFromBatchAsync(name, batch, priority, postExec, postParams, CancellationToken.None, sequential: sequential, - allowCollision: allowCollision); + private static Task Start(OrchestrationHarness h, string name, string batch, string? postExec = null, + string? postParams = null, bool sequential = false, int priority = 4, bool allowCollision = true) => + h.Start(name, batch, postExec, postParams, sequential, priority, allowCollision); [Fact] public async Task EveryTaskRunsOnce_WithItsBatchItemAsTaskJson_AndTheRunCompletes() @@ -352,10 +275,5 @@ public async Task ALowerBandRunsFirst_AndWithinABandTheOlderRun_NotTheAlphabetic Assert.Equal(["Urgent", "Zulu", "Alpha", "Background"], await ReadyNames(h)); } - private static async Task> ReadyNames(Harness h) - { - var names = new List(); - await foreach (var e in h.Store.ReadReadyAsync()) names.Add(e.Name); - return names; - } + private static Task> ReadyNames(OrchestrationHarness h) => h.ReadyNamesAsync(); } diff --git a/tests/Craft.Tests/OrchestrationCostTests.cs b/tests/Craft.Tests/OrchestrationCostTests.cs new file mode 100644 index 0000000..51be1ef --- /dev/null +++ b/tests/Craft.Tests/OrchestrationCostTests.cs @@ -0,0 +1,276 @@ +using Craft.Configuration; +using Craft.Orchestration; +using Craft.PowerShellHost; +using Craft.Storage; +using Microsoft.Extensions.Configuration; +using Microsoft.Extensions.Logging.Abstractions; +using Xunit.Abstractions; + +namespace Craft.Tests; + +/// +/// Storage cost pins. Each operation's table traffic is counted and held to a bound that does not grow with +/// the size of the queue, so a change that turns an O(batch) step into an O(queue) one fails here, long before +/// it shows up as a backlog on a large instance. Where a bound is a deliberate trade-off it says why. +/// +public class OrchestrationCostTests(ITestOutputHelper output) +{ + private const string Work = "OrchestratorWork", Ready = "OrchestratorReady", Names = "OrchestratorNames", + Finished = "OrchestratorFinished"; + + private static readonly TimeSpan Lease = TimeSpan.FromMinutes(30); + + private static (WorkStore Store, CountingTableStore Count) NewStore() + { + var count = new CountingTableStore(new MemoryTableStore()); + return (new WorkStore(NullLogger.Instance, new CraftSettings(), count), count); + } + + private static DateTime s_clock = new(2026, 10, 6, 0, 0, 0, DateTimeKind.Utc); + private static DateTime NextStart() => s_clock = s_clock.AddTicks(1); + + private static Task CreateAsync(WorkStore s, string name, int tasks, int maxConcurrency = 0, + string? postExec = null) + { + var started = NextStart(); + return s.CreateRunAsync(new RunHeader + { + RunKey = WorkStore.RunKeyFor(name, started), + Name = name, + StartedUtc = started, + TaskScriptName = "Invoke-CraftTask", + MaxConcurrency = maxConcurrency, + PostExecFunctionName = postExec, + }, Enumerable.Range(0, tasks).Select(i => new WorkStore.NewTask($"t{i}", new() { ["i"] = i })).ToList()); + } + + private static JobManager NewJobs(int poolSize = 4) + { + var settings = new CraftSettings(); + settings.Worker.BgPoolSize = poolSize; + var config = new ConfigurationBuilder().AddInMemoryCollection([]).Build(); + var repo = new ScriptRepository(NullLogger.Instance, settings); + var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); + var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); + return new JobManager(NullLogger.Instance, settings, limiter); + } + + private static WorkPump NewPump(WorkStore store, JobManager jobs, int batchSize = 4, int lowWater = 2) + { + var settings = new CraftSettings(); + settings.Worker.BgPoolSize = 4; + var config = new ConfigurationBuilder().AddInMemoryCollection(new Dictionary + { + ["JobQueueBatchSize"] = batchSize.ToString(System.Globalization.CultureInfo.InvariantCulture), + ["JobQueueLowWaterMark"] = lowWater.ToString(System.Globalization.CultureInfo.InvariantCulture), + }).Build(); + return new WorkPump(NullLogger.Instance, store, jobs, config, settings); + } + + [Theory] + [InlineData(10)] + [InlineData(5_000)] + public async Task CreatingARun_CostsAFixedNumberOfCalls_WhateverItsSize(int tasks) + { + var (s, c) = NewStore(); + await s.InitializeAsync(); + c.Reset(); + + await CreateAsync(s, "Sized", tasks); + + output.WriteLine($"work: {c.For(Work)} names: {c.For(Names)} ready: {c.For(Ready)}"); + Assert.Equal(2, c.For(Work).BatchUpserts); // payload rows, then pending rows (chunked by the backend) + Assert.Equal(1, c.For(Work).Upserts); // the header, last + Assert.Equal(1, c.For(Work).PointReads); // the created run read back + Assert.Equal(2, c.For(Names).Upserts); // latest-by-name and the active-outing row + Assert.Equal(1, c.For(Ready).Upserts); + } + + [Fact] + public async Task Claiming_ReadsOnlyTheRowsItClaims_FromAnyRunSize() + { + var (s, c) = NewStore(); + var run = await CreateAsync(s, "Big", 10_000); + c.Reset(); + + var claims = await s.ClaimAsync(run.RunKey, 49, "w", Lease, reclaimExpired: false); + + Assert.Equal(49, claims.Count); + var w = c.For(Work); + output.WriteLine($"plain claim: {w}"); + Assert.Equal((1, 49, 1, 0), (w.Queries, w.Rows, w.Submits, w.PointReads)); + + c.Reset(); + await s.ClaimAsync(run.RunKey, 49, "w", Lease, reclaimExpired: true); + w = c.For(Work); + output.WriteLine($"claim with lapsed-lease check: {w}"); + Assert.Equal(2, w.Queries); // running range, then pending range + Assert.Equal(49 + 49, w.Rows); // the 49 live claims are read to find lapsed ones + Assert.Equal(1, w.Submits); + } + + [Fact] + public async Task ARunAtItsConcurrencyLimit_CostsNoStorageReads_AndThePumpNeverHoldsMoreThanTheLimit() + { + var (s, c) = NewStore(); + await CreateAsync(s, "Capped", 10_000, maxConcurrency: 3); + var pump = NewPump(s, NewJobs(), batchSize: 49, lowWater: 1_000); + + Assert.Equal(3, await pump.RefillAsync(CancellationToken.None)); + c.Reset(); + for (var i = 0; i < 5; i++) Assert.Equal(0, await pump.RefillAsync(CancellationToken.None)); + + output.WriteLine($"5 refills at the limit: work {c.For(Work)} ready {c.For(Ready)}"); + Assert.Equal((0, 0, 0), (c.For(Work).PointReads, c.For(Work).Queries, c.For(Work).Submits)); + Assert.Equal(5, c.For(Ready).Queries); + } + + [Fact] + public async Task FinishingABatch_IsOneTransaction_AndOneReadyUpdate() + { + var (s, c) = NewStore(); + var run = await CreateAsync(s, "Fin", 10_000); + var claims = await s.ClaimAsync(run.RunKey, 49, "w", Lease, false); + c.Reset(); + + await s.FinishAsync(run.RunKey, claims.Select(x => new WorkStore.Finish(x.Seq, "Completed", Owner: "w")).ToList()); + + var w = c.For(Work); + output.WriteLine($"finish 49: {w} ready: {c.For(Ready)}"); + Assert.Equal(1, w.Submits); + Assert.Equal(49 + 2, w.PointReads); // each claimed row, the header before and after + Assert.Equal(0, w.Queries); + Assert.Equal(1, c.For(Ready).Upserts); + } + + [Fact] + public async Task ConcurrentFinishesOfOneRun_AreCoalescedIntoFewTransactions() + { + var (s, c) = NewStore(); + var run = await CreateAsync(s, "Coalesce", 100); + var claims = await s.ClaimAsync(run.RunKey, 40, "w", Lease, false); + var batcher = new FinishBatcher(s, NullLogger.Instance, TimeSpan.FromMilliseconds(30)); + c.Reset(); + + await Task.WhenAll(claims.Select(x => Task.Run(() => batcher.FinishAsync(run.RunKey, new WorkStore.Finish(x.Seq, "Completed", Owner: "w"))))); + + output.WriteLine($"40 concurrent finishes: {c.For(Work)}"); + Assert.InRange(c.For(Work).Submits, 1, 3); + Assert.Equal(40, (await s.GetRunAsync(run.RunKey))!.Done); + } + + [Fact] + public async Task TheStatusSnapshot_ReadsTheReadyListOnce_AndOnlyTheHeadOfTheQueue() + { + var (s, c) = NewStore(); + for (var i = 0; i < 5_000; i++) await CreateAsync(s, $"Run{i}", 2); + var reader = new JobQueueStatusReader(NullLogger.Instance, NewJobs(), s); + c.Reset(); + + var snap = (await reader.GetAsync())!; + + Assert.Equal(10_000, snap.Total); + output.WriteLine($"snapshot over 5,000 runs: ready {c.For(Ready)} work {c.For(Work)}"); + Assert.Equal((1, 5_000, 5), (c.For(Ready).Queries, c.For(Ready).Rows, c.For(Ready).Pages)); + Assert.InRange(c.For(Work).Queries, 0, 50); // HeadRuns + Assert.InRange(c.For(Work).Rows, 0, 100); + Assert.Equal(0, c.For(Work).PointReads); + } + + [Theory] + [InlineData(10)] + [InlineData(3_000)] + public async Task ARefill_CostsTheSame_WhateverTheBacklogBehindTheHead(int runs) + { + var (s, c) = NewStore(); + for (var i = 0; i < runs; i++) await CreateAsync(s, $"Run{i}", 5); + var pump = NewPump(s, NewJobs()); + c.Reset(); + + var claimed = await pump.RefillAsync(CancellationToken.None); + + Assert.Equal(4, claimed); + output.WriteLine($"refill with {runs} runs queued: ready {c.For(Ready)} work {c.For(Work)}"); + Assert.Equal((1, 1), (c.For(Ready).Queries, c.For(Ready).Pages)); + Assert.Equal(1, c.For(Work).PointReads); // the head run's header + Assert.Equal(2, c.For(Work).Queries); // its running range (first visit) and pending range + Assert.Equal(1, c.For(Work).Submits); + } + + [Fact] + public async Task LookingUpActiveRunsByName_IsOneRangeQuery_PlusOneReadPerOuting() + { + var (s, c) = NewStore(); + for (var i = 0; i < 1_000; i++) await CreateAsync(s, $"Other{i}", 1); + for (var i = 0; i < 3; i++) await CreateAsync(s, "Wanted", 1); + c.Reset(); + + Assert.Equal(3, (await s.GetActiveRunsAsync("Wanted")).Count); + Assert.Equal((1, 3), (c.For(Names).Queries, c.For(Names).Rows)); + Assert.Equal(3, c.For(Work).PointReads); + } + + [Fact] + public async Task TheRetentionSweep_ReadsOnlyRunsPastTheCutoff() + { + var (s, c) = NewStore(); + var old = new List(); + for (var i = 0; i < 20; i++) old.Add(await CreateAsync(s, $"Done{i}", 1)); + foreach (var run in old) + { + var claim = await s.ClaimAsync(run.RunKey, 1, "w", Lease, false); + await s.FinishAsync(run.RunKey, [new WorkStore.Finish(claim[0].Seq, "Completed", Owner: "w")]); + } + c.Reset(); + + Assert.Equal(0, await s.SweepFinishedAsync(TimeSpan.FromHours(1))); + Assert.Equal((1, 0), (c.For(Finished).Queries, c.For(Finished).Rows)); + + Assert.Equal(20, await s.SweepFinishedAsync(TimeSpan.Zero)); + } + + /// + /// The case that wedges a big instance: thousands of runs that have nothing to claim (waiting on children, + /// or on claims held elsewhere) sit ahead of one run that does. The pump gives each run it looks at a + /// storage read from a small per-refill budget, so if the blocked runs keep spending that budget the run + /// at the back is starved. Simulates ten minutes at one refill a second and requires the run at the back to + /// be reached quickly and then claimed from on nearly every refill. + /// + [Fact] + public async Task TheRunAtTheBackOfAQueueOfBlockedRuns_IsReached_AndKeepsBeingClaimed() + { + var (s, c) = NewStore(); + for (var i = 0; i < 2_000; i++) + { + var blocked = await CreateAsync(s, $"Blocked{i}", 1); + await s.ClaimAsync(blocked.RunKey, 1, "elsewhere", TimeSpan.FromDays(1), false); + } + await CreateAsync(s, "Tail", 100_000); + + var jobs = NewJobs(); + jobs.SetWorkResolver((_, _) => Task.FromResult?>(null)); + _ = Task.Run(() => jobs.StartAsync(CancellationToken.None)); + var pump = NewPump(s, jobs); + var now = new DateTime(2026, 10, 6, 0, 0, 0, DateTimeKind.Utc); + pump.Clock = () => now; + + var first = -1; + var claimedRefills = new List(); + c.Reset(); + for (var second = 0; second < 600; second++) + { + now = now.AddSeconds(1); + var claimed = await pump.RefillAsync(CancellationToken.None); + if (claimed > 0 && first < 0) first = second; + claimedRefills.Add(claimed > 0); + for (var spin = 0; spin < 200 && jobs.QueuedCount > 0; spin++) await Task.Delay(1); + } + await jobs.StopAsync(CancellationToken.None); + + var lateShare = claimedRefills.Skip(300).Count(x => x) / 300.0; + output.WriteLine($"first claim at refill {first}; claimed in {lateShare:P0} of the last 300 refills; work {c.For(Work)} ready {c.For(Ready)}"); + Assert.InRange(first, 0, 70); + Assert.InRange(c.For(Ready).Pages, 0, 600 * 3); // the whole Ready list each refill, 1,000 rows a page + Assert.True(lateShare >= 0.9, $"the run at the back was claimed in only {lateShare:P0} of the last 300 refills"); + } +} diff --git a/tests/Craft.Tests/OrchestrationHarness.cs b/tests/Craft.Tests/OrchestrationHarness.cs new file mode 100644 index 0000000..c71d0de --- /dev/null +++ b/tests/Craft.Tests/OrchestrationHarness.cs @@ -0,0 +1,145 @@ +using System.Collections.Concurrent; +using System.Text.Json; +using Craft.Configuration; +using Craft.Orchestration; +using Craft.PowerShellHost; +using Craft.Storage; +using Microsoft.Extensions.Configuration; +using Microsoft.Extensions.Logging.Abstractions; + +namespace Craft.Tests; + +/// +/// The real orchestrator with only its PowerShell calls faked. Tasks are batch items; each records itself, can +/// hold or wait on a gate, and returns {"tenant": TenantFilter} unless says otherwise. +/// Tracks the start order and the peak number of tasks running at once. +/// +internal sealed class FakeOrchestrator(JobManager jobs, WorkStore store, ResultStore results, IConfiguration config, + CraftSettings settings) + : OrchestratorService(NullLogger.Instance, null!, null!, jobs, store, results, config, settings) +{ + public const string PostExecScript = "Invoke-CraftPostExecution"; + + public readonly ConcurrentQueue> Tasks = new(); + public readonly ConcurrentQueue Started = new(); + public readonly ConcurrentQueue<(Dictionary Parameters, string[] Lines)> PostExecs = new(); + + /// The task's output; throw to fail it. + public Func, string>? Body; + + /// Awaited before the task does anything else (a gate, or a hold for one task). + public Func, Task>? BeforeRun; + + public Func? PostExecBody; + public int HoldMs; + public int Checkouts, Reclaims; + + private int _active, _maxActive; + public int MaxActive => Volatile.Read(ref _maxActive); + + public static string IdOf(Dictionary task) => task["TenantFilter"].ToString()!; + + internal override string? FindScript(string name) => name; + + internal override async Task RunScriptAsync(string path, Dictionary parameters, bool captureOutput, + PowerShellWorker? worker = null) + { + if (path == PostExecScript) + { + PostExecs.Enqueue((parameters, File.ReadAllLines((string)parameters["ResultsPath"]))); + if (PostExecBody != null) await PostExecBody(); + return string.Empty; + } + + var task = JsonSerializer.Deserialize>((string)parameters["TaskJson"])!; + Tasks.Enqueue(task); + Started.Enqueue(IdOf(task)); + var now = Interlocked.Increment(ref _active); + for (var seen = _maxActive; now > seen; seen = _maxActive) + if (Interlocked.CompareExchange(ref _maxActive, now, seen) == seen) break; + try + { + if (BeforeRun != null) await BeforeRun(task); + if (HoldMs > 0) await Task.Delay(HoldMs); + var output = Body?.Invoke(task) ?? JsonSerializer.Serialize(new { tenant = IdOf(task) }); + return captureOutput ? output : string.Empty; + } + finally + { + Interlocked.Decrement(ref _active); + } + } + + internal override PowerShellWorker? CheckoutSequentialWorker(CancellationToken ct) { Interlocked.Increment(ref Checkouts); return null; } + internal override void ReclaimSequentialWorker(PowerShellWorker? worker, bool faulted) => Interlocked.Increment(ref Reclaims); +} + +/// Store, pump and JobManager wired as in the host, over an in-memory table store, driven by hand. +internal sealed class OrchestrationHarness : IAsyncDisposable +{ + public required FakeOrchestrator Svc { get; init; } + public required WorkStore Store { get; init; } + public required WorkPump Pump { get; init; } + public required JobManager Jobs { get; init; } + public required ICraftTableStore Tables { get; init; } + + public static async Task CreateAsync(int poolSize = 4, ICraftTableStore? tables = null, + Action? configure = null) + { + var settings = new CraftSettings(); + settings.Worker.BgPoolSize = poolSize; + configure?.Invoke(settings); + // The limiter's starting concurrency otherwise follows the CPU count, which differs between machines. + var config = new ConfigurationBuilder().AddInMemoryCollection(new Dictionary + { + ["BackgroundBaseConcurrency"] = poolSize.ToString(System.Globalization.CultureInfo.InvariantCulture), + }).Build(); + var repo = new ScriptRepository(NullLogger.Instance, settings); + var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); + var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); + var jobs = new JobManager(NullLogger.Instance, settings, limiter); + tables ??= new MemoryTableStore(); + var store = new WorkStore(NullLogger.Instance, settings, tables); + var results = new ResultStore(NullLogger.Instance, settings, tables); + var svc = new FakeOrchestrator(jobs, store, results, config, settings); + await svc.ResumeInterruptedRunsAsync(CancellationToken.None); + var pump = new WorkPump(NullLogger.Instance, store, jobs, config, settings, svc); + _ = Task.Run(() => jobs.StartAsync(CancellationToken.None)); + return new OrchestrationHarness { Svc = svc, Store = store, Pump = pump, Jobs = jobs, Tables = tables }; + } + + public static string Batch(int n, string prefix = "t") => + JsonSerializer.Serialize(Enumerable.Range(0, n).Select(i => new { Name = "Job", TenantFilter = $"{prefix}{i}", N = i })); + + public Task Start(string name, string batch, string? postExec = null, string? postParams = null, + bool sequential = false, int priority = 4, bool allowCollision = true, int maxConcurrency = 0, bool stopOnFailure = false) => + Svc.StartFromBatchAsync(name, batch, priority, postExec, postParams, CancellationToken.None, sequential: sequential, + allowCollision: allowCollision, maxConcurrency: maxConcurrency, stopOnFailure: stopOnFailure); + + public async Task DriveUntil(Func> done, int timeoutMs = 10_000) + { + var deadline = Environment.TickCount64 + timeoutMs; + while (Environment.TickCount64 < deadline) + { + await Pump.RefillAsync(CancellationToken.None); + if (await done()) return true; + await Task.Delay(10); + } + return await done(); + } + + public Task DriveUntilFinished(string name, int timeoutMs = 10_000) => + DriveUntil(async () => await Store.GetRunByNameAsync(name) is { IsFinished: true }, timeoutMs); + + public Task DriveUntilAllFinished(int timeoutMs = 10_000) => + DriveUntil(async () => (await ReadyNamesAsync()).Count == 0, timeoutMs); + + public async Task> ReadyNamesAsync() + { + var names = new List(); + await foreach (var e in Store.ReadReadyAsync()) names.Add(e.Name); + return names; + } + + public async ValueTask DisposeAsync() => await Jobs.StopAsync(CancellationToken.None); +} diff --git a/tests/Craft.Tests/OrchestrationModeTests.cs b/tests/Craft.Tests/OrchestrationModeTests.cs new file mode 100644 index 0000000..f771f3d --- /dev/null +++ b/tests/Craft.Tests/OrchestrationModeTests.cs @@ -0,0 +1,221 @@ +using Craft.Storage; + +namespace Craft.Tests; + +/// +/// How a run's tasks are scheduled, end to end: fan-out (all at once), a concurrency limit (at most N at once, +/// workers released between tasks so other work interleaves), and sequential (one at a time on one pinned +/// worker, optionally stopping at the first failure). +/// +public class OrchestrationModeTests +{ + private static string Batch(int n, string prefix) => OrchestrationHarness.Batch(n, prefix); + + // ── concurrency limit ── + + [Fact] + public async Task AConcurrencyLimit_IsNeverExceeded_AndIsUsedInFull() + { + await using var h = await OrchestrationHarness.CreateAsync(poolSize: 6); + h.Svc.HoldMs = 40; + Assert.True(await h.Start("Capped", Batch(24, "c"), maxConcurrency: 2)); + + Assert.True(await h.DriveUntilFinished("Capped", 20_000)); + Assert.Equal(2, h.Svc.MaxActive); + Assert.Equal(24, h.Svc.Started.Distinct().Count()); + Assert.Equal(24, h.Svc.Started.Count); + } + + [Fact] + public async Task NoLimit_UsesTheWholePool() + { + await using var h = await OrchestrationHarness.CreateAsync(poolSize: 4); + h.Svc.HoldMs = 60; + Assert.True(await h.Start("Open", Batch(16, "o"))); + + Assert.True(await h.DriveUntilFinished("Open", 20_000)); + Assert.Equal(4, h.Svc.MaxActive); + } + + [Fact] + public async Task ALimitAboveThePoolSize_IsBoundedByThePool() + { + await using var h = await OrchestrationHarness.CreateAsync(poolSize: 3); + h.Svc.HoldMs = 40; + Assert.True(await h.Start("Roomy", Batch(12, "r"), maxConcurrency: 10)); + + Assert.True(await h.DriveUntilFinished("Roomy", 20_000)); + Assert.Equal(3, h.Svc.MaxActive); + } + + [Fact] + public async Task ALimitOfOne_RunsTheTasksInPayloadOrder() + { + await using var h = await OrchestrationHarness.CreateAsync(poolSize: 4); + h.Svc.HoldMs = 10; + Assert.True(await h.Start("OneByOne", Batch(6, "s"), maxConcurrency: 1)); + + Assert.True(await h.DriveUntilFinished("OneByOne")); + Assert.Equal(["s0", "s1", "s2", "s3", "s4", "s5"], h.Svc.Started); + Assert.Equal(1, h.Svc.MaxActive); + } + + [Fact] + public async Task ALimitOfOne_ReleasesTheWorkerBetweenTasks_SoUrgentWorkRunsInBetween() + { + Assert.Equal(["s0", "u0", "s1", "s2"], await InterleaveAsync(sequential: false)); + } + + [Fact] + public async Task ASequentialRun_HoldsItsWorker_SoUrgentWorkWaitsForTheWholeRun() + { + Assert.Equal(["s0", "s1", "s2", "u0"], await InterleaveAsync(sequential: true)); + } + + /// One worker. A three-task run (limit 1, or sequential) is held on its first task while an urgent + /// run is queued; the order everything then runs in shows whether the worker went back to the pool. + private static async Task> InterleaveAsync(bool sequential) + { + await using var h = await OrchestrationHarness.CreateAsync(poolSize: 1); + var gate = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + h.Svc.BeforeRun = t => FakeOrchestrator.IdOf(t) == "s0" ? gate.Task : Task.CompletedTask; + Assert.True(await h.Start("Slow", Batch(3, "s"), priority: 5, sequential: sequential, + maxConcurrency: sequential ? 0 : 1)); + + Assert.True(await h.DriveUntil(() => Task.FromResult(h.Svc.Started.Contains("s0")))); + Assert.True(await h.Start("Urgent", Batch(1, "u"), priority: 1)); + await h.DriveUntil(() => Task.FromResult(false), 200); + gate.SetResult(); + + Assert.True(await h.DriveUntilAllFinished()); + return [.. h.Svc.Started]; + } + + [Fact] + public async Task ALimitedRun_StillRunsItsPostExecutionOnce_AfterEveryTask() + { + await using var h = await OrchestrationHarness.CreateAsync(); + Assert.True(await h.Start("CappedAgg", Batch(5, "a"), "Agg", maxConcurrency: 1)); + + Assert.True(await h.DriveUntilFinished("CappedAgg")); + var post = Assert.Single(h.Svc.PostExecs); + Assert.Equal(5, post.Lines.Length); + } + + [Fact] + public async Task ALimitedRunWithFailures_StaysWithinItsLimit_AndFinishesWithErrors() + { + await using var h = await OrchestrationHarness.CreateAsync(poolSize: 6); + h.Svc.HoldMs = 20; + h.Svc.Body = t => int.Parse(FakeOrchestrator.IdOf(t)[1..], System.Globalization.CultureInfo.InvariantCulture) % 3 == 0 ? throw new InvalidOperationException("boom") : "{}"; + Assert.True(await h.Start("CappedFailing", Batch(12, "f"), maxConcurrency: 3)); + + Assert.True(await h.DriveUntilFinished("CappedFailing", 20_000)); + var run = (await h.Store.GetRunByNameAsync("CappedFailing"))!; + Assert.Equal(3, h.Svc.MaxActive); + Assert.Equal("CompletedWithErrors", run.Status); + Assert.Equal(4, run.Failed); + } + + [Fact] + public async Task TheLimitIsStored_SoAnotherProcessPickingUpTheRunHonoursIt() + { + var tables = new MemoryTableStore(); + await using (var first = await OrchestrationHarness.CreateAsync(tables: tables)) + Assert.True(await first.Start("Handover", Batch(10, "h"), maxConcurrency: 2)); + + await using var second = await OrchestrationHarness.CreateAsync(poolSize: 6, tables: tables); + second.Svc.HoldMs = 40; + Assert.True(await second.DriveUntilFinished("Handover", 20_000)); + Assert.Equal(2, second.Svc.MaxActive); + Assert.Equal(2, (await second.Store.GetRunByNameAsync("Handover"))!.MaxConcurrency); + } + + [Fact] + public async Task ALimitOnASequentialRun_IsDropped_AndANegativeLimitMeansNone() + { + await using var h = await OrchestrationHarness.CreateAsync(); + Assert.True(await h.Start("SeqCapped", Batch(2, "q"), sequential: true, maxConcurrency: 3)); + Assert.True(await h.Start("Negative", Batch(2, "n"), maxConcurrency: -5)); + + var seq = (await h.Store.GetRunByNameAsync("SeqCapped"))!; + Assert.True(seq.Sequential); + Assert.Equal(0, seq.MaxConcurrency); + Assert.Equal(0, (await h.Store.GetRunByNameAsync("Negative"))!.MaxConcurrency); + } + + // ── sequential: carry on (default) or stop on failure ── + + [Fact] + public async Task ASequentialRun_CarriesOnPastAFailedStep_ByDefault() + { + await using var h = await OrchestrationHarness.CreateAsync(); + h.Svc.Body = t => FakeOrchestrator.IdOf(t) == "s1" ? throw new InvalidOperationException("step down") : "{}"; + Assert.True(await h.Start("CarryOn", Batch(4, "s"), "Agg", sequential: true)); + + Assert.True(await h.DriveUntilFinished("CarryOn")); + Assert.Equal(["s0", "s1", "s2", "s3"], h.Svc.Started); + var run = (await h.Store.GetRunByNameAsync("CarryOn"))!; + Assert.Equal((1, 0, "CompletedWithErrors"), (run.Failed, run.Cancelled, run.Status)); + Assert.Equal(3, Assert.Single(h.Svc.PostExecs).Lines.Length); + } + + [Fact] + public async Task StopOnFailure_CancelsTheStepsAfterTheFirstFailure_AndStillAggregates() + { + await using var h = await OrchestrationHarness.CreateAsync(); + h.Svc.Body = t => FakeOrchestrator.IdOf(t) == "s2" ? throw new InvalidOperationException("step down") : "{}"; + Assert.True(await h.Start("Stopper", Batch(5, "s"), "Agg", sequential: true, stopOnFailure: true)); + + Assert.True(await h.DriveUntilFinished("Stopper")); + Assert.Equal(["s0", "s1", "s2"], h.Svc.Started); + var run = (await h.Store.GetRunByNameAsync("Stopper"))!; + Assert.Equal((1, 2, "CompletedWithErrors"), (run.Failed, run.Cancelled, run.Status)); + + var done = (await h.Store.GetTasksAsync(run.RunKey, 'D')).OrderBy(t => t.Seq).ToList(); + Assert.Equal(["Completed", "Completed", "Failed", "Cancelled", "Cancelled"], done.Select(t => t.Status)); + Assert.All(done.Skip(3), t => Assert.Equal(WorkStore.StoppedReason("Job_s2"), t.LastError)); + Assert.Equal(2, Assert.Single(h.Svc.PostExecs).Lines.Length); + Assert.Equal(1, h.Svc.Checkouts); + Assert.Equal(1, h.Svc.Reclaims); + } + + [Fact] + public async Task StopOnFailure_WhenTheLastStepFails_HasNothingToCancel() + { + await using var h = await OrchestrationHarness.CreateAsync(); + h.Svc.Body = t => FakeOrchestrator.IdOf(t) == "s2" ? throw new InvalidOperationException("last") : "{}"; + Assert.True(await h.Start("StopLast", Batch(3, "s"), sequential: true, stopOnFailure: true)); + + Assert.True(await h.DriveUntilFinished("StopLast")); + var run = (await h.Store.GetRunByNameAsync("StopLast"))!; + Assert.Equal((1, 0), (run.Failed, run.Cancelled)); + } + + [Fact] + public async Task StopOnFailure_IsIgnoredForAFanOutRun_WhichAlwaysCarriesOn() + { + await using var h = await OrchestrationHarness.CreateAsync(); + h.Svc.Body = t => FakeOrchestrator.IdOf(t) == "f0" ? throw new InvalidOperationException("boom") : "{}"; + Assert.True(await h.Start("FanStop", Batch(4, "f"), stopOnFailure: true)); + + Assert.True(await h.DriveUntilFinished("FanStop")); + var run = (await h.Store.GetRunByNameAsync("FanStop"))!; + Assert.False(run.StopOnFailure); + Assert.Equal(4, h.Svc.Started.Count); + Assert.Equal((1, 0), (run.Failed, run.Cancelled)); + } + + [Fact] + public async Task TheRunsModeIsStoredOnItsHeader() + { + await using var h = await OrchestrationHarness.CreateAsync(); + Assert.True(await h.Start("ModeSeq", Batch(1, "a"), sequential: true, stopOnFailure: true)); + Assert.True(await h.Start("ModeCap", Batch(1, "b"), maxConcurrency: 7)); + + var seq = (await h.Store.GetRunByNameAsync("ModeSeq"))!; + var cap = (await h.Store.GetRunByNameAsync("ModeCap"))!; + Assert.Equal((true, 0, true), (seq.Sequential, seq.MaxConcurrency, seq.StopOnFailure)); + Assert.Equal((false, 7, false), (cap.Sequential, cap.MaxConcurrency, cap.StopOnFailure)); + } +} diff --git a/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs b/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs index 6dd2b0f..85a3e21 100644 --- a/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs +++ b/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs @@ -1,3 +1,4 @@ +using System.Collections.ObjectModel; using System.Collections.Concurrent; using System.Management.Automation; using System.Management.Automation.Runspaces; @@ -172,6 +173,88 @@ public void SelfParent_IsDroppedAtEnqueue() Assert.Null(pending!.ParentRunName); } + /// Run the real Start-CraftOrchestrator inside a production-configured worker, under a caller + /// context that the worker stamps into the runspace, and return what reached the bridge and the result. + private static async Task<(OrchestratorBridge.PendingOrchestration? Pending, string Result)> RunWrapperAsync( + string name, string inputObject, OperationContext.Invocation? caller = null) + { + var script = Path.Combine(AppContext.BaseDirectory, "Runtime", "CraftRuntime", "Start-CraftOrchestrator.ps1"); + var worker = await NewPinnedWorkerAsync(); + try + { + Collection output; + using (OperationContext.Set(caller ?? new OperationContext.Invocation("Push-Task"))) + output = await worker.InvokeScriptAsync(ScriptBlock.Create( + $"try {{ . '{script}'; Start-CraftOrchestrator -InputObject {inputObject} }} catch {{ \"ERROR: $_\" }}")); + var pending = TakePending(name); + if (pending?.BatchFilePath is { } path && File.Exists(path)) File.Delete(path); + return (pending, string.Join(",", output.Select(o => o?.ToString()))); + } + finally + { + worker.Dispose(); + } + } + + [Fact] + public async Task Wrapper_AChildInheritsOnlyItsParentsPriority_AndNamesItsExactParentRun() + { + var parent = new OperationContext.Invocation("Push-Task") { RunName = "WrapParent", RunKey = "WrapParent~8de0a1b2c3d4e5f", Priority = 7 }; + + var (pending, result) = await RunWrapperAsync("WrapChild", + "@{ OrchestratorName = 'WrapChild'; Batch = @(@{ FunctionName = 'X'; TenantFilter = 'a.com' }) }", parent); + + Assert.Equal("Craft-WrapChild", result); + Assert.NotNull(pending); + Assert.Equal(7, pending!.Priority); + Assert.Equal("WrapParent~8de0a1b2c3d4e5f", pending.ParentRunName); + Assert.Equal((false, true, 0, false), (pending.Sequential, pending.AllowCollision, pending.MaxConcurrency, pending.StopOnFailure)); + } + + [Fact] + public async Task Wrapper_AChildsOwnPriorityAndModeWin_OverItsParent() + { + var parent = new OperationContext.Invocation("Push-Task") { RunName = "WrapSeqParent", Priority = 7 }; + + var (pending, _) = await RunWrapperAsync("WrapOwnMode", + "@{ OrchestratorName = 'WrapOwnMode'; Priority = 2; Sequential = $true; StopOnFailure = $true; MaxConcurrency = 3; AllowCollision = $false; Batch = @(@{ FunctionName = 'X' }) }", + parent); + + Assert.NotNull(pending); + Assert.Equal(2, pending!.Priority); + Assert.Equal((true, false, 3, true), (pending.Sequential, pending.AllowCollision, pending.MaxConcurrency, pending.StopOnFailure)); + } + + [Fact] + public async Task Wrapper_WithoutAParent_UsesTheDefaultBand_AndNoLineage() + { + var (pending, _) = await RunWrapperAsync("WrapTopLevel", "@{ OrchestratorName = 'WrapTopLevel'; Batch = @(@{ FunctionName = 'X' }) }"); + + Assert.NotNull(pending); + Assert.Equal(4, pending!.Priority); + Assert.Null(pending.ParentRunName); + } + + [Fact] + public async Task Wrapper_WithoutCollisions_SkipsAndSaysSo_WhileARunOfThatNameIsQueued() + { + OrchestratorBridge.QueueOrchestration("WrapBusy", "[]", 4); + try + { + var (pending, result) = await RunWrapperAsync("WrapBusy", + "@{ OrchestratorName = 'WrapBusy'; AllowCollision = $false; Batch = @(@{ FunctionName = 'X' }) }"); + + Assert.Equal("Craft-WrapBusy-Skipped", result); + Assert.NotNull(pending); // the first, queued directly above + Assert.True(string.IsNullOrEmpty(pending!.BatchFilePath)); + Assert.Null(TakePending("WrapBusy")); // and no second one + } + finally + { + TakePending("WrapBusy"); + } + } + [Fact] public async Task Drain_ReleasesTheParent_WhenTheChildIsNeverCreated() { diff --git a/tests/Craft.Tests/WorkStoreTests.cs b/tests/Craft.Tests/WorkStoreTests.cs index 6bff310..c6332a5 100644 --- a/tests/Craft.Tests/WorkStoreTests.cs +++ b/tests/Craft.Tests/WorkStoreTests.cs @@ -119,6 +119,54 @@ public async Task ATaskThatKeepsDying_IsFailedAfterThreeAttempts() Assert.Equal("CompletedWithErrors", header.Status); } + private static Task CreateModeAsync(WorkStore s, string name, int tasks, bool sequential = false, + bool stopOnFailure = false) + { + var h = Header(name); + return s.CreateRunAsync(new RunHeader + { + RunKey = h.RunKey, + Name = h.Name, + StartedUtc = h.StartedUtc, + TaskScriptName = h.TaskScriptName, + Sequential = sequential, + StopOnFailure = stopOnFailure, + }, Tasks(tasks)); + } + + [Fact] + public async Task ASequentialStepThatKeepsDying_StopsAStopOnFailureRun() + { + var (s, _) = New(); + var run = await CreateModeAsync(s, "SeqPoison", 4, sequential: true, stopOnFailure: true); + for (var i = 0; i < 3; i++) + { + Assert.NotNull(await s.ClaimSequentialAsync(run.RunKey, $"w{i}", TimeSpan.FromMilliseconds(1))); + await Task.Delay(20); + } + + Assert.Null(await s.ClaimSequentialAsync(run.RunKey, "w9", Lease)); + var done = (await s.GetTasksAsync(run.RunKey, 'D')).OrderBy(t => t.Seq).ToList(); + Assert.Equal(["Failed", "Cancelled", "Cancelled", "Cancelled"], done.Select(t => t.Status)); + Assert.True((await s.GetRunAsync(run.RunKey))!.IsFinished); + } + + [Fact] + public async Task ASequentialStepThatKeepsDying_IsFailed_AndTheRunCarriesOn_ByDefault() + { + var (s, _) = New(); + var run = await CreateModeAsync(s, "SeqPoisonCarry", 3, sequential: true); + for (var i = 0; i < 3; i++) + { + Assert.NotNull(await s.ClaimSequentialAsync(run.RunKey, $"w{i}", TimeSpan.FromMilliseconds(1))); + await Task.Delay(20); + } + + var next = await s.ClaimSequentialAsync(run.RunKey, "w9", Lease); + Assert.Equal(1, next!.Seq); + Assert.Equal("Failed", Assert.Single(await s.GetTasksAsync(run.RunKey, 'D')).Status); + } + [Fact] public async Task TwoClaimersRacingForTheSameRows_NeverBothWin() { From ceda47b7d105cda09ce1dccbcc768e525ece833d Mon Sep 17 00:00:00 2001 From: Zacgoose <107489668+Zacgoose@users.noreply.github.com> Date: Tue, 6 Oct 2026 01:44:07 +0800 Subject: [PATCH 10/24] fix(jobs): finish the claim of a job cancelled just after the dispatcher dequeued it - a cancel that no longer finds the job in the queue is now reported by the dispatcher's skip (or the job's start) instead of being dropped, so the claim is finished as Cancelled rather than lapsing and running half an hour later - tests: lease renewal, graceful-stop release, buffered cancel, reprioritise, stale Ready entries, legacy table drop, result cleanup, and the log lines the health tooling parses --- Services/Orchestration/JobManager.cs | 31 ++- Services/Orchestration/OrchestratorService.cs | 2 +- Services/Orchestration/WorkPump.cs | 2 +- tests/Craft.Tests/MemoryTableStore.cs | 13 + tests/Craft.Tests/OrchestrationHarness.cs | 23 +- .../OrchestrationLifecycleTests.cs | 233 ++++++++++++++++++ 6 files changed, 294 insertions(+), 10 deletions(-) create mode 100644 tests/Craft.Tests/OrchestrationLifecycleTests.cs diff --git a/Services/Orchestration/JobManager.cs b/Services/Orchestration/JobManager.cs index 8957f45..ced1c7c 100644 --- a/Services/Orchestration/JobManager.cs +++ b/Services/Orchestration/JobManager.cs @@ -57,6 +57,12 @@ public class JobManager : BackgroundService // ── Tracking ── private readonly ConcurrentDictionary _jobs = new(); + /// + /// Jobs cancelled while queued, with whether the state writer was told. A cancel that finds the job still + /// in the queue tells it straight away; one that lands after the dispatcher has dequeued the job cannot see + /// its descriptor, so the dispatcher (or the job's start) tells it instead. Without that, the claim behind + /// the job was never finished, lapsed half an hour later, and ran after all. + /// private readonly ConcurrentDictionary _cancelledJobIds = new(); private readonly ConcurrentDictionary> _pendingWork = new(); @@ -275,9 +281,11 @@ protected override async Task ExecuteAsync(CancellationToken stoppingToken) // The work ref must go too: CancelJob only marks the id, so leaving the entry here // stranded the captured closure (and everything it captured) in _pendingWork forever — // nothing else ever removes it, not even CleanupOldJobs. - if (_cancelledJobIds.TryRemove(job.Record.Id, out _)) + if (_cancelledJobIds.TryRemove(job.Record.Id, out var notified)) { _pendingWork.TryRemove(job.Record.Id, out _); + if (!notified && job.Descriptor is { } cancelled) + NotifyStateWriter(w => w.Cancelled(cancelled), "cancellation", job.Record.Name); _limiter.ReleaseSlot(); slotHeld = false; continue; @@ -361,6 +369,14 @@ private async Task RunJobAsync(QueuedJob job, CancellationToken ct) }; opScope = OperationContext.Set(parentInvocation); + // Cancelled after the dispatcher's own check but before the job got going. + if (_cancelledJobIds.TryRemove(job.Record.Id, out var notified)) + { + if (!notified && job.Descriptor is { } cancelled) + NotifyStateWriter(w => w.Cancelled(cancelled), "cancellation", job.Record.Name); + return; + } + job.Record.Status = "Running"; job.Record.StartedUtc = DateTime.UtcNow; @@ -536,10 +552,15 @@ public bool CancelJob(string jobId) record.Status = "Cancelled"; record.CompletedUtc = DateTime.UtcNow; record.LastError = "Cancelled by user"; - _cancelledJobIds.TryAdd(jobId, true); + // Under the queue lock, so the dispatcher either still finds the entry (and we tell the state writer + // here) or has already dequeued it (and tells it when it skips the job). JobDescriptor? descriptor; - lock (_queueLock) descriptor = FindLiveEntry(record)?.Descriptor; + lock (_queueLock) + { + descriptor = FindLiveEntry(record)?.Descriptor; + _cancelledJobIds[jobId] = descriptor != null; + } if (descriptor is { } d) NotifyStateWriter(w => w.Cancelled(d), "cancellation", record.Name); @@ -558,6 +579,8 @@ public int CancelRun(string runName) var descriptors = new List(toCancel.Count); lock (_queueLock) { + // Marked under the lock for the same reason as CancelJob: found here means told here. + foreach (var record in toCancel) _cancelledJobIds[record.Id] = false; var wanted = toCancel.ToDictionary(r => r.Id, r => r); foreach (var (entry, _) in _pendingQueue.UnorderedItems) { @@ -566,6 +589,7 @@ public int CancelRun(string runName) if (!ReferenceEquals(entry.Record, rec)) continue; if (_reprioritized.TryGetValue(rec.Id, out var live) && entry.Epoch != live) continue; descriptors.Add(d); + _cancelledJobIds[rec.Id] = true; } } @@ -574,7 +598,6 @@ public int CancelRun(string runName) record.Status = "Cancelled"; record.CompletedUtc = DateTime.UtcNow; record.LastError = "Run cancelled by user"; - _cancelledJobIds.TryAdd(record.Id, true); } foreach (var d in descriptors) diff --git a/Services/Orchestration/OrchestratorService.cs b/Services/Orchestration/OrchestratorService.cs index 132285f..2e27809 100644 --- a/Services/Orchestration/OrchestratorService.cs +++ b/Services/Orchestration/OrchestratorService.cs @@ -711,7 +711,7 @@ public async Task RunStatusSweepLoopAsync(CancellationToken ct) catch (OperationCanceledException) { } } - private async Task LogRunStatusAsync(CancellationToken ct) + internal async Task LogRunStatusAsync(CancellationToken ct) { if (!_logger.IsEnabled(LogLevel.Information)) return; // Running jobs are counted per run name, so runs sharing a name take them oldest first. diff --git a/Services/Orchestration/WorkPump.cs b/Services/Orchestration/WorkPump.cs index d857786..8ceb4d7 100644 --- a/Services/Orchestration/WorkPump.cs +++ b/Services/Orchestration/WorkPump.cs @@ -213,7 +213,7 @@ internal async Task RefillAsync(CancellationToken ct) } /// Renew claims in their last third, so a long buffer wait or a long task never loses its lease. - private async Task RenewAsync(CancellationToken ct) + internal async Task RenewAsync(CancellationToken ct) { var now = Clock(); var due = _inFlight.Where(kv => kv.Value.LeaseUntil - now < _lease / 3).ToList(); diff --git a/tests/Craft.Tests/MemoryTableStore.cs b/tests/Craft.Tests/MemoryTableStore.cs index e3fe843..d9e8a23 100644 --- a/tests/Craft.Tests/MemoryTableStore.cs +++ b/tests/Craft.Tests/MemoryTableStore.cs @@ -20,6 +20,19 @@ internal sealed class MemoryTableStore : ICraftTableStore public int Submits { get; private set; } + /// Tables deleted through , in order. + public List DroppedTables { get; } = []; + + public Task DeleteTableAsync(string table, CancellationToken ct = default) + { + lock (_lock) + { + DroppedTables.Add(table); + _tables.Remove(table); + } + return Task.CompletedTask; + } + private static readonly Comparer<(string, string)> Order = Comparer<(string, string)>.Create((a, b) => { var c = string.CompareOrdinal(a.Item1, b.Item1); diff --git a/tests/Craft.Tests/OrchestrationHarness.cs b/tests/Craft.Tests/OrchestrationHarness.cs index c71d0de..e2e3de8 100644 --- a/tests/Craft.Tests/OrchestrationHarness.cs +++ b/tests/Craft.Tests/OrchestrationHarness.cs @@ -5,6 +5,7 @@ using Craft.PowerShellHost; using Craft.Storage; using Microsoft.Extensions.Configuration; +using Microsoft.Extensions.Logging; using Microsoft.Extensions.Logging.Abstractions; namespace Craft.Tests; @@ -15,8 +16,8 @@ namespace Craft.Tests; /// Tracks the start order and the peak number of tasks running at once. /// internal sealed class FakeOrchestrator(JobManager jobs, WorkStore store, ResultStore results, IConfiguration config, - CraftSettings settings) - : OrchestratorService(NullLogger.Instance, null!, null!, jobs, store, results, config, settings) + CraftSettings settings, ILogger logger) + : OrchestratorService(logger, null!, null!, jobs, store, results, config, settings) { public const string PostExecScript = "Invoke-CraftPostExecution"; @@ -82,6 +83,7 @@ internal sealed class OrchestrationHarness : IAsyncDisposable public required WorkPump Pump { get; init; } public required JobManager Jobs { get; init; } public required ICraftTableStore Tables { get; init; } + public required CapturingLogger Log { get; init; } public static async Task CreateAsync(int poolSize = 4, ICraftTableStore? tables = null, Action? configure = null) @@ -101,11 +103,12 @@ public static async Task CreateAsync(int poolSize = 4, ICr tables ??= new MemoryTableStore(); var store = new WorkStore(NullLogger.Instance, settings, tables); var results = new ResultStore(NullLogger.Instance, settings, tables); - var svc = new FakeOrchestrator(jobs, store, results, config, settings); + var log = new CapturingLogger(); + var svc = new FakeOrchestrator(jobs, store, results, config, settings, log); await svc.ResumeInterruptedRunsAsync(CancellationToken.None); var pump = new WorkPump(NullLogger.Instance, store, jobs, config, settings, svc); _ = Task.Run(() => jobs.StartAsync(CancellationToken.None)); - return new OrchestrationHarness { Svc = svc, Store = store, Pump = pump, Jobs = jobs, Tables = tables }; + return new OrchestrationHarness { Svc = svc, Store = store, Pump = pump, Jobs = jobs, Tables = tables, Log = log }; } public static string Batch(int n, string prefix = "t") => @@ -143,3 +146,15 @@ public async Task> ReadyNamesAsync() public async ValueTask DisposeAsync() => await Jobs.StopAsync(CancellationToken.None); } + +/// Keeps every rendered log line (at any level) for tests that pin what operators and tooling read. +internal sealed class CapturingLogger : ILogger +{ + public readonly ConcurrentQueue<(LogLevel Level, string Message)> Lines = new(); + + public IDisposable? BeginScope(TState state) where TState : notnull => null; + public bool IsEnabled(LogLevel logLevel) => true; + + public void Log(LogLevel logLevel, EventId eventId, TState state, Exception? exception, + Func formatter) => Lines.Enqueue((logLevel, formatter(state, exception))); +} diff --git a/tests/Craft.Tests/OrchestrationLifecycleTests.cs b/tests/Craft.Tests/OrchestrationLifecycleTests.cs new file mode 100644 index 0000000..969ff6e --- /dev/null +++ b/tests/Craft.Tests/OrchestrationLifecycleTests.cs @@ -0,0 +1,233 @@ +using System.Text.RegularExpressions; +using Craft.Configuration; +using Craft.Orchestration; +using Craft.PowerShellHost; +using Craft.Storage; +using Microsoft.Extensions.Configuration; +using Microsoft.Extensions.Logging; +using Microsoft.Extensions.Logging.Abstractions; + +namespace Craft.Tests; + +/// +/// Everything around a run that is not its scheduling mode: leases, shutdown, operator actions, startup cleanup, +/// result cleanup, and the log lines operators and the health tooling read. +/// +public class OrchestrationLifecycleTests +{ + private static string Batch(int n, string prefix) => OrchestrationHarness.Batch(n, prefix); + + private static JobManager NewJobs() + { + var settings = new CraftSettings(); + settings.Worker.BgPoolSize = 4; + var config = new ConfigurationBuilder().AddInMemoryCollection([]).Build(); + var repo = new ScriptRepository(NullLogger.Instance, settings); + var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); + var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); + return new JobManager(NullLogger.Instance, settings, limiter); + } + + private static (WorkStore Store, WorkPump Pump, JobManager Jobs) NewIdlePump(int leaseSeconds = 1800) + { + var settings = new CraftSettings(); + settings.Worker.BgPoolSize = 4; + var store = new WorkStore(NullLogger.Instance, settings, new MemoryTableStore()); + var jobs = NewJobs(); + var config = new ConfigurationBuilder().AddInMemoryCollection(new Dictionary + { + ["JobQueueLeaseSeconds"] = leaseSeconds.ToString(System.Globalization.CultureInfo.InvariantCulture), + }).Build(); + return (store, new WorkPump(NullLogger.Instance, store, jobs, config, settings), jobs); + } + + private static async Task CreateAsync(WorkStore s, string name, int tasks) + { + var started = DateTime.UtcNow; + return await s.CreateRunAsync(new RunHeader + { + RunKey = WorkStore.RunKeyFor(name, started), + Name = name, + StartedUtc = started, + TaskScriptName = "Invoke-CraftTask", + }, Enumerable.Range(0, tasks).Select(i => new WorkStore.NewTask($"t{i}", [])).ToList()); + } + + // ── leases and shutdown ── + + [Fact] + public async Task AClaimHeldPastTwoThirdsOfItsLease_IsRenewed_SoALongTaskNeverLosesIt() + { + var (store, pump, _) = NewIdlePump(leaseSeconds: 60); + var run = await CreateAsync(store, "Long", 2); + var now = DateTime.UtcNow; + pump.Clock = () => now; + Assert.Equal(2, await pump.RefillAsync(CancellationToken.None)); + var before = (await store.GetTasksAsync(run.RunKey, 'R')).Select(t => t.LeaseUntil!.Value).Min(); + + now = now.AddSeconds(30); // half way: not yet due + await Task.Delay(20); + await pump.RenewAsync(CancellationToken.None); + Assert.Equal(before, (await store.GetTasksAsync(run.RunKey, 'R')).Select(t => t.LeaseUntil!.Value).Min()); + + now = now.AddSeconds(15); // three quarters: renewed + await pump.RenewAsync(CancellationToken.None); + Assert.All(await store.GetTasksAsync(run.RunKey, 'R'), t => Assert.True(t.LeaseUntil > before)); + } + + [Fact] + public async Task OnShutdown_ClaimsThatNeverStarted_AreHandedBack_WithTheirAttemptRefunded() + { + var (store, pump, _) = NewIdlePump(); + var run = await CreateAsync(store, "Stopping", 6); + Assert.Equal(4, await pump.RefillAsync(CancellationToken.None)); + + await pump.StopAsync(CancellationToken.None); + + Assert.Empty(await store.GetTasksAsync(run.RunKey, 'R')); + var pending = await store.GetTasksAsync(run.RunKey, 'P'); + Assert.Equal(6, pending.Count); + Assert.All(pending, t => Assert.Equal(0, t.Attempt)); + } + + /// + /// Depending on timing, the cancel lands while the job is still queued or just after the dispatcher has + /// dequeued it. The second case once dropped the state-writer notification: the claim was never finished, + /// lapsed half an hour later and ran after all. Either way the task must end up Cancelled. + /// + [Fact] + public async Task CancellingABufferedJob_RecordsItsTaskCancelled_SoTheClaimCannotLapseAndRunIt() + { + await using var h = await OrchestrationHarness.CreateAsync(poolSize: 1); + var gate = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + h.Svc.BeforeRun = t => FakeOrchestrator.IdOf(t) == "b0" ? gate.Task : Task.CompletedTask; + Assert.True(await h.Start("Buffered", Batch(3, "b"))); + var run = (await h.Store.GetRunByNameAsync("Buffered"))!; + + Assert.True(await h.DriveUntil(() => Task.FromResult(h.Jobs.GetJobs(status: "Queued").Count > 0)), "no job was ever buffered"); + var queued = h.Jobs.GetJobs(status: "Queued").First(); + Assert.True(h.Jobs.CancelJob(queued.Id)); + gate.SetResult(); + + var finished = await h.DriveUntilFinished("Buffered"); + var state = string.Join(",", (await h.Store.GetTasksAsync(run.RunKey)).Select(t => $"{t.TaskId}:{t.State}:{t.Status}:{t.Owner}")); + Assert.True(finished, $"run did not finish; started={string.Join(",", h.Svc.Started)} tasks={state} cancelled={queued.Name}"); + var done = await h.Store.GetTasksAsync(run.RunKey, 'D'); + Assert.Single(done, t => t.Status == "Cancelled"); + Assert.Equal(2, h.Svc.Started.Count); + } + + // ── operator actions ── + + [Fact] + public async Task Reprioritizing_MovesTheWholeRunToItsNewBand() + { + await using var h = await OrchestrationHarness.CreateAsync(); + Assert.True(await h.Start("First", Batch(2, "f"), priority: 4)); + Assert.True(await h.Start("Second", Batch(2, "s"), priority: 4)); + + Assert.True(await h.Svc.ReprioritizeRunAsync("Second", 1)); + + Assert.Equal(["Second", "First"], await h.ReadyNamesAsync()); + Assert.Equal(1, (await h.Store.GetRunByNameAsync("Second"))!.Priority); + } + + [Fact] + public async Task CancellingOneQueuedTaskByName_FindsItInWhicheverStackedRunHoldsIt() + { + await using var h = await OrchestrationHarness.CreateAsync(); + Assert.True(await h.Start("Twin", Batch(2, "a"))); + Assert.True(await h.Start("Twin", Batch(2, "b"))); + + Assert.True(await h.Svc.TryCancelQueuedTaskAsync("Twin", "Job_b1")); + Assert.False(await h.Svc.TryCancelQueuedTaskAsync("Twin", "Job_b1")); + Assert.False(await h.Svc.TryCancelQueuedTaskAsync("Twin", "Job_nope")); + } + + [Fact] + public async Task ClearingTheQueue_CancelsEveryPendingTaskOfEveryRun() + { + await using var h = await OrchestrationHarness.CreateAsync(); + Assert.True(await h.Start("A", Batch(3, "a"))); + Assert.True(await h.Start("B", Batch(4, "b"), priority: 9)); + + Assert.Equal(7, await h.Svc.ClearQueueAsync()); + Assert.Empty(await h.ReadyNamesAsync()); + } + + [Fact] + public async Task AReadyEntryWhoseRunIsGone_IsDroppedByThePump() + { + await using var h = await OrchestrationHarness.CreateAsync(); + Assert.True(await h.Start("Vanished", Batch(2, "v"))); + var run = (await h.Store.GetRunByNameAsync("Vanished"))!; + await h.Tables.DeletePartitionAsync("OrchestratorWork", run.RunKey); + + await h.Pump.RefillAsync(CancellationToken.None); + Assert.Empty(await h.ReadyNamesAsync()); + } + + // ── startup and cleanup ── + + [Fact] + public async Task Startup_DropsThePreviousDesignsTables() + { + var tables = new MemoryTableStore(); + await using var h = await OrchestrationHarness.CreateAsync(tables: tables); + + Assert.Equal(["OrchestratorQueue", "OrchestratorQueueIndex", "OrchestratorTasks", "OrchestratorRuns", "OrchestratorResults"], + tables.DroppedTables); + } + + [Fact] + public async Task ACompletedRunsResults_AreDeleted_OnceItsAggregationHasReadThem() + { + var tables = new MemoryTableStore(); + await using var h = await OrchestrationHarness.CreateAsync(tables: tables); + Assert.True(await h.Start("Results", Batch(5, "r"), "Agg")); + + Assert.True(await h.DriveUntilFinished("Results")); + Assert.Equal(5, Assert.Single(h.Svc.PostExecs).Lines.Length); + Assert.Empty(tables.All("OrchestratorTaskResults")); + } + + // ── what operators and tooling read ── + + [Fact] + public async Task TheLogLines_HealthToolingParses_KeepTheirShape() + { + await using var h = await OrchestrationHarness.CreateAsync(poolSize: 1); + var gate = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + h.Svc.BeforeRun = t => FakeOrchestrator.IdOf(t) == "l0" ? gate.Task : Task.CompletedTask; + h.Svc.Body = t => FakeOrchestrator.IdOf(t) == "l2" ? throw new InvalidOperationException("boom") : "{}"; + Assert.True(await h.Start("Logged", Batch(3, "l"), "Agg", priority: 6)); + Assert.True(await h.DriveUntil(() => Task.FromResult(h.Svc.Started.Contains("l0")))); + await h.Svc.LogRunStatusAsync(CancellationToken.None); + gate.SetResult(); + Assert.True(await h.DriveUntilFinished("Logged")); + + var lines = h.Log.Lines.Select(l => l.Message).ToList(); + // The patterns Get-CraftInstanceStatus.ps1 matches; change them together or not at all. + Assert.Contains(lines, l => Regex.IsMatch(l, @"\] Run Logged created with 3 tasks at P6")); + Assert.Contains(lines, l => Regex.IsMatch(l, @"\] Run Logged T\+[\d.]+min: \d+/3 done 1 running 2 pending 0 failed")); + Assert.Contains(lines, l => l.Contains("Dispatching PostExecution") && Regex.IsMatch(l, @"for run Logged\b")); + Assert.Contains(lines, l => Regex.IsMatch(l, @"\] Run Logged finalized: CompletedWithErrors \(2/1/0/3\)")); + Assert.Contains(h.Log.Lines, l => l.Level == LogLevel.Debug && l.Message.StartsWith("[Scheduler] Task completed: ", StringComparison.Ordinal)); + } + + [Fact] + public async Task APostExecutionThatGivesUp_SaysSoInTheLineTheToolingMatches() + { + await using var h = await OrchestrationHarness.CreateAsync(); + h.Svc.PostExecBody = () => throw new InvalidOperationException("aggregate down"); + Assert.True(await h.Start("GivesUp", Batch(1, "g"), "Agg")); + + Assert.True(await h.DriveUntil(async () => + { + h.Pump.ForgetBackoff(); + return await h.Store.GetRunByNameAsync("GivesUp") is { IsFinished: true }; + })); + Assert.Contains(h.Log.Lines, l => l.Message.Contains("giving up and cleaning up results") + && Regex.IsMatch(l.Message, @"PostExecution for GivesUp\b")); + } +} From 8667f174d290e87266501e430bdb0fe4e12e2bce Mon Sep 17 00:00:00 2001 From: Zacgoose <107489668+Zacgoose@users.noreply.github.com> Date: Tue, 6 Oct 2026 01:45:27 +0800 Subject: [PATCH 11/24] test(orchestration): pin end-to-end table calls per task and per small run --- tests/Craft.Tests/OrchestrationCostTests.cs | 39 +++++++++++++++++++++ 1 file changed, 39 insertions(+) diff --git a/tests/Craft.Tests/OrchestrationCostTests.cs b/tests/Craft.Tests/OrchestrationCostTests.cs index 51be1ef..74e839d 100644 --- a/tests/Craft.Tests/OrchestrationCostTests.cs +++ b/tests/Craft.Tests/OrchestrationCostTests.cs @@ -229,6 +229,45 @@ public async Task TheRetentionSweep_ReadsOnlyRunsPastTheCutoff() Assert.Equal(20, await s.SweepFinishedAsync(TimeSpan.Zero)); } + [Fact] + public async Task AWholeFanOut_CostsAFewTableCallsPerTask_EndToEnd() + { + var count = new CountingTableStore(new MemoryTableStore()); + await using var h = await OrchestrationHarness.CreateAsync(poolSize: 8, tables: count); + count.Reset(); + Assert.True(await h.Start("Throughput", OrchestrationHarness.Batch(1_000, "t"), "Agg")); + + Assert.True(await h.DriveUntilFinished("Throughput", 60_000)); + var t = count.Total(); + output.WriteLine($"1,000-task fan-out end to end: {t} work {count.For(Work)}"); + // Measured: see the output line. Bounds leave headroom for scheduling noise, not for a new per-task call. + Assert.InRange(t.Submits / 1000.0, 0, PerTaskSubmits); + Assert.InRange(t.PointReads / 1000.0, 0, PerTaskPointReads); + Assert.InRange(t.Queries / 1000.0, 0, PerTaskQueries); + Assert.InRange((t.Upserts + t.BatchUpserts) / 1000.0, 0, PerTaskWrites); + } + + [Fact] + public async Task ManySmallRuns_CostAFewTableCallsPerRun_EndToEnd() + { + var count = new CountingTableStore(new MemoryTableStore()); + await using var h = await OrchestrationHarness.CreateAsync(poolSize: 8, tables: count); + count.Reset(); + for (var i = 0; i < 300; i++) Assert.True(await h.Start($"Small{i}", OrchestrationHarness.Batch(1, $"s{i}_"))); + + Assert.True(await h.DriveUntilAllFinished(60_000)); + var t = count.Total(); + output.WriteLine($"300 single-task runs end to end: {t}"); + Assert.InRange((t.Submits + t.PointReads + t.Queries + t.Upserts + t.BatchUpserts + t.Deletes) / 300.0, 0, PerSmallRunCalls); + } + + // Measured 2026-10-06 (three runs, steady): per task 0.25 transactions (claims and finishes batched), + // 3.66 point reads (header and payload at dispatch, the claimed row and header at finish), 0.39 queries and + // 1.13 writes (each task's result row for the aggregation, plus a Ready update per finish batch); per + // single-task run 21.5 calls in all. Bounds are about 1.5x. + private const double PerTaskSubmits = 0.4, PerTaskPointReads = 5.5, PerTaskQueries = 0.6, PerTaskWrites = 1.7, + PerSmallRunCalls = 32; + /// /// The case that wedges a big instance: thousands of runs that have nothing to claim (waiting on children, /// or on claims held elsewhere) sit ahead of one run that does. The pump gives each run it looks at a From 7cc9a3b45e4b82dc797c347fd9ef8d6af2811c47 Mon Sep 17 00:00:00 2001 From: Zacgoose <107489668+Zacgoose@users.noreply.github.com> Date: Tue, 6 Oct 2026 02:33:04 +0800 Subject: [PATCH 12/24] fix(powershell): keep a call's output when the previous call on the worker threw - invocations pass their own output collection; after a failed async invocation the next EndInvoke returned null, so the following call on that worker lost its output (a sequential run dropped the result of the step after each failure) --- Services/PowerShellHost/PowerShellWorker.cs | 17 ++++--- .../PowerShellWorkerOutputTests.cs | 49 +++++++++++++++++++ 2 files changed, 60 insertions(+), 6 deletions(-) create mode 100644 tests/Craft.Tests/PowerShellWorkerOutputTests.cs diff --git a/Services/PowerShellHost/PowerShellWorker.cs b/Services/PowerShellHost/PowerShellWorker.cs index 33a0be5..f57ba9c 100644 --- a/Services/PowerShellHost/PowerShellWorker.cs +++ b/Services/PowerShellHost/PowerShellWorker.cs @@ -337,13 +337,16 @@ public async Task> InvokeAsync(string functionName, Diction if (prof) buildTicks = System.Diagnostics.Stopwatch.GetTimestamp() - bStart; var rStart = prof ? System.Diagnostics.Stopwatch.GetTimestamp() : 0; - var asyncResult = _pwsh.BeginInvoke(); - var results = await Task.Factory.FromAsync(asyncResult, _pwsh.EndInvoke); + // A fresh output collection per invocation: after an invocation that threw, the PowerShell object's + // own output buffer comes back null from the next EndInvoke, silently dropping that call's output. + using var outputs = new PSDataCollection(); + var asyncResult = _pwsh.BeginInvoke(null, outputs); + await Task.Factory.FromAsync(asyncResult, _pwsh.EndInvoke); ct.ThrowIfCancellationRequested(); if (prof) runTicks = System.Diagnostics.Stopwatch.GetTimestamp() - rStart; var cpStart = prof ? System.Diagnostics.Stopwatch.GetTimestamp() : 0; - var coll = new Collection(results?.ToList() ?? new List()); + var coll = new Collection(outputs.ReadAll()); if (prof) copyTicks = System.Diagnostics.Stopwatch.GetTimestamp() - cpStart; return coll; } @@ -382,11 +385,13 @@ public async Task> InvokeScriptAsync(ScriptBlock scriptBloc if (ct.CanBeCanceled) registration = ct.Register(() => _pwsh.Stop()); - var asyncResult = _pwsh.BeginInvoke(); - var results = await Task.Factory.FromAsync(asyncResult, _pwsh.EndInvoke); + // A fresh output collection per invocation; see InvokeAsync. + using var outputs = new PSDataCollection(); + var asyncResult = _pwsh.BeginInvoke(null, outputs); + await Task.Factory.FromAsync(asyncResult, _pwsh.EndInvoke); ct.ThrowIfCancellationRequested(); - return new Collection(results?.ToList() ?? new List()); + return new Collection(outputs.ReadAll()); } catch (PipelineStoppedException) when (ct.IsCancellationRequested) { diff --git a/tests/Craft.Tests/PowerShellWorkerOutputTests.cs b/tests/Craft.Tests/PowerShellWorkerOutputTests.cs new file mode 100644 index 0000000..11a4f9f --- /dev/null +++ b/tests/Craft.Tests/PowerShellWorkerOutputTests.cs @@ -0,0 +1,49 @@ +using System.Management.Automation; +using System.Management.Automation.Runspaces; +using Craft.PowerShellHost; +using Microsoft.Extensions.Logging.Abstractions; + +namespace Craft.Tests; + +/// +/// A worker is reused across invocations, and an invocation that throws must not cost the next one its output. +/// It did: after a failed asynchronous invocation, the PowerShell object's own output buffer came back null from +/// the next EndInvoke, so that call returned nothing. A sequential run lost the result of the step after every +/// failed step; any reused worker lost one call's output after any failure. +/// +public class PowerShellWorkerOutputTests +{ + private static async Task NewWorkerAsync() + { + var iss = InitialSessionState.CreateDefault2(); + iss.Commands.Add(new SessionStateFunctionEntry("Step", "param($N) if ($N -eq 2) { throw \"boom $N\" }; \"out$N\"")); + var worker = new PowerShellWorker(96, iss, NullLogger.Instance); + worker.Runspace.ThreadOptions = PSThreadOptions.ReuseThread; + if (worker.Runspace.RunspaceStateInfo.State == RunspaceState.BeforeOpen) worker.Runspace.Open(); + await worker.InvokeScriptAsync(ScriptBlock.Create("$null")); + return worker; + } + + [Fact] + public async Task TheCallAfterAFailedCall_StillReturnsItsOutput() + { + using var worker = await NewWorkerAsync(); + var outputs = new List(); + for (var i = 0; i < 5; i++) + { + try { outputs.Add(string.Join(",", (await worker.InvokeAsync("Step", new() { ["N"] = i })).Select(o => o.ToString()))); } + catch (RuntimeException) { outputs.Add("threw"); } + } + + Assert.Equal(["out0", "out1", "threw", "out3", "out4"], outputs); + } + + [Fact] + public async Task TheScriptAfterAFailedScript_StillReturnsItsOutput() + { + using var worker = await NewWorkerAsync(); + await Assert.ThrowsAsync(() => worker.InvokeScriptAsync(ScriptBlock.Create("throw 'boom'"))); + + Assert.Equal("after", Assert.Single(await worker.InvokeScriptAsync(ScriptBlock.Create("'after'"))).ToString()); + } +} From d0e22d54e5800b17d29af9fdeeba82804aab2918 Mon Sep 17 00:00:00 2001 From: Zacgoose <107489668+Zacgoose@users.noreply.github.com> Date: Tue, 6 Oct 2026 02:33:05 +0800 Subject: [PATCH 13/24] feat(orchestration): one instance works the queue, with repairable indexes and run diagnostics - instance lock row: the pump claims nothing until it holds it, keeps it while its own claimed tasks still run on shutdown, then releases it; claims and sequential drivers of any other process are taken back on sight while held - owner ids are unique per process start - in-process run-changed events wake the pump on create, finish, release, child and band changes, and also refill as soon as the job buffer drains - active-run row written first and removed last; Ready and Finished are rebuilt from it at startup and whenever an index write has failed for good - index writes retried; failed finishes retried in the background for 30 min - a task finished twice in one batch is applied once (one entity per transaction) - range reads send a page size; status, claim and cancel reads are bounded; queued rows carry their run key and position so cancelling one is a point read - status views no longer count claims made after the cached snapshot twice - InspectRun and RepairIndexes on the bridge --- Services/Bridges/OrchestratorBridge.cs | 21 + Services/Bridges/WorkerMetricsBridge.cs | 2 +- Services/Orchestration/FinishBatcher.cs | 62 ++- Services/Orchestration/JobManager.cs | 9 + .../Orchestration/JobQueueStatusReader.cs | 40 +- Services/Orchestration/OrchestratorService.cs | 99 ++++- Services/Orchestration/WorkPump.cs | 182 ++++++++- Services/Storage/AzureTableStore.cs | 5 +- Services/Storage/ICraftTableStore.cs | 6 +- Services/Storage/WorkStore.cs | 261 +++++++++++-- tests/Craft.Tests/CountingTableStore.cs | 14 +- tests/Craft.Tests/FaultyTableStore.cs | 43 +++ .../Craft.Tests/JobQueueStatusReaderTests.cs | 29 ++ tests/Craft.Tests/MemoryTableStore.cs | 2 +- .../Craft.Tests/OrchestrationAzuriteTests.cs | 124 +++++- tests/Craft.Tests/OrchestrationHarness.cs | 12 +- .../Craft.Tests/OrchestrationRecoveryTests.cs | 365 ++++++++++++++++++ .../OrchestrationThroughputTests.cs | 73 ++++ .../OrchestratorBridgeLineageTests.cs | 15 +- tests/Craft.Tests/WorkStoreTests.cs | 4 +- 20 files changed, 1282 insertions(+), 86 deletions(-) create mode 100644 tests/Craft.Tests/FaultyTableStore.cs create mode 100644 tests/Craft.Tests/OrchestrationRecoveryTests.cs create mode 100644 tests/Craft.Tests/OrchestrationThroughputTests.cs diff --git a/Services/Bridges/OrchestratorBridge.cs b/Services/Bridges/OrchestratorBridge.cs index 128b6ac..815b445 100644 --- a/Services/Bridges/OrchestratorBridge.cs +++ b/Services/Bridges/OrchestratorBridge.cs @@ -125,6 +125,27 @@ public static bool IsRunActive(string name) return s_service != null && Task.Run(() => s_service.IsRunActiveAsync(name)).GetAwaiter().GetResult(); } + private static readonly System.Text.Json.JsonSerializerOptions s_inspectJson = new() { WriteIndented = true }; + + /// + /// Why a run is or is not moving, as JSON: counts and mode, whether the scheduler can see it, its claims + /// and who holds them, child runs it waits for, its aggregation, the instance lock, and a diagnosis. + /// Takes a run key, or a run name (every unfinished run of it, else the latest). + /// PS usage: [Craft.Services.OrchestratorBridge]::InspectRun('MailboxRules_contoso.com'). + /// + public static string InspectRun(string nameOrKey) => s_service == null + ? "{\"error\":\"orchestrator not initialised\"}" + : System.Text.Json.JsonSerializer.Serialize(Task.Run(() => s_service.InspectRunAsync(nameOrKey)).GetAwaiter().GetResult(), s_inspectJson); + + /// + /// Rebuild the Ready and Finished indexes from the active-run list now, as the pump does at startup: relists + /// unfinished runs, retires finished ones, removes runs whose creation never finished. Returns a JSON summary. + /// PS usage: [Craft.Services.OrchestratorBridge]::RepairIndexes(). + /// + public static string RepairIndexes() => s_service == null + ? "{\"error\":\"orchestrator not initialised\"}" + : System.Text.Json.JsonSerializer.Serialize(Task.Run(() => s_service.RepairIndexesAsync()).GetAwaiter().GetResult(), s_inspectJson); + /// Synchronous drain — blocks until all pending orchestrations are started. public static void DrainPending() { diff --git a/Services/Bridges/WorkerMetricsBridge.cs b/Services/Bridges/WorkerMetricsBridge.cs index c40d3ce..cbe09a0 100644 --- a/Services/Bridges/WorkerMetricsBridge.cs +++ b/Services/Bridges/WorkerMetricsBridge.cs @@ -660,7 +660,7 @@ public static bool CancelJob(string jobId) var row = snap?.Rows.FirstOrDefault(r => !r.Claimed && $"{r.RunName}-{r.TaskId}" == jobId); if (row == null) return false; - return await s_orchestrator.TryCancelQueuedTaskAsync(row.RunName, row.TaskId); + return await s_orchestrator.TryCancelQueuedTaskAsync(row.RunKey, row.Seq); }); } diff --git a/Services/Orchestration/FinishBatcher.cs b/Services/Orchestration/FinishBatcher.cs index ecbb49e..2090070 100644 --- a/Services/Orchestration/FinishBatcher.cs +++ b/Services/Orchestration/FinishBatcher.cs @@ -4,8 +4,13 @@ namespace Craft.Orchestration; /// /// Coalesces task finishes per run, so tasks of one run finishing together share one transaction instead of -/// each paying for its own (and racing each other on the run header). Callers await their own outcome: a -/// finish is durable when the await returns, which is what lets a worker slot go. +/// each paying for its own (and racing each other on the run header). Callers await their own outcome. +/// +/// A finish that cannot be written (storage unavailable, or losing races for longer than the store retries) +/// is not dropped: it is retried in the background, backing off to a minute between attempts, for +/// . Each attempt is guarded by the claim's owner, so a claim taken over meanwhile is +/// never overwritten. Only if every attempt fails does the task fall back to its claim lapsing and running +/// again, and that is logged as an error naming the run. /// public sealed class FinishBatcher(WorkStore store, ILogger logger, TimeSpan? window = null) { @@ -13,6 +18,16 @@ public sealed class FinishBatcher(WorkStore store, ILogger logger, TimeSpan? win private readonly object _lock = new(); private readonly Dictionary Done)>> _pending = new(StringComparer.Ordinal); + /// How long a finish that cannot be written keeps being retried. + internal TimeSpan GiveUpAfter { get; set; } = TimeSpan.FromMinutes(30); + + /// The first background retry delay, doubling to a minute; tests shorten it. + internal TimeSpan FirstRetry { get; set; } = TimeSpan.FromSeconds(1); + + /// Finishes waiting on a background retry, for status and tests. + public int Retrying => Volatile.Read(ref _retrying); + private int _retrying; + public Task FinishAsync(string runKey, WorkStore.Finish finish) { var done = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); @@ -37,16 +52,53 @@ private async Task FlushAfterWindowAsync(string runKey) _pending.Remove(runKey); } + var finishes = batch.Select(b => b.Finish).ToList(); try { - var outcome = await store.FinishAsync(runKey, batch.Select(b => b.Finish).ToList()); + var outcome = await store.FinishAsync(runKey, finishes); foreach (var (_, done) in batch) done.TrySetResult(outcome); } catch (Exception ex) { - logger.LogWarning(ex, "[Orchestrator] Could not record {Count} finished task(s) of {Run}; their leases lapse and they run again", + logger.LogWarning(ex, "[Orchestrator] Could not record {Count} finished task(s) of {Run}; retrying in the background", batch.Count, runKey); - foreach (var (_, done) in batch) done.TrySetException(ex); + foreach (var (_, done) in batch) done.TrySetResult(null); + _ = RetryAsync(runKey, finishes); + } + } + + private async Task RetryAsync(string runKey, List finishes) + { + Interlocked.Add(ref _retrying, finishes.Count); + try + { + var delay = FirstRetry; + var giveUpAt = DateTime.UtcNow + GiveUpAfter; + for (var attempt = 1; ; attempt++) + { + await Task.Delay(delay); + try + { + await store.FinishAsync(runKey, finishes); + logger.LogInformation("[Orchestrator] Recorded {Count} finished task(s) of {Run} after {Attempts} retr{Ies}", + finishes.Count, runKey, attempt, attempt == 1 ? "y" : "ies"); + return; + } + catch (Exception ex) + { + if (DateTime.UtcNow >= giveUpAt) + { + logger.LogError(ex, "[Orchestrator] Gave up recording {Count} finished task(s) of {Run} after {Attempts} attempts; they run again when their claims lapse", + finishes.Count, runKey, attempt + 1); + return; + } + delay = delay * 2 > TimeSpan.FromMinutes(1) ? TimeSpan.FromMinutes(1) : delay * 2; + } + } + } + finally + { + Interlocked.Add(ref _retrying, -finishes.Count); } } } diff --git a/Services/Orchestration/JobManager.cs b/Services/Orchestration/JobManager.cs index ced1c7c..06d33d1 100644 --- a/Services/Orchestration/JobManager.cs +++ b/Services/Orchestration/JobManager.cs @@ -85,6 +85,10 @@ public class JobManager : BackgroundService public int ActiveCount => _activeCount; public int QueuedCount { get { lock (_queueLock) return _pendingQueue.Count; } } + /// Raised each time a job leaves the queue for a worker, so a feeder can top the queue up at once + /// instead of waiting for its next poll. + public event Action? Dispatched; + /// /// Is this job still in flight — queued or running? /// @@ -261,6 +265,11 @@ protected override async Task ExecuteAsync(CancellationToken stoppingToken) { _pendingQueue.TryDequeue(out job, out _); } + if (job != null) + { + try { Dispatched?.Invoke(); } + catch (Exception ex) { _logger.LogDebug(ex, "[JobManager] A dispatch listener failed"); } + } if (job == null) { diff --git a/Services/Orchestration/JobQueueStatusReader.cs b/Services/Orchestration/JobQueueStatusReader.cs index 2c81328..3ea970a 100644 --- a/Services/Orchestration/JobQueueStatusReader.cs +++ b/Services/Orchestration/JobQueueStatusReader.cs @@ -33,10 +33,18 @@ public JobQueueStatusReader(ILogger logger, JobManager job } /// A task waiting in storage, as the job listings show it. - public sealed record QueuedRow(string RunName, string TaskId, int Priority, DateTime QueuedUtc, bool Claimed); + /// A task waiting in storage, as the job listings show it; and + /// address its row, so acting on it is a point read. + public sealed record QueuedRow(string RunName, string TaskId, int Priority, DateTime QueuedUtc, bool Claimed, + string RunKey = "", int Seq = 0); + /// A run name's durable counts (summed over runs sharing the name). Merged views subtract what this + /// process holds right now, never the counts held when the snapshot was taken. public sealed record RunQueueInfo(int Unclaimed, int Claimed, int MinPriority, DateTime? OldestQueuedUtc, - int Total, int Done, string? Reference); + int Total, int Done, string? Reference, int Failed = 0) + { + public int Outstanding => Math.Max(0, Total - Done); + } /// One Ready scan. is the head of the queue only; the counts cover every run. public sealed record QueueSnapshot(DateTime TakenUtc, IReadOnlyList Rows, int Total, int Unclaimed, @@ -108,17 +116,14 @@ private async Task BuildSnapshotAsync(CancellationToken ct) if (outstanding > claimed && (oldest == null || e.StartedUtc < oldest)) oldest = e.StartedUtc; byRun[e.Name] = byRun.TryGetValue(e.Name, out var same) ? new RunQueueInfo(same.Unclaimed + outstanding - claimed, same.Claimed + claimed, Math.Min(same.MinPriority, e.Band), - same.OldestQueuedUtc, same.Total + e.Total, same.Done + e.Done, same.Reference ?? e.Reference) - : new RunQueueInfo(outstanding - claimed, claimed, e.Band, e.StartedUtc, e.Total, e.Done, e.Reference); + same.OldestQueuedUtc, same.Total + e.Total, same.Done + e.Done, same.Reference ?? e.Reference, same.Failed + e.Failed) + : new RunQueueInfo(outstanding - claimed, claimed, e.Band, e.StartedUtc, e.Total, e.Done, e.Reference, e.Failed); if (head.Count < HeadRows && runsListed < HeadRuns && outstanding > claimed) { runsListed++; - foreach (var t in await _store.GetTasksAsync(e.RunKey, 'P', ct)) - { - if (head.Count >= HeadRows) break; - head.Add(new QueuedRow(e.Name, t.TaskId, e.Band, e.StartedUtc, false)); - } + foreach (var t in await _store.GetTasksAsync(e.RunKey, 'P', HeadRows - head.Count, ct)) + head.Add(new QueuedRow(e.Name, t.TaskId, e.Band, e.StartedUtc, false, e.RunKey, t.Seq)); } } @@ -135,8 +140,11 @@ public async Task GetSummaryAsync(CancellationToken ct = default) var snap = await GetAsync(ct: ct); if (snap == null) return summary; - summary.QueuedDurable = snap.Unclaimed; - summary.Queued += snap.Unclaimed; + // Waiting in storage = everything outstanding less what this process holds now (its queued and running + // orchestrator jobs). Using the snapshot's own subtraction would count claims made since it was taken twice. + var held = _jobs.GetJobs().Count(j => j.RunName != null && j.Status is "Queued" or "Running"); + summary.QueuedDurable = Math.Max(0, snap.Total - held); + summary.Queued += summary.QueuedDurable; if (snap.OldestUnclaimedUtc is { } oldest && (summary.OldestQueuedUtc == null || oldest < summary.OldestQueuedUtc)) summary.OldestQueuedUtc = oldest; return summary; @@ -190,10 +198,14 @@ public async Task> GetRunSummariesAsync(CancellationToken ct summaries.Add(summary); byName[run] = summary; } + // Storage is the truth for a run still going: outstanding work is running here or waiting, and done + // splits into completed and failed. Local job history can include earlier outings of the name. summary.Reference ??= info.Reference; - summary.Queued += info.Unclaimed; - summary.Total = Math.Max(summary.Total, info.Total); - summary.Completed = Math.Max(summary.Completed, info.Done - summary.Failed); + summary.Total = info.Total; + summary.Running = Math.Min(summary.Running, info.Outstanding); + summary.Queued = info.Outstanding - summary.Running; + summary.Failed = info.Failed; + summary.Completed = Math.Max(0, info.Done - info.Failed); summary.CompletedUtc = null; } diff --git a/Services/Orchestration/OrchestratorService.cs b/Services/Orchestration/OrchestratorService.cs index 2e27809..ddd965c 100644 --- a/Services/Orchestration/OrchestratorService.cs +++ b/Services/Orchestration/OrchestratorService.cs @@ -30,9 +30,14 @@ public class OrchestratorService : IJobDescriptorStateWriter private readonly ConcurrentDictionary _scripts = new(StringComparer.OrdinalIgnoreCase); private readonly ConcurrentDictionary _lastStatusLog = new(); - /// Identifies this process's claims, and is what a lease is checked against. + /// Identifies this process's claims, and is what a lease is checked against. Unique per process start, + /// so a container restarted under the same host name never mistakes its predecessor's claims for its own. public string Owner { get; } + /// {host}/{pid}/{random}: readable in a claim row, unique per process start. + public static string NewOwnerId() => + $"{Environment.GetEnvironmentVariable("HOSTNAME") ?? Environment.MachineName}/{Environment.ProcessId}/{Guid.NewGuid():N}"[..^24]; + /// How long a claim is held before anyone may take it back. Longer than any task may run. public TimeSpan Lease { get; } @@ -70,7 +75,7 @@ public OrchestratorService( _store = store; _results = results; _settings = settings; - Owner = Environment.GetEnvironmentVariable("HOSTNAME") ?? $"instance-{Environment.ProcessId}"; + Owner = NewOwnerId(); Lease = TimeSpan.FromSeconds(Math.Max(60, configuration.GetValue("JobQueueLeaseSeconds", 1800))); _finisher = new FinishBatcher(store, logger); _store.AfterFinish = AfterFinishAsync; @@ -536,7 +541,7 @@ private Func BuildSequentialRunWork(RunHeader header, J _logger.LogError(ex, "[Scheduler] Sequential task failed: {TaskId} — continuing with the next step", step.TaskId); } - if (await _store.ClaimSequentialAsync(header.RunKey, Owner, Lease, continuing: true, jobCt) is not { } next) break; + if (await _store.ClaimSequentialAsync(header.RunKey, Owner, Lease, continuing: true, ct: jobCt) is not { } next) break; step = next; } } @@ -589,7 +594,16 @@ private static Dictionary TaskInvocation(Dictionary 0, total); } - /// Cancel one task that is still pending in storage. False when it is not pending (or not found). + /// Cancel one pending task by its row (from a queue listing): a point read. False when it is not pending. + public async Task TryCancelQueuedTaskAsync(string runKey, int seq) + { + if (await _store.GetPendingAsync(runKey, seq) == null) return false; + var outcome = await _store.FinishAsync(runKey, [new WorkStore.Finish(seq, "Cancelled", "Cancelled by user")], 'P'); + return outcome?.Applied > 0; + } + + /// Cancel one task that is still pending in storage, found by run name and task id. Reads the run's + /// pending range to find it (task ids are not keys), so prefer the run-key overload when the row is known. public async Task TryCancelQueuedTaskAsync(string runName, string taskId) { foreach (var header in await TargetRunsAsync(runName)) @@ -637,6 +651,83 @@ public void Cancelled(JobDescriptor descriptor) _ = _finisher.FinishAsync(runKey, new WorkStore.Finish(descriptor.Seq, "Cancelled", "Cancelled by user", Owner)); } + // ── diagnosis ── + + public sealed record ClaimView(string TaskId, int Seq, string? Owner, DateTimeOffset? LeaseUntil, int Attempt, bool HeldHere); + + public sealed record RunView(string RunKey, string Name, string Status, string Phase, int Priority, DateTime StartedUtc, + DateTime? CompletedUtc, string Mode, int Total, int Done, int Failed, int Cancelled, bool Listed, string Pending, + IReadOnlyList Running, IReadOnlyList WaitingOnChildren, string? PostExecStatus, string? Driver, + IReadOnlyList Diagnosis); + + public sealed record Inspection(string Query, string ThisProcess, string? LockHolder, DateTimeOffset? LockLeaseUntil, + IReadOnlyList Runs); + + /// + /// Everything needed to see why a run is (or is not) moving, from storage, in a few bounded reads per run: + /// its counts and mode, whether the scheduler can see it, its claims and who holds them, the child runs it + /// waits for, its aggregation, and the instance lock, plus a plain-language diagnosis. Looks a run up by key, + /// or every unfinished run of a name, or else the latest finished one. + /// + public async Task InspectRunAsync(string nameOrKey, CancellationToken ct = default) + { + var runs = await TargetRunsAsync(nameOrKey); + if (runs.Count == 0 && await _store.ResolveRunAsync(TableKeys.Sanitize(nameOrKey), ct) is { } byKey) runs = [byKey]; + if (runs.Count == 0 && await _store.GetRunByNameAsync(TableKeys.Sanitize(nameOrKey), ct) is { } latest) runs = [latest]; + var lockRow = await _store.GetInstanceLockAsync(ct); + var lockLive = lockRow != null && lockRow.LeaseUntil > DateTimeOffset.UtcNow; + + var views = new List(); + foreach (var h in runs) + { + var pending = await _store.GetTasksAsync(h.RunKey, 'P', 1001, ct); + var running = (await _store.GetTasksAsync(h.RunKey, 'R', 500, ct)) + .Select(t => new ClaimView(t.TaskId, t.Seq, t.Owner, t.LeaseUntil, t.Attempt, t.Owner == Owner)).ToList(); + var children = await _store.GetChildWaitsAsync(h.RunKey, ct: ct); + var listed = h.IsFinished || await _store.IsListedAsync(h, ct); + var tasksPending = pending.Count(t => t.Seq != WorkStore.AggregateSeq); + var pendingText = tasksPending > 1000 ? "1000+" : tasksPending.ToString(System.Globalization.CultureInfo.InvariantCulture); + var mode = h.Sequential ? (h.StopOnFailure ? "sequential, stop on failure" : "sequential") + : h.MaxConcurrency > 0 ? $"at most {h.MaxConcurrency} at once" : "fan-out"; + + var why = new List(); + if (h.IsFinished) + { + why.Add($"Finished {h.Status} at {h.CompletedUtc:O}."); + } + else + { + if (!listed) why.Add("Not on the Ready list, so the scheduler cannot see it. RepairIndexes (or a restart) relists it."); + if (!lockLive) why.Add("Nobody holds the instance lock: no process is claiming work."); + else if (lockRow!.Owner != Owner) why.Add($"The instance lock is held by {lockRow.Owner}, not this process; that process is the one claiming."); + var stale = running.Where(r => !r.HeldHere).ToList(); + var here = running.Count - stale.Count; + if (here > 0) why.Add($"{here} task(s) running in this process."); + if (stale.Count > 0) + why.Add(lockLive && lockRow!.Owner == Owner + ? $"{stale.Count} claim(s) held by a process that no longer works the queue ({string.Join(", ", stale.Select(s => s.Owner).Distinct())}); taken back the next time the run is read." + : $"{stale.Count} claim(s) held by {string.Join(", ", stale.Select(s => s.Owner).Distinct())}."); + if (children.Count > 0) why.Add($"Waiting for {children.Count} child run(s): {string.Join(", ", children.Select(c => c.Split('|')[0]))}."); + if (h.MaxConcurrency > 0 && !h.Sequential && tasksPending > 0 && here >= h.MaxConcurrency) + why.Add($"At its concurrency limit of {h.MaxConcurrency}; the next task starts when one finishes."); + else if (tasksPending > 0 && running.Count == 0) + why.Add($"{pendingText} task(s) pending, waiting for a worker in band P{h.Priority} (lower bands, and older runs in this band, go first)."); + if (h.Phase == RunPhase.Aggregate) + why.Add($"Every task is done; its aggregation (Push-{h.PostExecFunctionName}) is {(running.Any(r => r.Seq == WorkStore.AggregateSeq) ? "running" : "waiting to be claimed")}."); + if (h.CancelRequested) why.Add("Cancel requested: pending tasks are cancelled and running ones finish."); + } + + views.Add(new RunView(h.RunKey, h.Name, h.Status, h.Phase.ToString(), h.Priority, h.StartedUtc, h.CompletedUtc, mode, + h.Total, h.Done, h.Failed, h.Cancelled, listed, pendingText, running, children, h.PostExecStatus, + h.DriverOwner == null ? null : $"{h.DriverOwner} until {h.DriverLease:O}", why)); + } + return new Inspection(nameOrKey, Owner, lockRow?.Owner, lockRow?.LeaseUntil, views); + } + + /// Rebuild the Ready and Finished indexes from the active-run list now (the pump does this at startup). + public Task RepairIndexesAsync(CancellationToken ct = default) => + _store.RepairIndexesAsync(TimeSpan.FromMinutes(10), ct); + // ── lookups ── public string? GetRunReference(string runName) => Task.Run(async () => diff --git a/Services/Orchestration/WorkPump.cs b/Services/Orchestration/WorkPump.cs index 8ceb4d7..e171ac6 100644 --- a/Services/Orchestration/WorkPump.cs +++ b/Services/Orchestration/WorkPump.cs @@ -8,8 +8,10 @@ namespace Craft.Orchestration; /// oldest run first), claims from those runs' partitions until the batch is full, and hands the claims to the /// JobManager as descriptors. A run with nothing claimable is skipped for a while rather than read every tick. /// -/// The pump holds no state that matters after a crash: claims it held lapse and are claimed again, by this -/// process or any other. +/// One process works the queue: the pump claims nothing until it holds the instance lock (a single row, renewed +/// every few seconds and released on shutdown), so a recycle never has two processes claiming at once. While it +/// holds the lock, any claim in storage owned by another process belongs to one that has stopped, and is taken +/// back the first time its run is read. The pump holds no other state that matters after a crash. /// public class WorkPump : BackgroundService { @@ -39,7 +41,7 @@ public class WorkPump : BackgroundService /// 30 s, doubling each time it is found empty again, up to 15 minutes, until its counts move. /// private readonly Dictionary _skip = new(StringComparer.Ordinal); - private readonly System.Collections.Concurrent.ConcurrentQueue _released = new(); + private readonly System.Collections.Concurrent.ConcurrentQueue _changed = new(); private static readonly TimeSpan EmptyBackoff = TimeSpan.FromSeconds(30); private static readonly TimeSpan MaxEmptyBackoff = TimeSpan.FromMinutes(15); @@ -50,6 +52,18 @@ public class WorkPump : BackgroundService /// large backlog has to be scanned past. private const int ReadyPageSize = 1000; + /// The instance lock: how long it is held for without renewal, and how often it is renewed. + private readonly TimeSpan _lockLease; + private readonly TimeSpan _lockRenewEvery; + private DateTime _lockRenewedAt; + private volatile bool _holdsLock; + + /// Whether this pump holds the instance lock (and so may treat other owners' claims as dead). + internal bool HoldsLock => _holdsLock; + + /// How long a run may sit with its creation unfinished before the startup repair removes it. + private static readonly TimeSpan AbandonUnfinishedCreation = TimeSpan.FromMinutes(10); + /// Claims handed to the JobManager, by job id, with when their lease runs out. private readonly Dictionary _inFlight = new(StringComparer.Ordinal); @@ -61,26 +75,55 @@ public WorkPump(ILogger logger, WorkStore store, JobManager jobs, ICon _jobs = jobs; _orchestrator = orchestrator; _claimGate = orchestrator?.RecoveryDone; - _owner = orchestrator?.Owner ?? Environment.GetEnvironmentVariable("HOSTNAME") ?? $"instance-{Environment.ProcessId}"; + _owner = orchestrator?.Owner ?? OrchestratorService.NewOwnerId(); _batchSize = Math.Max(1, configuration.GetValue("JobQueueBatchSize", Math.Max(1, settings.Worker.BgPoolSize))); _lowWater = Math.Max(0, configuration.GetValue("JobQueueLowWaterMark", 2)); _lease = orchestrator?.Lease ?? TimeSpan.FromSeconds(Math.Max(60, configuration.GetValue("JobQueueLeaseSeconds", 1800))); _pollInterval = TimeSpan.FromMilliseconds(Math.Max(100, configuration.GetValue("JobQueuePollIntervalMs", 1000))); _idlePollInterval = TimeSpan.FromMilliseconds(Math.Max(_pollInterval.TotalMilliseconds, configuration.GetValue("JobQueueIdlePollIntervalMs", 10_000))); - _store.Released += _released.Enqueue; + _lockLease = TimeSpan.FromSeconds(Math.Max(5, configuration.GetValue("InstanceLockSeconds", 30))); + _lockRenewEvery = _lockLease / 3; + _store.RunChanged += runKey => + { + _changed.Enqueue(runKey); + Wake(); + }; + _jobs.Dispatched += () => + { + if (_jobs.QueuedCount <= _lowWater) Wake(); + }; + } + + /// + /// Refill now rather than at the next poll: the JobManager's buffer has drained to the low-water mark, or a + /// run was created or moved on in this process. Wakes are coalesced to one refill per + /// , so a burst of finishes does not re-read the Ready list for each one. + /// + private void Wake() + { + try + { + if (_wake.CurrentCount == 0) _wake.Release(); + } + catch (SemaphoreFullException) { } } + private readonly SemaphoreSlim _wake = new(0, 1); + private static readonly TimeSpan MinRefillGap = TimeSpan.FromMilliseconds(50); + protected override async Task ExecuteAsync(CancellationToken stoppingToken) { _logger.LogInformation("[WorkPump] Started: owner={Owner} batch={Batch} lowWater={Low} lease={Lease}s", _owner, _batchSize, _lowWater, _lease.TotalSeconds); - if (_claimGate is { IsCompleted: false }) + try { - try { await _claimGate.WaitAsync(stoppingToken); } - catch (OperationCanceledException) { return; } + if (_claimGate is { IsCompleted: false }) await _claimGate.WaitAsync(stoppingToken); + await AcquireLockAsync(stoppingToken); } + catch (OperationCanceledException) { return; } + _ = RepairAsync(stoppingToken); var idleTicks = 0; while (!stoppingToken.IsCancellationRequested) @@ -88,6 +131,7 @@ protected override async Task ExecuteAsync(CancellationToken stoppingToken) var claimed = 0; try { + if (!await KeepLockAsync(stoppingToken)) await AcquireLockAsync(stoppingToken); claimed = await RefillAsync(stoppingToken); await RenewAsync(stoppingToken); } @@ -102,13 +146,109 @@ protected override async Task ExecuteAsync(CancellationToken stoppingToken) ? _pollInterval : TimeSpan.FromMilliseconds(Math.Min(_idlePollInterval.TotalMilliseconds, _pollInterval.TotalMilliseconds * (1L << Math.Min(idleTicks, 20)))); - try { await Task.Delay(delay, stoppingToken); } + if (delay > _lockRenewEvery) delay = _lockRenewEvery; + try + { + if (await _wake.WaitAsync(delay, stoppingToken)) + { + idleTicks = 0; + await Task.Delay(MinRefillGap, stoppingToken); + } + } catch (OperationCanceledException) { break; } } } internal void ForgetBackoff() => _skip.Clear(); + /// + /// Wait until this process holds the instance lock. A predecessor that shut down cleanly released it, so this + /// is immediate after a normal recycle; one that crashed holds it until its lease runs out. + /// + internal async Task AcquireLockAsync(CancellationToken ct) + { + string? waitingOn = null; + while (true) + { + var (held, holder) = await _store.TryHoldInstanceLockAsync(_owner, _lockLease, ct); + if (held) + { + _holdsLock = true; + _lockRenewedAt = DateTime.UtcNow; + _logger.LogInformation("[WorkPump] Holding the instance lock as {Owner}; claims of any other process are taken back on sight", + _owner); + return; + } + if (holder?.Owner != waitingOn) + { + waitingOn = holder?.Owner; + _logger.LogInformation("[WorkPump] Waiting for the instance lock, held by {Holder} until {Until:O}", + holder?.Owner, holder?.LeaseUntil); + } + var wait = (holder?.LeaseUntil ?? DateTimeOffset.UtcNow) - DateTimeOffset.UtcNow; + await Task.Delay(wait < TimeSpan.FromSeconds(1) ? TimeSpan.FromSeconds(1) : wait > _lockRenewEvery ? _lockRenewEvery : wait, ct); + } + } + + /// Renew the instance lock when due. False when it was lost (a renewal failed long enough for + /// another process to take it): claiming stops until it is held again. + internal async Task KeepLockAsync(CancellationToken ct) + { + if (!_holdsLock) return false; + if (DateTime.UtcNow - _lockRenewedAt < _lockRenewEvery) return true; + try + { + var (held, holder) = await _store.TryHoldInstanceLockAsync(_owner, _lockLease, ct); + if (held) + { + _lockRenewedAt = DateTime.UtcNow; + return true; + } + _holdsLock = false; + _logger.LogCritical("[WorkPump] Lost the instance lock to {Holder}; claiming stops until it is held again", holder?.Owner); + return false; + } + catch (Exception ex) when (ex is not OperationCanceledException) + { + // A storage blip: keep claiming while the lease we hold is still good, and say so if it is not. + if (DateTime.UtcNow - _lockRenewedAt < _lockLease) return true; + _holdsLock = false; + _logger.LogCritical(ex, "[WorkPump] Could not renew the instance lock before it ran out; claiming stops until it is held again"); + return false; + } + } + + private int _repairing; + private int _indexFailuresSeen; + + /// + /// Rebuild the indexes from the active-run list: once when the lock is first held, and again whenever an index + /// write has failed for good since the last pass (so a run whose Ready entry never landed is relisted now, not + /// at the next restart). Runs beside claiming; one pass at a time. + /// + private async Task RepairAsync(CancellationToken ct) + { + if (Interlocked.Exchange(ref _repairing, 1) == 1) return; + _indexFailuresSeen = _store.IndexFailures; + try + { + var r = await _store.RepairIndexesAsync(AbandonUnfinishedCreation, ct); + _logger.LogInformation("[WorkPump] Index repair: {Active} active run(s), {Relisted} relisted, {Retired} retired, {Removed} unfinished creation(s) removed, {Young} still being created", + r.Active, r.Relisted, r.Retired, r.Removed, r.Young); + } + catch (Exception ex) when (ex is not OperationCanceledException) + { + _logger.LogError(ex, "[WorkPump] Index repair failed; it runs again at the next start, or via RepairIndexes"); + } + finally + { + Volatile.Write(ref _repairing, 0); + } + } + + /// The background repair started by the last refill that saw a failed index write; tests await it. + internal Task? LastRepair { get; private set; } + /// The pump's notion of now, for its backoff and renewal timing; tests replace it. internal Func Clock { get; set; } = () => DateTime.UtcNow; @@ -130,7 +270,8 @@ private void Forget() internal async Task RefillAsync(CancellationToken ct) { Forget(); - while (_released.TryDequeue(out var releasedRun)) _skip.Remove(releasedRun); + while (_changed.TryDequeue(out var changedRun)) _skip.Remove(changedRun); + if (_store.IndexFailures != _indexFailuresSeen && Volatile.Read(ref _repairing) == 0) LastRepair = RepairAsync(ct); if (_jobs.QueuedCount > _lowWater) return 0; var need = _batchSize - _jobs.QueuedCount; var claimed = 0; @@ -168,12 +309,12 @@ internal async Task RefillAsync(CancellationToken ct) var probe = new WorkStore.ClaimProbe(); if (header.Sequential) { - var step = await _store.ClaimSequentialAsync(header.RunKey, _owner, _lease, ct: ct); + var step = await _store.ClaimSequentialAsync(header.RunKey, _owner, _lease, othersAreDead: _holdsLock, ct: ct); claims = step == null ? [] : [step]; } else { - claims = await _store.ClaimAsync(header.RunKey, want, _owner, _lease, reclaimExpired: true, probe, ct); + claims = await _store.ClaimAsync(header.RunKey, want, _owner, _lease, reclaimExpired: true, probe, othersAreDead: _holdsLock, ct); } if (probe.PendingExhausted) @@ -235,5 +376,22 @@ public override async Task StopAsync(CancellationToken cancellationToken) try { await _store.ReleaseAsync(v.Claim.RunKey, v.Claim.Seq, _owner, refundAttempt: true, cancellationToken); } catch (Exception ex) { _logger.LogDebug(ex, "[WorkPump] Could not release {Job} on shutdown", id); } } + if (_holdsLock) + { + // Hold the lock while this process still runs claimed tasks: a successor holding it would take those + // claims back as dead and run them a second time. If shutdown is cut short, the lock simply lapses. + try + { + while (_jobs.GetJobs(status: "Running").Any(j => _inFlight.ContainsKey(j.Id))) + { + await KeepLockAsync(cancellationToken); + await Task.Delay(250, cancellationToken); + } + } + catch (OperationCanceledException) { return; } + try { await _store.ReleaseInstanceLockAsync(_owner, cancellationToken); } + catch (Exception ex) { _logger.LogDebug(ex, "[WorkPump] Could not release the instance lock on shutdown"); } + _holdsLock = false; + } } } diff --git a/Services/Storage/AzureTableStore.cs b/Services/Storage/AzureTableStore.cs index fca5a91..f9e0d3e 100644 --- a/Services/Storage/AzureTableStore.cs +++ b/Services/Storage/AzureTableStore.cs @@ -510,11 +510,12 @@ public async IAsyncEnumerable QueryTableAsync(string table, string? fi } public async IAsyncEnumerable QueryRowKeyRangeAsync(string table, string partitionKey, string fromRowKey, - string toRowKey, IReadOnlyList? properties = null, + string toRowKey, IReadOnlyList? properties = null, int? maxPerPage = null, [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) { var filter = $"PartitionKey eq '{Escape(partitionKey)}' and RowKey ge '{Escape(fromRowKey)}' and RowKey lt '{Escape(toRowKey)}'"; - await foreach (var entity in EnumerateAsync(table, () => Client(table).QueryAsync(filter: filter, select: properties, cancellationToken: ct), ct)) + await foreach (var entity in EnumerateAsync(table, () => Client(table).QueryAsync(filter: filter, maxPerPage: maxPerPage, + select: properties, cancellationToken: ct), ct)) yield return ToRow(entity); } diff --git a/Services/Storage/ICraftTableStore.cs b/Services/Storage/ICraftTableStore.cs index 19096d1..f74ac07 100644 --- a/Services/Storage/ICraftTableStore.cs +++ b/Services/Storage/ICraftTableStore.cs @@ -92,10 +92,12 @@ IAsyncEnumerable QueryTableAsync(string table, string? filter, int max /// /// Rows of one partition with <= RowKey < /// (ordinal), optionally projected (name the keys too if you read them). Split entities are not - /// reassembled, so use it only on tables whose rows are never split. + /// reassembled, so use it only on tables whose rows are never split. is + /// the page size asked of the service ($top); a caller that needs a few rows should pass it, or each + /// request returns up to 1,000. /// async IAsyncEnumerable QueryRowKeyRangeAsync(string table, string partitionKey, string fromRowKey, - string toRowKey, IReadOnlyList? properties = null, + string toRowKey, IReadOnlyList? properties = null, int? maxPerPage = null, [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) { await foreach (var row in QueryPartitionAsync(table, partitionKey, ct)) diff --git a/Services/Storage/WorkStore.cs b/Services/Storage/WorkStore.cs index 31f6fea..5f67d80 100644 --- a/Services/Storage/WorkStore.cs +++ b/Services/Storage/WorkStore.cs @@ -19,9 +19,12 @@ namespace Craft.Storage; /// whose lease lapses, and the next claim takes it back. /// /// Three small tables sit beside it: Ready (one row per run with work, ordered by band then start time, -/// read by the scheduler), Names (latest run per name, and every unfinished run by name, since runs of one -/// name may overlap) and Finished (completion order, for retention). -/// All three are hints derived from the Work rows: a stale one costs a read, never a wrong answer. +/// read by the scheduler), Names (latest run per name; every unfinished run, partition A; and the +/// instance lock) and Finished (completion order, for retention). +/// +/// The active-run row is written before anything else of a run and removed after everything else, so it is +/// the authoritative list of runs that exist; Ready and Finished are indexes derived from the runs and can be +/// rebuilt from it (). A stale index row costs a read, never a wrong answer. /// public sealed class WorkStore { @@ -90,12 +93,14 @@ public static string RunKeyFor(string name, DateTime startedUtc) => private static string ReadyPartition(int band) => "P" + Math.Clamp(band, 0, 99).ToString("D2", CultureInfo.InvariantCulture); private static string ReadyKey(RunHeader h) => $"{h.StartedUtc.Ticks.ToString("D19", CultureInfo.InvariantCulture)}|{h.RunKey}"; - /// A run's rows in one state, in seq order; 0 means all. + /// A run's rows in one state, in seq order; 0 means all. A bounded read asks + /// the service for only that many rows a page. private async IAsyncEnumerable Range(string runKey, char state, int max = 0, [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) { var count = 0; - await foreach (var row in _store.QueryRowKeyRangeAsync(_work, runKey, $"{state}|", $"{state}}}", null, ct)) + await foreach (var row in _store.QueryRowKeyRangeAsync(_work, runKey, $"{state}|", $"{state}}}", + maxPerPage: max > 0 ? Math.Min(max, 1000) : null, ct: ct)) { yield return row; if (max > 0 && ++count >= max) yield break; @@ -124,6 +129,11 @@ public async Task CreateRunAsync(RunHeader header, IReadOnlyList CreateRunAsync(RunHeader header, IReadOnlyList _store.UpsertAsync(_names, new StoreRow("N", header.Name) { Properties = { ["RunKey"] = header.RunKey } }, ct), ct); + await IndexAsync("list it as ready", header.RunKey, () => PublishReadyAsync(header, ct), ct); + Changed(header.RunKey); return (await GetRunAsync(header.RunKey, ct))!; } @@ -173,7 +184,7 @@ public async Task> GetActiveRunsAsync(string name, CancellationT { var runs = new List(); var stale = new List(); - await foreach (var row in _store.QueryRowKeyRangeAsync(_names, ActivePartition, $"{name}~", $"{name}~g", null, ct)) + await foreach (var row in _store.QueryRowKeyRangeAsync(_names, ActivePartition, $"{name}~", $"{name}~g", ct: ct)) { if (row.GetString("Name") != name) continue; if (await GetRunAsync(row.RowKey, ct) is { IsFinished: false } run) runs.Add(run); @@ -235,15 +246,20 @@ public sealed record TaskRow(int Seq, string TaskId, char State, string? Status, r.GetString("Status"), r.GetInt32("Attempt") ?? 0, r.GetString("Owner"), r.GetDateTimeOffset("LeaseUntil"), r.GetString("LastError")); - /// Every task row of a run (P, R and D), for status views and cancel lookups. - public async Task> GetTasksAsync(string runKey, char? state = null, CancellationToken ct = default) + /// A run's task rows (P, R and D, or one state), in seq order. bounds the + /// rows read per state; 0 reads them all, which on a large run is a long read, so status views pass a bound. + public async Task> GetTasksAsync(string runKey, char? state = null, int max = 0, CancellationToken ct = default) { var rows = new List(); foreach (var s in state is { } one ? [one] : new[] { 'P', 'R', 'D' }) - await foreach (var r in Range(runKey, s, ct: ct)) rows.Add(ToTask(r)); + await foreach (var r in Range(runKey, s, max, ct)) rows.Add(ToTask(r)); return rows; } + /// One pending task by its position, or null when it is not pending. A point read. + public async Task GetPendingAsync(string runKey, int seq, CancellationToken ct = default) => + await _store.GetAsync(_work, runKey, Key('P', seq), ct) is { } row ? ToTask(row) : null; + // ── claim ── public sealed record ClaimedTask(string RunKey, int Seq, string TaskId, int Attempt); @@ -256,9 +272,16 @@ public sealed class ClaimProbe public DateTimeOffset? EarliestLeaseUntil { get; internal set; } } - /// Raised with the run key when a claim is handed back to pending, so a scheduler that had written - /// the run off as drained looks at it again. - public event Action? Released; + /// Raised in-process with the run key whenever a run's work may have changed (created, a task + /// finished or was handed back, a child added, moved band), so a scheduler that had written the run off + /// looks at it again without depending on the Ready index write having landed. + public event Action? RunChanged; + + private void Changed(string runKey) + { + try { RunChanged?.Invoke(runKey); } + catch (Exception ex) { _logger.LogWarning(ex, "[WorkStore] A run-changed listener failed for {Run}", runKey); } + } /// /// Move up to pending tasks to running under , plus, with @@ -267,7 +290,7 @@ public sealed class ClaimProbe /// race returns empty and the caller moves on. /// public async Task> ClaimAsync(string runKey, int max, string owner, TimeSpan lease, - bool reclaimExpired, ClaimProbe? probe = null, CancellationToken ct = default) + bool reclaimExpired, ClaimProbe? probe = null, bool othersAreDead = false, CancellationToken ct = default) { max = Math.Min(max, MaxPerTransaction); if (max <= 0) return []; @@ -280,14 +303,17 @@ public async Task> ClaimAsync(string runKey, int max, // lease from a partial scan only makes the scheduler look again sooner. await foreach (var r in Range(runKey, 'R', MaxRunningScan, ct)) { - if (r.GetDateTimeOffset("LeaseUntil") is not { } until || until <= now) + var dead = r.GetDateTimeOffset("LeaseUntil") is not { } until || until <= now + || (othersAreDead && r.GetString("Owner") != owner); + if (dead) { if (reclaimExpired && expired.Count < max) expired.Add(r); else if (probe != null) probe.EarliestLeaseUntil = now; } - else if (probe != null && (probe.EarliestLeaseUntil is not { } seen || until < seen)) + else if (probe != null && r.GetDateTimeOffset("LeaseUntil") is { } live + && (probe.EarliestLeaseUntil is not { } seen || live < seen)) { - probe.EarliestLeaseUntil = until; + probe.EarliestLeaseUntil = live; } if (probe == null && expired.Count >= max) break; } @@ -343,14 +369,17 @@ public async Task> ClaimAsync(string runKey, int max, /// drivers; a lapsed driver's step is reclaimed with it. /// /// True for the driver claiming its own next step; false (the pump) defers to any live driver. + /// The caller holds the instance lock, so a driver lease held by any other process + /// belongs to a process that has stopped, and its step is taken over at once. public async Task ClaimSequentialAsync(string runKey, string owner, TimeSpan lease, bool continuing = false, - CancellationToken ct = default) + bool othersAreDead = false, CancellationToken ct = default) { var headerRow = await _store.GetAsync(_work, runKey, HeaderKey, ct); if (headerRow == null) return null; var header = RunHeader.FromRow(headerRow); var now = DateTimeOffset.UtcNow; - if (header.DriverOwner != null && header.DriverLease > now && !(continuing && header.DriverOwner == owner)) return null; + var driverLive = header.DriverOwner != null && header.DriverLease > now && !(othersAreDead && header.DriverOwner != owner); + if (driverLive && !(continuing && header.DriverOwner == owner)) return null; StoreRow? row = null; await foreach (var r in Range(runKey, 'R', ct: ct)) { row = r; break; } @@ -368,7 +397,7 @@ public async Task> ClaimAsync(string runKey, int max, await CancelPendingAsync(runKey, StoppedReason(row.GetString("TaskId")), ct); return null; } - return await ClaimSequentialAsync(runKey, owner, lease, continuing, ct); + return await ClaimSequentialAsync(runKey, owner, lease, continuing, othersAreDead, ct); } var until = now.Add(lease); @@ -421,7 +450,7 @@ public async Task ReleaseAsync(string runKey, int seq, string owner, bool await _rate.TakeAsync(runKey, 2, ct); var released = await _store.TrySubmitAsync(_work, runKey, [StoreOp.Delete(row), StoreOp.Insert(PendingRow(runKey, seq, row.GetString("TaskId")!, attempt))], ct); - if (released) Released?.Invoke(runKey); + if (released) Changed(runKey); return released; } @@ -489,8 +518,12 @@ public sealed record FinishOutcome(RunHeader Header, bool ReachedBarrier, bool C var barrier = false; var completed = false; + // An entity may appear once in a transaction: a task finished twice in one batch (a cancel and a + // completion landing together, or a retried finish) is applied once. + var seen = new HashSet(StringComparer.Ordinal); foreach (var f in chunk) { + if (!seen.Add(f.ChildKey is { } ck ? $"C|{ck}" : Seq(f.Seq))) continue; if (f.ChildKey is { } child) { var placeholder = await _store.GetAsync(_work, runKey, $"C|{child}", ct); @@ -563,7 +596,8 @@ public sealed record FinishOutcome(RunHeader Header, bool ReachedBarrier, bool C { var after = (await GetRunAsync(runKey, ct)) ?? header; if (completed) await RetireAsync(after, ct); - else await PublishReadyAsync(after, ct); + else await IndexAsync("update its Ready counts", runKey, () => PublishReadyAsync(after, ct), ct); + Changed(runKey); var outcome = new FinishOutcome(after, barrier, completed, applied); if ((barrier || completed) && AfterFinish is { } hook) { @@ -574,18 +608,63 @@ public sealed record FinishOutcome(RunHeader Header, bool ReachedBarrier, bool C } } - _logger.LogWarning("[WorkStore] Finishing {Count} task(s) of {Run} kept losing races; the next attempt retries", chunk.Count, runKey); - return null; + throw new InvalidOperationException($"Finishing {chunk.Count} task(s) of {runKey} kept losing races to other writers"); } /// Take a finished run off the Ready list and record it for retention. + /// Retire a finished run: the Finished row first (so retention will always find it), then off the + /// Ready list, then off the active list last. Each step is idempotent, so the repair can redo any of them. private async Task RetireAsync(RunHeader h, CancellationToken ct) { _rate.Forget(h.RunKey); - await _store.DeleteAsync(_ready, ReadyPartition(h.Priority), ReadyKey(h), ct); - await _store.DeleteAsync(_names, ActivePartition, h.RunKey, ct); + await IndexAsync("record it for retention", h.RunKey, () => RecordFinishedAsync(h, ct), ct); + await IndexAsync("take it off the Ready list", h.RunKey, () => _store.DeleteAsync(_ready, ReadyPartition(h.Priority), ReadyKey(h), ct), ct); + await IndexAsync("take it off the active list", h.RunKey, () => _store.DeleteAsync(_names, ActivePartition, h.RunKey, ct), ct); + } + + private Task RecordFinishedAsync(RunHeader h, CancellationToken ct) + { var done = (h.CompletedUtc ?? DateTime.UtcNow).Ticks.ToString("D19", CultureInfo.InvariantCulture); - await _store.UpsertAsync(_finished, new StoreRow("F", $"{done}|{h.RunKey}") { Properties = { ["RunKey"] = h.RunKey } }, ct); + return _store.UpsertAsync(_finished, new StoreRow("F", $"{done}|{h.RunKey}") { Properties = { ["RunKey"] = h.RunKey } }, ct); + } + + private static readonly TimeSpan[] IndexRetryDelays = + [TimeSpan.FromMilliseconds(200), TimeSpan.FromMilliseconds(500), TimeSpan.FromSeconds(1), TimeSpan.FromSeconds(2), TimeSpan.FromSeconds(4)]; + + /// Index writes that failed for good; a scheduler repairs the indexes when this moves. + public int IndexFailures => Volatile.Read(ref _indexFailures); + private int _indexFailures; + + /// Delays between index-write retries; tests shorten them. + internal TimeSpan[] IndexRetries { get; set; } = IndexRetryDelays; + + /// + /// An index write that follows a committed change to a run. Retried through a short storage blip; if it + /// still fails the change stands, the in-process event keeps this instance + /// scheduling correctly, and (run at startup, or on demand) puts the + /// index right. Logged as an error naming the run, so a run missing from a listing has an explanation. + /// + private async Task IndexAsync(string what, string runKey, Func write, CancellationToken ct) + { + for (var attempt = 0; ; attempt++) + { + try + { + await write(); + return; + } + catch (Exception ex) when (ex is not OperationCanceledException || !ct.IsCancellationRequested) + { + if (attempt >= IndexRetries.Length) + { + Interlocked.Increment(ref _indexFailures); + _logger.LogError(ex, "[WorkStore] Could not {What} for run {Run}; the index repair (startup, or RepairIndexes) fixes it", + what, runKey); + return; + } + await Task.Delay(IndexRetries[attempt], ct); + } + } } /// Remove a stale Ready entry (its run is gone or finished). @@ -612,7 +691,11 @@ public async Task AddChildAsync(string parentKey, string childKey, Cancell StoreOp.Insert(new StoreRow(parentKey, $"C|{childKey}") { Properties = { ["Child"] = childKey } }), StoreOp.Replace(header.ToRow(headerRow.ETag)), }; - if (await _store.TrySubmitAsync(_work, parentKey, ops, ct)) return true; + if (await _store.TrySubmitAsync(_work, parentKey, ops, ct)) + { + Changed(parentKey); + return true; + } } return false; } @@ -638,6 +721,121 @@ public async Task AddChildAsync(string parentKey, string childKey, Cancell } } + // ── the instance lock ── + + private const string LockPartition = "$instance", LockKey = "lock"; + + /// Who holds the instance lock, and until when (null when nobody does). + public sealed record InstanceLock(string Owner, DateTimeOffset LeaseUntil, DateTimeOffset AcquiredUtc); + + public async Task GetInstanceLockAsync(CancellationToken ct = default) + { + await InitializeAsync(ct); + var row = await _store.GetAsync(_names, LockPartition, LockKey, ct); + return row?.GetString("Owner") is { } owner + ? new InstanceLock(owner, row.GetDateTimeOffset("LeaseUntil") ?? DateTimeOffset.MinValue, + row.GetDateTimeOffset("AcquiredUtc") ?? DateTimeOffset.MinValue) + : null; + } + + /// + /// Take or keep the instance lock: the one row that says which process works the queue. Succeeds when + /// nobody holds it, its lease has run out, or already holds it (a renewal). Guarded + /// by the row's ETag, so two processes racing for it cannot both win. Returns the holder afterwards. + /// + public async Task<(bool Held, InstanceLock? Holder)> TryHoldInstanceLockAsync(string owner, TimeSpan lease, + CancellationToken ct = default) + { + await InitializeAsync(ct); + var now = DateTimeOffset.UtcNow; + var row = await _store.GetAsync(_names, LockPartition, LockKey, ct); + var holder = row?.GetString("Owner"); + var until = row?.GetDateTimeOffset("LeaseUntil") ?? DateTimeOffset.MinValue; + if (row != null && holder != owner && until > now) + return (false, new InstanceLock(holder!, until, row.GetDateTimeOffset("AcquiredUtc") ?? DateTimeOffset.MinValue)); + + var acquired = holder == owner ? row!.GetDateTimeOffset("AcquiredUtc") ?? now : now; + var next = new StoreRow(LockPartition, LockKey) + { + ETag = row?.ETag, + Properties = { ["Owner"] = owner, ["LeaseUntil"] = now.Add(lease), ["AcquiredUtc"] = acquired }, + }; + var ok = await _store.TrySubmitAsync(_names, LockPartition, [row == null ? StoreOp.Insert(next) : StoreOp.Replace(next)], ct); + return ok ? (true, new InstanceLock(owner, now.Add(lease), acquired)) : (false, await GetInstanceLockAsync(ct)); + } + + /// Give the instance lock up, if still holds it, so a successor starts at once. + public async Task ReleaseInstanceLockAsync(string owner, CancellationToken ct = default) + { + var row = await _store.GetAsync(_names, LockPartition, LockKey, ct); + if (row?.GetString("Owner") == owner) + await _store.TrySubmitAsync(_names, LockPartition, [StoreOp.Delete(row)], ct); + } + + // ── index repair ── + + /// What a repair pass found and did. + public sealed record RepairResult(int Active, int Relisted, int Retired, int Removed, int Young); + + /// + /// Rebuild the indexes from the active-run list, which is written before a run's other rows and removed + /// after them. For each run on it: a run that finished is retired (Finished row, off Ready, off the list); + /// a run still going gets its Ready entry rewritten; a run whose header never landed (its creation was + /// interrupted) is removed once it is older than . Every step is + /// idempotent, so this is safe while the queue is being worked. One point read per active run. + /// + public async Task RepairIndexesAsync(TimeSpan abandonAfter, CancellationToken ct = default) + { + await InitializeAsync(ct); + var rows = new List(); + await foreach (var row in _store.QueryPartitionAsync(_names, ActivePartition, ct)) rows.Add(row); + + int relisted = 0, retired = 0, removed = 0, young = 0; + var cutoff = DateTimeOffset.UtcNow - abandonAfter; + foreach (var row in rows) + { + var runKey = row.RowKey; + var header = await GetRunAsync(runKey, ct); + if (header == null) + { + if ((row.GetDateTimeOffset("StartedUtc") ?? DateTimeOffset.MinValue) > cutoff) { young++; continue; } + _logger.LogWarning("[WorkStore] Removing run {Run}: its creation never finished", runKey); + await _store.DeletePartitionAsync(_work, runKey, ct); + await _store.DeletePartitionAsync(_results, runKey, ct); + await _store.DeleteAsync(_names, ActivePartition, runKey, ct); + removed++; + } + else if (header.IsFinished) + { + await RecordFinishedAsync(header, ct); + await _store.DeleteAsync(_ready, ReadyPartition(header.Priority), ReadyKey(header), ct); + await _store.DeleteAsync(_names, ActivePartition, runKey, ct); + retired++; + } + else + { + await PublishReadyAsync(header, ct); + Changed(runKey); + relisted++; + } + } + return new RepairResult(rows.Count, relisted, retired, removed, young); + } + + // ── inspection ── + + /// Whether the run has its entry on the Ready list (the scheduler only sees runs that do). + public async Task IsListedAsync(RunHeader h, CancellationToken ct = default) => + await _store.GetAsync(_ready, ReadyPartition(h.Priority), ReadyKey(h), ct) != null; + + /// The run's child placeholders: runs it is waiting for (at most ). + public async Task> GetChildWaitsAsync(string runKey, int max = 1000, CancellationToken ct = default) + { + var children = new List(); + await foreach (var r in Range(runKey, 'C', max, ct)) children.Add(r.RowKey[2..]); + return children; + } + // ── retention ── /// Delete runs that finished before the retention cutoff: their Work and Results partitions and @@ -646,7 +844,7 @@ public async Task SweepFinishedAsync(TimeSpan retention, CancellationToken { var cutoff = (DateTime.UtcNow - retention).Ticks.ToString("D19", CultureInfo.InvariantCulture); var expired = new List(); - await foreach (var row in _store.QueryRowKeyRangeAsync(_finished, "F", "", cutoff, null, ct)) + await foreach (var row in _store.QueryRowKeyRangeAsync(_finished, "F", "", cutoff, ct: ct)) expired.Add(row); foreach (var row in expired) @@ -700,6 +898,7 @@ public async Task SetPriorityAsync(string runKey, int priority, Cancellati if (!ok || before == null) return false; await _store.DeleteAsync(_ready, ReadyPartition(before.Priority), ReadyKey(before), ct); if (await GetRunAsync(runKey, ct) is { IsFinished: false } after) await PublishReadyAsync(after, ct); + Changed(runKey); return true; } diff --git a/tests/Craft.Tests/CountingTableStore.cs b/tests/Craft.Tests/CountingTableStore.cs index 6a1d285..b977fe2 100644 --- a/tests/Craft.Tests/CountingTableStore.cs +++ b/tests/Craft.Tests/CountingTableStore.cs @@ -15,8 +15,11 @@ internal sealed class CountingTableStore(ICraftTableStore inner) : ICraftTableSt public sealed class Counts { public int PointReads, Queries, Rows, Pages, Submits, Upserts, BatchUpserts, Deletes; + + /// Range queries that did not ask for a page size, so each request may return 1,000 rows. + public int UnboundedRanges; public override string ToString() => - $"reads={PointReads} queries={Queries} rows={Rows} pages={Pages} submits={Submits} upserts={Upserts} batches={BatchUpserts} deletes={Deletes}"; + $"reads={PointReads} queries={Queries} rows={Rows} pages={Pages} submits={Submits} upserts={Upserts} batches={BatchUpserts} deletes={Deletes} unboundedRanges={UnboundedRanges}"; } private readonly ConcurrentDictionary _byTable = new(StringComparer.Ordinal); @@ -32,6 +35,7 @@ public Counts Total() { t.PointReads += c.PointReads; t.Queries += c.Queries; t.Rows += c.Rows; t.Pages += c.Pages; t.Submits += c.Submits; t.Upserts += c.Upserts; t.BatchUpserts += c.BatchUpserts; t.Deletes += c.Deletes; + t.UnboundedRanges += c.UnboundedRanges; } return t; } @@ -96,8 +100,12 @@ public IAsyncEnumerable QueryTableAsync(string table, string? filter, Count(table, inner.QueryTableAsync(table, filter, maxPerPage, ct), Math.Max(1, maxPerPage), ct); public IAsyncEnumerable QueryRowKeyRangeAsync(string table, string partitionKey, string fromRowKey, string toRowKey, - IReadOnlyList? properties = null, CancellationToken ct = default) => - Count(table, inner.QueryRowKeyRangeAsync(table, partitionKey, fromRowKey, toRowKey, properties, ct), 1000, ct); + IReadOnlyList? properties = null, int? maxPerPage = null, CancellationToken ct = default) + { + if (maxPerPage == null) Interlocked.Increment(ref For(table).UnboundedRanges); + return Count(table, inner.QueryRowKeyRangeAsync(table, partitionKey, fromRowKey, toRowKey, properties, maxPerPage, ct), + maxPerPage ?? 1000, ct); + } public Task TrySubmitAsync(string table, string partitionKey, IReadOnlyList ops, CancellationToken ct = default) { diff --git a/tests/Craft.Tests/FaultyTableStore.cs b/tests/Craft.Tests/FaultyTableStore.cs new file mode 100644 index 0000000..0b8f7ed --- /dev/null +++ b/tests/Craft.Tests/FaultyTableStore.cs @@ -0,0 +1,43 @@ +using Craft.Storage; + +namespace Craft.Tests; + +/// Fails chosen writes, so a test can stop a sequence of writes at an exact point. +internal sealed class FaultyTableStore(ICraftTableStore inner) : ICraftTableStore +{ + public Func? FailUpsert; // (table, pk, rk) => fail + public Func, bool>? FailSubmit; + public Func? FailDelete; // (table, rk) => fail + + public Task PingAsync(CancellationToken ct = default) => inner.PingAsync(ct); + public Task EnsureTableAsync(string table, CancellationToken ct = default) => inner.EnsureTableAsync(table, ct); + public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) => + FailUpsert?.Invoke(table, row.PartitionKey, row.RowKey) == true + ? throw new InvalidOperationException($"injected upsert failure on {table}") + : inner.UpsertAsync(table, row, ct); + public Task UpsertBatchAsync(string table, string partitionKey, IReadOnlyList rows, CancellationToken ct = default) => + inner.UpsertBatchAsync(table, partitionKey, rows, ct); + public Task TryReplaceBatchAsync(string table, string partitionKey, IReadOnlyList rows, CancellationToken ct = default) => + inner.TryReplaceBatchAsync(table, partitionKey, rows, ct); + public Task GetAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) => + inner.GetAsync(table, partitionKey, rowKey, ct); + public IAsyncEnumerable QueryPartitionAsync(string table, string partitionKey, CancellationToken ct = default) => + inner.QueryPartitionAsync(table, partitionKey, ct); + public IAsyncEnumerable QueryTableAsync(string table, CancellationToken ct = default) => inner.QueryTableAsync(table, ct); + public IAsyncEnumerable QueryTableAsync(string table, string? filter, int maxPerPage, CancellationToken ct = default) => + inner.QueryTableAsync(table, filter, maxPerPage, ct); + public IAsyncEnumerable QueryRowKeyRangeAsync(string table, string partitionKey, string fromRowKey, string toRowKey, + IReadOnlyList? properties = null, int? maxPerPage = null, CancellationToken ct = default) => + inner.QueryRowKeyRangeAsync(table, partitionKey, fromRowKey, toRowKey, properties, maxPerPage, ct); + public Task TrySubmitAsync(string table, string partitionKey, IReadOnlyList ops, CancellationToken ct = default) => + FailSubmit?.Invoke(table, ops) == true + ? throw new InvalidOperationException($"injected transaction failure on {table}") + : inner.TrySubmitAsync(table, partitionKey, ops, ct); + public Task DeleteTableAsync(string table, CancellationToken ct = default) => inner.DeleteTableAsync(table, ct); + public Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) => + FailDelete?.Invoke(table, rowKey) == true + ? throw new InvalidOperationException($"injected delete failure on {table}") + : inner.DeleteAsync(table, partitionKey, rowKey, ct); + public Task DeletePartitionAsync(string table, string partitionKey, CancellationToken ct = default) => + inner.DeletePartitionAsync(table, partitionKey, ct); +} diff --git a/tests/Craft.Tests/JobQueueStatusReaderTests.cs b/tests/Craft.Tests/JobQueueStatusReaderTests.cs index b5d1e0b..5a58f95 100644 --- a/tests/Craft.Tests/JobQueueStatusReaderTests.cs +++ b/tests/Craft.Tests/JobQueueStatusReaderTests.cs @@ -69,6 +69,35 @@ public async Task WorkThisProcessHolds_IsNotCountedAsWaiting() Assert.Equal(3, summary.QueuedDurable); } + /// + /// The snapshot is cached for seconds while this process keeps claiming. Merged views once added the live + /// local queue to the snapshot's own "unclaimed", which had already subtracted the claims held when it was + /// taken, so every claim made since was counted twice (seen live: 202 queued on a 200-task run). + /// + [Fact] + public async Task ClaimsMadeAfterTheSnapshot_AreNotCountedTwice() + { + var (reader, store, jobs) = New(); + var run = await Create(store, "Busy", 5, 0); + void Hold(IEnumerable claims) + { + foreach (var c in claims) jobs.Enqueue(new JobDescriptor("Busy", c.TaskId, 4) { RunKey = c.RunKey, Seq = c.Seq }, $"Busy-{c.TaskId}", $"{c.RunKey}|{c.Seq}"); + } + Hold(await store.ClaimAsync(run.RunKey, 2, "me", TimeSpan.FromMinutes(5), false)); + await reader.GetAsync(TimeSpan.FromMinutes(5)); // cached with 2 held + Hold(await store.ClaimAsync(run.RunKey, 2, "me", TimeSpan.FromMinutes(5), false)); + + var summary = await reader.GetSummaryAsync(); + Assert.Equal(1, summary.QueuedDurable); + Assert.Equal(5, summary.Queued + summary.Running); + + var busy = Assert.Single(await reader.GetRunSummariesAsync(), r => r.Name == "Busy"); + Assert.Equal(5, busy.Total); + Assert.True(busy.Queued + busy.Running + busy.Completed + busy.Failed <= busy.Total, + $"queued {busy.Queued} running {busy.Running} completed {busy.Completed} failed {busy.Failed} over total {busy.Total}"); + Assert.Equal(5, busy.Queued + busy.Running); + } + [Fact] public async Task RunsSharingAName_AreSummedUnderThatName() { diff --git a/tests/Craft.Tests/MemoryTableStore.cs b/tests/Craft.Tests/MemoryTableStore.cs index d9e8a23..27caba3 100644 --- a/tests/Craft.Tests/MemoryTableStore.cs +++ b/tests/Craft.Tests/MemoryTableStore.cs @@ -173,7 +173,7 @@ public async IAsyncEnumerable QueryPartitionAsync(string table, string } public async IAsyncEnumerable QueryRowKeyRangeAsync(string table, string partitionKey, string fromRowKey, - string toRowKey, IReadOnlyList? properties = null, + string toRowKey, IReadOnlyList? properties = null, int? maxPerPage = null, [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) { foreach (var r in Snapshot(table, t => t.Between((partitionKey, fromRowKey), (partitionKey, toRowKey)))) diff --git a/tests/Craft.Tests/OrchestrationAzuriteTests.cs b/tests/Craft.Tests/OrchestrationAzuriteTests.cs index 01b324e..d5f41d2 100644 --- a/tests/Craft.Tests/OrchestrationAzuriteTests.cs +++ b/tests/Craft.Tests/OrchestrationAzuriteTests.cs @@ -13,7 +13,7 @@ namespace Craft.Tests; [Collection(LargeAllocationSerialTests.Name)] public class OrchestrationAzuriteTests { - private static async Task TryCreateAsync(int poolSize = 4) + private static async Task TryCreateAsync(int poolSize = 4, Func? wrap = null) { var settings = new CraftSettings(); var connection = Environment.GetEnvironmentVariable("CRAFT_TEST_TABLE_CONNECTION"); @@ -32,7 +32,7 @@ public class OrchestrationAzuriteTests } var prefix = "azo" + Guid.NewGuid().ToString("N")[..10]; - return await OrchestrationHarness.CreateAsync(poolSize, tables, s => s.Orchestrator.TablePrefix = prefix); + return await OrchestrationHarness.CreateAsync(poolSize, wrap?.Invoke(tables) ?? tables, s => s.Orchestrator.TablePrefix = prefix); } private static string Batch(int n, string prefix) => OrchestrationHarness.Batch(n, prefix); @@ -143,4 +143,124 @@ public async Task AClaimLeftByADeadProcess_IsTakenBackAfterItsLease() Assert.True(await h.DriveUntilFinished("AzOrphan", 30_000)); Assert.All(await h.Store.GetTasksAsync(run.RunKey, 'D'), t => Assert.Equal(2, t.Attempt)); } + + // ── the hardening guarantees, on a real backend ── + + private static readonly TimeSpan Lease = TimeSpan.FromMinutes(30); + + private static Task CreateRunAsync(WorkStore s, string name, int tasks, string? postExec = null) + { + var started = DateTime.UtcNow; + return s.CreateRunAsync(new RunHeader + { + RunKey = WorkStore.RunKeyFor(name, started), + Name = name, + StartedUtc = started, + TaskScriptName = "Invoke-CraftTask", + PostExecFunctionName = postExec, + }, Enumerable.Range(0, tasks).Select(i => new WorkStore.NewTask($"t{i}", [])).ToList()); + } + + [Fact] + public async Task AFinishOfFortyNineTasksThatReachesTheBarrier_IsExactlyOneHundredEntities_AndIsAccepted() + { + await using var h = await TryCreateAsync(); + if (h == null) return; + var run = await CreateRunAsync(h.Store, "AzHundred", 49, "Agg"); + var claims = await h.Store.ClaimAsync(run.RunKey, 49, "w", Lease, false); + Assert.Equal(49, claims.Count); + + // 49 x (delete running + insert done) + the aggregation insert + the header = 100, the documented maximum. + var outcome = await h.Store.FinishAsync(run.RunKey, claims.Select(c => new WorkStore.Finish(c.Seq, "Completed", Owner: "w")).ToList()); + + Assert.True(outcome!.ReachedBarrier); + Assert.Equal(RunPhase.Aggregate, (await h.Store.GetRunAsync(run.RunKey))!.Phase); + } + + [Fact] + public async Task ARangeReadsPageSize_IsAPageNotALimit() + { + await using var h = await TryCreateAsync(); + if (h == null) return; + var run = await CreateRunAsync(h.Store, "AzPaged", 23); + + var rows = 0; + await foreach (var _ in h.Tables.QueryRowKeyRangeAsync($"{h.Prefix}Work", run.RunKey, "P|", "P}", maxPerPage: 5)) rows++; + + Assert.Equal(23, rows); + } + + [Fact] + public async Task TheInstanceLock_HoldsOnARealBackend() + { + await using var h = await TryCreateAsync(); + if (h == null) return; + + var race = await Task.WhenAll(h.Store.TryHoldInstanceLockAsync("racer-a", TimeSpan.FromSeconds(2)), + h.Store.TryHoldInstanceLockAsync("racer-b", TimeSpan.FromSeconds(2))); + Assert.Single(race, r => r.Held); + var winner = race.Single(r => r.Held).Holder!.Owner; + var loser = winner == "racer-a" ? "racer-b" : "racer-a"; + Assert.False((await h.Store.TryHoldInstanceLockAsync(loser, Lease)).Held); + + await Task.Delay(TimeSpan.FromSeconds(2.5)); + Assert.True((await h.Store.TryHoldInstanceLockAsync(loser, Lease)).Held); + await h.Store.ReleaseInstanceLockAsync(loser); + Assert.Null(await h.Store.GetInstanceLockAsync()); + } + + [Fact] + public async Task AStoppedProcesssClaims_AreTakenBackByTheLockHolder_OnARealBackend() + { + await using var h = await TryCreateAsync(); + if (h == null) return; + Assert.True(await h.Start("AzInherited", Batch(3, "i"))); + var run = (await h.Store.GetRunByNameAsync("AzInherited"))!; + Assert.Equal(3, (await h.Store.ClaimAsync(run.RunKey, 3, "old-host/7/aaaa", TimeSpan.FromDays(1), false)).Count); + + await h.Pump.AcquireLockAsync(CancellationToken.None); + Assert.True(await h.DriveUntilFinished("AzInherited", 30_000)); + Assert.All(await h.Store.GetTasksAsync(run.RunKey, 'D'), t => Assert.Equal(2, t.Attempt)); + } + + [Fact] + public async Task CrashesAtWriteBoundaries_AreRepaired_OnARealBackend() + { + FaultyTableStore? faulty = null; + await using var h = await TryCreateAsync(wrap: t => faulty = new FaultyTableStore(t)); + if (h == null) return; + h.Store.IndexRetries = [TimeSpan.Zero]; + + faulty!.FailUpsert = (table, _, rk) => table == $"{h.Prefix}Work" && rk == WorkStore.HeaderKey; + await Assert.ThrowsAsync(() => CreateRunAsync(h.Store, "AzHalfMade", 3)); + faulty.FailUpsert = null; + + var done = await CreateRunAsync(h.Store, "AzUnretired", 1); + var claim = await h.Store.ClaimAsync(done.RunKey, 1, "w", Lease, false); + faulty.FailDelete = (table, _) => table == $"{h.Prefix}Ready" || table == $"{h.Prefix}Names"; + await h.Store.FinishAsync(done.RunKey, [new WorkStore.Finish(claim[0].Seq, "Completed", Owner: "w")]); + faulty.FailDelete = null; + + var r = await h.Store.RepairIndexesAsync(TimeSpan.Zero); + + Assert.Equal((1, 1), (r.Removed, r.Retired)); + var active = 0; + await foreach (var _ in h.Tables.QueryPartitionAsync($"{h.Prefix}Names", "A")) active++; + Assert.Equal(0, active); + Assert.Equal(1, await h.Store.SweepFinishedAsync(TimeSpan.Zero)); + } + + [Fact] + public async Task ATaskFinishedTwiceInOneBatch_IsAppliedOnce_OnARealBackend() + { + await using var h = await TryCreateAsync(); + if (h == null) return; + var run = await CreateRunAsync(h.Store, "AzTwice", 2); + var claim = (await h.Store.ClaimAsync(run.RunKey, 1, "w", Lease, false))[0]; + + var outcome = await h.Store.FinishAsync(run.RunKey, + [new WorkStore.Finish(claim.Seq, "Completed", Owner: "w"), new WorkStore.Finish(claim.Seq, "Cancelled", Owner: "w")]); + + Assert.Equal(1, outcome!.Applied); + } } diff --git a/tests/Craft.Tests/OrchestrationHarness.cs b/tests/Craft.Tests/OrchestrationHarness.cs index e2e3de8..5a643ae 100644 --- a/tests/Craft.Tests/OrchestrationHarness.cs +++ b/tests/Craft.Tests/OrchestrationHarness.cs @@ -84,6 +84,7 @@ internal sealed class OrchestrationHarness : IAsyncDisposable public required JobManager Jobs { get; init; } public required ICraftTableStore Tables { get; init; } public required CapturingLogger Log { get; init; } + public required string Prefix { get; init; } public static async Task CreateAsync(int poolSize = 4, ICraftTableStore? tables = null, Action? configure = null) @@ -108,7 +109,16 @@ public static async Task CreateAsync(int poolSize = 4, ICr await svc.ResumeInterruptedRunsAsync(CancellationToken.None); var pump = new WorkPump(NullLogger.Instance, store, jobs, config, settings, svc); _ = Task.Run(() => jobs.StartAsync(CancellationToken.None)); - return new OrchestrationHarness { Svc = svc, Store = store, Pump = pump, Jobs = jobs, Tables = tables, Log = log }; + return new OrchestrationHarness + { + Svc = svc, + Store = store, + Pump = pump, + Jobs = jobs, + Tables = tables, + Log = log, + Prefix = settings.Orchestrator.TablePrefix, + }; } public static string Batch(int n, string prefix = "t") => diff --git a/tests/Craft.Tests/OrchestrationRecoveryTests.cs b/tests/Craft.Tests/OrchestrationRecoveryTests.cs new file mode 100644 index 0000000..de750c6 --- /dev/null +++ b/tests/Craft.Tests/OrchestrationRecoveryTests.cs @@ -0,0 +1,365 @@ +using Craft.Configuration; +using Craft.Orchestration; +using Craft.PowerShellHost; +using Craft.Storage; +using Microsoft.Extensions.Configuration; +using Microsoft.Extensions.Logging.Abstractions; + +namespace Craft.Tests; + +/// +/// The guarantees that replace the old recovery machinery, each driven to its failure point: one process works +/// the queue (the instance lock), a stopped process's claims are taken back on sight, a failed index write never +/// strands a run, a crash at any write boundary is repaired from the active-run list, a finish that cannot be +/// written is retried rather than lost, and explains a stuck run. +/// +public class OrchestrationRecoveryTests +{ + private static readonly TimeSpan Lease = TimeSpan.FromMinutes(30); + + private static WorkStore NewStore(ICraftTableStore tables) => + new(NullLogger.Instance, new CraftSettings(), tables) { IndexRetries = [TimeSpan.Zero] }; + + private static JobManager NewJobs() + { + var settings = new CraftSettings(); + settings.Worker.BgPoolSize = 4; + var config = new ConfigurationBuilder().AddInMemoryCollection([]).Build(); + var repo = new ScriptRepository(NullLogger.Instance, settings); + var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); + var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); + return new JobManager(NullLogger.Instance, settings, limiter); + } + + private static WorkPump NewPump(WorkStore store, int lockSeconds = 5) + { + var settings = new CraftSettings(); + settings.Worker.BgPoolSize = 4; + var config = new ConfigurationBuilder().AddInMemoryCollection(new Dictionary + { + ["InstanceLockSeconds"] = lockSeconds.ToString(System.Globalization.CultureInfo.InvariantCulture), + }).Build(); + return new WorkPump(NullLogger.Instance, store, NewJobs(), config, settings); + } + + private static Task CreateAsync(WorkStore s, string name, int tasks, bool sequential = false, string? postExec = null) + { + var started = DateTime.UtcNow; + return s.CreateRunAsync(new RunHeader + { + RunKey = WorkStore.RunKeyFor(name, started), + Name = name, + StartedUtc = started, + TaskScriptName = "Invoke-CraftTask", + Sequential = sequential, + PostExecFunctionName = postExec, + }, Enumerable.Range(0, tasks).Select(i => new WorkStore.NewTask($"t{i}", [])).ToList()); + } + + // ── the instance lock ── + + [Fact] + public async Task OneProcessHoldsTheLock_TheNextGetsItOnceReleased() + { + var s = NewStore(new MemoryTableStore()); + + Assert.True((await s.TryHoldInstanceLockAsync("a", Lease)).Held); + var (held, holder) = await s.TryHoldInstanceLockAsync("b", Lease); + Assert.False(held); + Assert.Equal("a", holder!.Owner); + Assert.True((await s.TryHoldInstanceLockAsync("a", Lease)).Held); // renewal + + await s.ReleaseInstanceLockAsync("b"); // not the holder: no effect + Assert.Equal("a", (await s.GetInstanceLockAsync())!.Owner); + await s.ReleaseInstanceLockAsync("a"); + Assert.True((await s.TryHoldInstanceLockAsync("b", Lease)).Held); + } + + [Fact] + public async Task ALapsedLock_IsTakenOver() + { + var s = NewStore(new MemoryTableStore()); + Assert.True((await s.TryHoldInstanceLockAsync("crashed", TimeSpan.FromMilliseconds(30))).Held); + await Task.Delay(60); + + Assert.True((await s.TryHoldInstanceLockAsync("successor", Lease)).Held); + } + + [Fact] + public async Task TwoProcessesRacingForAFreeLock_OnlyOneGetsIt() + { + var mem = new MemoryTableStore(); + var s = NewStore(mem); + await s.InitializeAsync(); + var raced = false; + (bool Held, WorkStore.InstanceLock? Holder) rival = default; + mem.BeforeSubmit = async () => + { + if (raced) return; + raced = true; + mem.BeforeSubmit = null; + rival = await s.TryHoldInstanceLockAsync("rival", Lease); + }; + + var mine = await s.TryHoldInstanceLockAsync("me", Lease); + + Assert.True(rival.Held); + Assert.False(mine.Held); + Assert.Equal("rival", (await s.GetInstanceLockAsync())!.Owner); + } + + [Fact] + public async Task APumpWaitsForTheLock_AndStartsOnceItsPredecessorShutsDown() + { + var s = NewStore(new MemoryTableStore()); + var first = NewPump(s); + var second = NewPump(s); + await first.AcquireLockAsync(CancellationToken.None); + + var waiting = second.AcquireLockAsync(CancellationToken.None); + await Task.Delay(300); + Assert.False(waiting.IsCompleted); + Assert.False(second.HoldsLock); + + await first.StopAsync(CancellationToken.None); + await waiting.WaitAsync(TimeSpan.FromSeconds(5)); + Assert.True(second.HoldsLock); + } + + [Fact] + public async Task APumpTakesTheLock_FromAPredecessorThatCrashed_OnceItsLeaseRunsOut() + { + var s = NewStore(new MemoryTableStore()); + await s.TryHoldInstanceLockAsync("crashed/1/abc", TimeSpan.FromSeconds(2)); + var pump = NewPump(s); + + var started = DateTime.UtcNow; + await pump.AcquireLockAsync(CancellationToken.None).WaitAsync(TimeSpan.FromSeconds(10)); + + Assert.True(DateTime.UtcNow - started >= TimeSpan.FromSeconds(1.5)); + Assert.True(pump.HoldsLock); + } + + [Fact] + public async Task APumpThatLosesTheLock_StopsClaiming() + { + var s = NewStore(new MemoryTableStore()); + var pump = NewPump(s, lockSeconds: 6); + await pump.AcquireLockAsync(CancellationToken.None); + await s.ReleaseInstanceLockAsync((await s.GetInstanceLockAsync())!.Owner); + Assert.True((await s.TryHoldInstanceLockAsync("usurper", Lease)).Held); + + await Task.Delay(TimeSpan.FromSeconds(2.1)); // past a renewal interval (lease / 3) + Assert.False(await pump.KeepLockAsync(CancellationToken.None)); + Assert.False(pump.HoldsLock); + } + + // ── a stopped process's claims ── + + [Fact] + public async Task HoldingTheLock_AStoppedProcesssLiveClaimsAreTakenBackAtOnce() + { + var s = NewStore(new MemoryTableStore()); + var run = await CreateAsync(s, "Inherited", 3); + Assert.Equal(3, (await s.ClaimAsync(run.RunKey, 3, "old-host/7/aaaa", Lease, false)).Count); + var pump = NewPump(s); + + Assert.Equal(0, await pump.RefillAsync(CancellationToken.None)); // not holding the lock: left alone + pump.ForgetBackoff(); + await pump.AcquireLockAsync(CancellationToken.None); + Assert.Equal(3, await pump.RefillAsync(CancellationToken.None)); + Assert.All(await s.GetTasksAsync(run.RunKey, 'R'), t => Assert.Equal(2, t.Attempt)); + } + + [Fact] + public async Task HoldingTheLock_AStoppedProcesssSequentialDriver_IsTakenOverAtOnce() + { + var s = NewStore(new MemoryTableStore()); + var run = await CreateAsync(s, "SeqInherited", 3, sequential: true); + Assert.NotNull(await s.ClaimSequentialAsync(run.RunKey, "old-host/7/aaaa", Lease)); + var pump = NewPump(s); + await pump.AcquireLockAsync(CancellationToken.None); + + Assert.Equal(1, await pump.RefillAsync(CancellationToken.None)); + var step = Assert.Single(await s.GetTasksAsync(run.RunKey, 'R')); + Assert.Equal((0, 2), (step.Seq, step.Attempt)); + } + + // ── index writes that fail ── + + [Fact] + public async Task AFailedReadyUpdate_DoesNotStrandARun_ThisProcessIsWorking() + { + var faulty = new FaultyTableStore(new MemoryTableStore()); + await using var h = await OrchestrationHarness.CreateAsync(tables: faulty); + h.Store.IndexRetries = [TimeSpan.Zero]; + Assert.True(await h.Start("Stranded", OrchestrationHarness.Batch(2, "s"), "Agg")); + faulty.FailUpsert = (table, _, _) => table == "OrchestratorReady"; // every later Ready update fails + + Assert.True(await h.DriveUntilFinished("Stranded")); + Assert.Single(h.Svc.PostExecs); + } + + [Fact] + public async Task ARunWhoseReadyEntryNeverLanded_IsRelistedByTheRepairTheFailureTriggers() + { + var faulty = new FaultyTableStore(new MemoryTableStore()); + await using var h = await OrchestrationHarness.CreateAsync(tables: faulty); + h.Store.IndexRetries = [TimeSpan.Zero]; + faulty.FailUpsert = (table, _, _) => table == "OrchestratorReady"; + Assert.True(await h.Start("Unlisted", OrchestrationHarness.Batch(2, "u"))); + faulty.FailUpsert = null; + + Assert.Empty(await h.ReadyNamesAsync()); + Assert.Contains((await h.Svc.InspectRunAsync("Unlisted")).Runs[0].Diagnosis, d => d.StartsWith("Not on the Ready list", StringComparison.Ordinal)); + + await h.Pump.RefillAsync(CancellationToken.None); // sees the failure, repairs + await h.Pump.LastRepair!; + Assert.True(await h.DriveUntilFinished("Unlisted")); + } + + // ── a crash at a write boundary, repaired from the active-run list ── + + [Fact] + public async Task ACreationThatDiedBeforeItsHeader_IsRemoved_OnceItIsOldEnough() + { + var faulty = new FaultyTableStore(new MemoryTableStore()); + var s = NewStore(faulty); + faulty.FailUpsert = (table, _, rk) => table == "OrchestratorWork" && rk == WorkStore.HeaderKey; + await Assert.ThrowsAsync(() => CreateAsync(s, "HalfMade", 3)); + faulty.FailUpsert = null; + var runKey = Assert.Single(await ActiveKeysAsync(faulty)); + + var young = await s.RepairIndexesAsync(TimeSpan.FromHours(1)); + Assert.Equal((1, 0), (young.Young, young.Removed)); + + var old = await s.RepairIndexesAsync(TimeSpan.Zero); + Assert.Equal(1, old.Removed); + Assert.Empty(await ActiveKeysAsync(faulty)); + Assert.Empty(await s.GetTasksAsync(runKey)); + } + + [Fact] + public async Task ARunThatFinishedButWasNotRetired_IsRetiredByTheRepair() + { + var faulty = new FaultyTableStore(new MemoryTableStore()); + var s = NewStore(faulty); + var run = await CreateAsync(s, "Unretired", 1); + var claim = await s.ClaimAsync(run.RunKey, 1, "w", Lease, false); + faulty.FailDelete = (table, _) => table is "OrchestratorReady" or "OrchestratorNames"; + await s.FinishAsync(run.RunKey, [new WorkStore.Finish(claim[0].Seq, "Completed", Owner: "w")]); + faulty.FailDelete = null; + Assert.Single(await ActiveKeysAsync(faulty)); + Assert.True(s.IndexFailures > 0); + + var r = await s.RepairIndexesAsync(TimeSpan.FromMinutes(10)); + + Assert.Equal(1, r.Retired); + Assert.Empty(await ActiveKeysAsync(faulty)); + Assert.Empty(await NamesAsync(s)); + Assert.Equal(1, await s.SweepFinishedAsync(TimeSpan.Zero)); // retention still finds it + } + + internal static async Task> ActiveKeysAsync(FaultyTableStore t) + { + var keys = new List(); + await foreach (var r in t.QueryPartitionAsync("OrchestratorNames", "A")) keys.Add(r.RowKey); + return keys; + } + + private static async Task> NamesAsync(WorkStore s) + { + var names = new List(); + await foreach (var e in s.ReadReadyAsync()) names.Add(e.Name); + return names; + } + + // ── finishes that cannot be written ── + + [Fact] + public async Task AFinishThatCannotBeWritten_IsRetried_AndTheTaskIsNotRunAgain() + { + var faulty = new FaultyTableStore(new MemoryTableStore()); + await using var h = await OrchestrationHarness.CreateAsync(tables: faulty); + var failures = 0; + faulty.FailSubmit = (table, ops) => table == "OrchestratorWork" && ops.Any(o => o.Row.RowKey.StartsWith("D|", StringComparison.Ordinal)) + && Interlocked.Increment(ref failures) <= 3; + Assert.True(await h.Start("Flaky", OrchestrationHarness.Batch(4, "f"))); + + Assert.True(await h.DriveUntilFinished("Flaky", 30_000)); + Assert.True(failures >= 3); + Assert.Equal(4, h.Svc.Started.Count); + } + + [Fact] + public async Task ATaskFinishedTwiceInOneBatch_IsAppliedOnce() + { + var s = NewStore(new MemoryTableStore()); + var run = await CreateAsync(s, "Twice", 2); + var claim = (await s.ClaimAsync(run.RunKey, 1, "w", Lease, false))[0]; + + var outcome = await s.FinishAsync(run.RunKey, + [new WorkStore.Finish(claim.Seq, "Completed", Owner: "w"), new WorkStore.Finish(claim.Seq, "Cancelled", Owner: "w")]); + + Assert.Equal(1, outcome!.Applied); + Assert.Equal("Completed", Assert.Single(await s.GetTasksAsync(run.RunKey, 'D')).Status); + } + + // ── explaining a stuck run ── + + [Fact] + public async Task Inspection_SaysWhyARunIsNotMoving() + { + await using var h = await OrchestrationHarness.CreateAsync(); + + Assert.True(await h.Start("Capped", OrchestrationHarness.Batch(5, "c"), maxConcurrency: 2)); + var capped = (await h.Store.GetRunByNameAsync("Capped"))!; + Assert.Equal(2, (await h.Store.ClaimAsync(capped.RunKey, 2, h.Svc.Owner, Lease, false)).Count); + var cappedView = (await h.Svc.InspectRunAsync("Capped")).Runs.Single(); + Assert.Equal(2, cappedView.Running.Count(r => r.HeldHere)); + Assert.Contains(cappedView.Diagnosis, d => d.StartsWith("At its concurrency limit of 2", StringComparison.Ordinal)); + Assert.Equal("3", cappedView.Pending); + + Assert.True(await h.Start("Orphaned", OrchestrationHarness.Batch(1, "o"))); + var orphaned = (await h.Store.GetRunByNameAsync("Orphaned"))!; + await h.Store.ClaimAsync(orphaned.RunKey, 1, "gone-host/9/beef", Lease, false); + await h.Store.TryHoldInstanceLockAsync(h.Svc.Owner, Lease); + Assert.Contains((await h.Svc.InspectRunAsync("Orphaned")).Runs.Single().Diagnosis, + d => d.Contains("held by a process that no longer works the queue (gone-host/9/beef)")); + + Assert.True(await h.Start("Parent", OrchestrationHarness.Batch(1, "p"))); + var parent = (await h.Store.GetRunByNameAsync("Parent"))!; + Assert.NotNull(h.Svc.RegisterPendingChild(parent.RunKey, "Kid")); + Assert.Contains((await h.Svc.InspectRunAsync("Parent")).Runs.Single().Diagnosis, d => d.StartsWith("Waiting for 1 child run(s): Kid", StringComparison.Ordinal)); + + Assert.True(await h.Start("Quick", OrchestrationHarness.Batch(1, "q"))); + Assert.True(await h.DriveUntilFinished("Quick")); + Assert.Contains((await h.Svc.InspectRunAsync("Quick")).Runs.Single().Diagnosis, d => d.StartsWith("Finished Completed", StringComparison.Ordinal)); + } + + // ── bounded reads ── + + [Fact] + public async Task OnAHugeRun_StatusClaimAndInspection_ReadOnlyWhatTheyNeed() + { + var count = new CountingTableStore(new MemoryTableStore()); + await using var h = await OrchestrationHarness.CreateAsync(tables: count); + Assert.True(await h.Start("Huge", OrchestrationHarness.Batch(20_000, "h"))); + var reader = new JobQueueStatusReader(NullLogger.Instance, h.Jobs, h.Store); + count.Reset(); + + await reader.GetAsync(); + Assert.InRange(count.For("OrchestratorWork").Rows, 0, JobQueueStatusReader.HeadRows); + Assert.Equal(0, count.For("OrchestratorWork").UnboundedRanges); + + count.Reset(); + var run = (await h.Store.GetRunByNameAsync("Huge"))!; + await h.Store.ClaimAsync(run.RunKey, 49, "w", Lease, reclaimExpired: true, new WorkStore.ClaimProbe()); + Assert.Equal(0, count.For("OrchestratorWork").UnboundedRanges); + + count.Reset(); + await h.Svc.InspectRunAsync("Huge"); + Assert.Equal(0, count.For("OrchestratorWork").UnboundedRanges); + Assert.InRange(count.For("OrchestratorWork").Rows, 0, 1001 + 500); + } +} diff --git a/tests/Craft.Tests/OrchestrationThroughputTests.cs b/tests/Craft.Tests/OrchestrationThroughputTests.cs new file mode 100644 index 0000000..f022a26 --- /dev/null +++ b/tests/Craft.Tests/OrchestrationThroughputTests.cs @@ -0,0 +1,73 @@ +using System.Diagnostics; + +namespace Craft.Tests; + +/// +/// The pump's real loop, not single refills: it must refill as soon as the JobManager's buffer drains and as soon +/// as a run appears, not on its poll. On the poll alone, short tasks ran at about one batch a second (pool-size +/// tasks per second) and an idle instance took up to the 10 s idle poll to notice a new run. +/// +public class OrchestrationThroughputTests +{ + private static async Task StartedAsync() + { + var h = await OrchestrationHarness.CreateAsync(poolSize: 4); + h.Svc.MarkRecoveryDone(); + await h.Pump.StartAsync(CancellationToken.None); + var deadline = Environment.TickCount64 + 10_000; + while (!h.Pump.HoldsLock && Environment.TickCount64 < deadline) await Task.Delay(20); + Assert.True(h.Pump.HoldsLock); + return h; + } + + private static async Task WaitUntil(Func> done, int timeoutMs) + { + var deadline = Environment.TickCount64 + timeoutMs; + while (Environment.TickCount64 < deadline) + { + if (await done()) return true; + await Task.Delay(10); + } + return await done(); + } + + [Fact] + public async Task ShortTasks_AreNotHeldToOneBatchASecond() + { + await using var h = await StartedAsync(); + try + { + var sw = Stopwatch.StartNew(); + Assert.True(await h.Start("Quick", OrchestrationHarness.Batch(200, "q"))); + + Assert.True(await WaitUntil(async () => await h.Store.GetRunByNameAsync("Quick") is { IsFinished: true }, 30_000)); + sw.Stop(); + // On the poll alone this took about 50 s (four tasks a second). + Assert.True(sw.Elapsed < TimeSpan.FromSeconds(10), $"200 short tasks took {sw.Elapsed.TotalSeconds:F1}s"); + } + finally + { + await h.Pump.StopAsync(CancellationToken.None); + } + } + + [Fact] + public async Task AnIdleInstance_ClaimsANewRunAtOnce() + { + await using var h = await StartedAsync(); + try + { + await Task.Delay(TimeSpan.FromSeconds(4)); // let the idle poll back off + var sw = Stopwatch.StartNew(); + Assert.True(await h.Start("Fresh", OrchestrationHarness.Batch(1, "f"))); + + Assert.True(await WaitUntil(() => Task.FromResult(h.Svc.Started.Contains("f0")), 15_000)); + sw.Stop(); + Assert.True(sw.Elapsed < TimeSpan.FromSeconds(1), $"the first task of a new run started after {sw.Elapsed.TotalMilliseconds:F0}ms"); + } + finally + { + await h.Pump.StopAsync(CancellationToken.None); + } + } +} diff --git a/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs b/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs index 85a3e21..6b79e9b 100644 --- a/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs +++ b/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs @@ -236,22 +236,25 @@ public async Task Wrapper_WithoutAParent_UsesTheDefaultBand_AndNoLineage() } [Fact] - public async Task Wrapper_WithoutCollisions_SkipsAndSaysSo_WhileARunOfThatNameIsQueued() + public async Task Wrapper_WithoutCollisions_SkipsAndSaysSo_WhileARunOfThatNameIsActive() { - OrchestratorBridge.QueueOrchestration("WrapBusy", "[]", 4); + // Against a real active run rather than a queued entry: the bridge queue is process-wide and any + // test's PostExecution drains it. + var (svc, store) = NewStoreBackedService(); + await CreateRunAsync(store, "WrapBusy"); + var previousService = s_serviceField.GetValue(null); try { + OrchestratorBridge.Initialize(svc); var (pending, result) = await RunWrapperAsync("WrapBusy", "@{ OrchestratorName = 'WrapBusy'; AllowCollision = $false; Batch = @(@{ FunctionName = 'X' }) }"); Assert.Equal("Craft-WrapBusy-Skipped", result); - Assert.NotNull(pending); // the first, queued directly above - Assert.True(string.IsNullOrEmpty(pending!.BatchFilePath)); - Assert.Null(TakePending("WrapBusy")); // and no second one + Assert.Null(pending); } finally { - TakePending("WrapBusy"); + s_serviceField.SetValue(null, previousService); } } diff --git a/tests/Craft.Tests/WorkStoreTests.cs b/tests/Craft.Tests/WorkStoreTests.cs index c6332a5..2fb5ea2 100644 --- a/tests/Craft.Tests/WorkStoreTests.cs +++ b/tests/Craft.Tests/WorkStoreTests.cs @@ -268,8 +268,8 @@ public async Task RetentionDeletesRunsFinishedBeforeTheCutoff() /// /// The header is rewritten in every finish transaction, and a transaction cannot split a row, so a - /// property over Azure's 64 KiB limit there would fail every finish of the run. Azurite does not enforce - /// the limit, so this is checked on the row itself. + /// property over Azure's 64 KiB limit there would fail every finish of the run. Checked on the row itself, + /// which catches it without needing a real backend or a run large enough to hit the limit. /// [Fact] public async Task TheHeaderStaysSmall_HoweverLargeThePostExecutionParameters() From edc5a87358a897367205a011b52e96a0cd0740b4 Mon Sep 17 00:00:00 2001 From: Zacgoose <107489668+Zacgoose@users.noreply.github.com> Date: Tue, 6 Oct 2026 02:04:58 +0800 Subject: [PATCH 14/24] test(e2e): add orchestration engine regression section - PerfApi probes: recording task/PostExecution, run starter, durable run reader, bridge and legacy-table endpoints - checks for PostExecution contract, failures, child gating and inheritance, collisions, cancel, sequential runs, priority order, status consistency, restart recovery and perf gates - MaxConcurrency/StopOnFailure checks gated on image capability (SKIP when absent) --- .../API/Modules/PerfApi/PerfApi.psd1 | 2 +- .../API/Modules/PerfApi/PerfApi.psm1 | 290 +++++++++ perf-harness/docker-compose.e2e-azure.yml | 6 + .../scripts/run-e2e-orchestration.ps1 | 602 ++++++++++++++++++ perf-harness/scripts/run-e2e.ps1 | 29 +- 5 files changed, 922 insertions(+), 7 deletions(-) create mode 100644 perf-harness/scripts/run-e2e-orchestration.ps1 diff --git a/perf-harness/api-harness/API/Modules/PerfApi/PerfApi.psd1 b/perf-harness/api-harness/API/Modules/PerfApi/PerfApi.psd1 index 28b8e31..3d33224 100644 --- a/perf-harness/api-harness/API/Modules/PerfApi/PerfApi.psd1 +++ b/perf-harness/api-harness/API/Modules/PerfApi/PerfApi.psd1 @@ -5,7 +5,7 @@ Author = 'CRAFT perf-harness' Description = 'Synthetic HTTP endpoints for load-testing CRAFT in http-only mode. Not for production.' PowerShellVersion = '7.2' - FunctionsToExport = @('Invoke-PerfPing', 'Invoke-PerfEcho', 'Invoke-PerfCpu', 'Invoke-PerfSleep', 'Invoke-PerfJson', 'Invoke-PerfFile', 'Invoke-PerfBgEnqueue', 'Push-PerfBg', 'Push-PerfBgLeaf', 'Invoke-PerfManyRuns', 'Push-PerfHold', 'Invoke-PerfThreads', 'Invoke-PerfTableOp', 'Push-PerfCheck', 'Invoke-PerfCheckCounts', 'Push-PerfSeq', 'Invoke-PerfSeqResult', 'Push-PerfSeqWorker', 'Invoke-PerfSeqWorkerEnqueue', 'Invoke-PerfSeqWorkerResult', 'Invoke-PerfThreadBreakdown', 'Invoke-PerfSeedRuns', 'Invoke-ListPerf', 'Invoke-PerfWhoami', 'Invoke-PerfTimerTick', 'Invoke-PerfTimerCount', 'Invoke-PerfPublish', 'Invoke-PerfAllocation', 'Invoke-PerfRuns') + FunctionsToExport = @('Invoke-PerfPing', 'Invoke-PerfEcho', 'Invoke-PerfCpu', 'Invoke-PerfSleep', 'Invoke-PerfJson', 'Invoke-PerfFile', 'Invoke-PerfBgEnqueue', 'Push-PerfBg', 'Push-PerfBgLeaf', 'Invoke-PerfManyRuns', 'Push-PerfHold', 'Invoke-PerfThreads', 'Invoke-PerfTableOp', 'Push-PerfCheck', 'Invoke-PerfCheckCounts', 'Push-PerfSeq', 'Invoke-PerfSeqResult', 'Push-PerfSeqWorker', 'Invoke-PerfSeqWorkerEnqueue', 'Invoke-PerfSeqWorkerResult', 'Invoke-PerfThreadBreakdown', 'Invoke-PerfSeedRuns', 'Invoke-ListPerf', 'Invoke-PerfWhoami', 'Invoke-PerfTimerTick', 'Invoke-PerfTimerCount', 'Invoke-PerfPublish', 'Invoke-PerfAllocation', 'Invoke-PerfRuns', 'Push-PerfE2E', 'Push-PerfE2EPost', 'Invoke-PerfE2EStart', 'Invoke-PerfE2EState', 'Invoke-PerfE2ERuns', 'Invoke-PerfE2EBridge', 'Invoke-PerfE2ELegacy') CmdletsToExport = @() VariablesToExport = @() AliasesToExport = @() diff --git a/perf-harness/api-harness/API/Modules/PerfApi/PerfApi.psm1 b/perf-harness/api-harness/API/Modules/PerfApi/PerfApi.psm1 index 8b3661f..2601ca0 100644 --- a/perf-harness/api-harness/API/Modules/PerfApi/PerfApi.psm1 +++ b/perf-harness/api-harness/API/Modules/PerfApi/PerfApi.psm1 @@ -562,3 +562,293 @@ function Invoke-PerfPublish { [Craft.Services.RealtimeBridge]::Publish($userId, $jobId, $mode, $data, "/perf/$jobId", "View job") return @{ StatusCode = 200; Body = @{ ok = $true; endpoint = 'PerfPublish'; jobId = $jobId; mode = $mode; userId = $userId } } } + +# -- Orchestration e2e probes (scripts/run-e2e-orchestration.ps1) ----------------------------------------- +# Each check gets its own namespace (ns). Tasks and PostExecutions record what they saw under that ns, in the +# shared cache by default or in the E2EProbe table (sink=table) when the record has to survive a restart. +# Records are hashtables: kind T (task: run label, idx, start/end ticks, worker, stamped priority/run), +# kind P (PostExecution: what Results and Parameters held) and kind Q (what a child enqueue returned). + +function Get-PerfE2ECache([string]$Ns) { [Craft.Services.PowerShellRunnerService]::GetSharedCache("E2E:$Ns") } + +function Get-PerfE2ETable { + $Tc = [Azure.Data.Tables.TableClient]::new($env:AzureWebJobsStorage, 'E2EProbe') + $Flags = [Craft.Services.PowerShellRunnerService]::GetSharedCache('E2EProbeInit') + if (-not $Flags['created']) { $Tc.CreateIfNotExists() | Out-Null; $Flags['created'] = $true } + $Tc +} + +function Write-PerfE2ERecord([string]$Ns, [string]$Sink, [string]$Key, [hashtable]$Fields) { + if ($Sink -eq 'table') { + $E = [Azure.Data.Tables.TableEntity]::new($Ns, $Key) + foreach ($K in $Fields.Keys) { $E[$K] = $Fields[$K] } + (Get-PerfE2ETable).UpsertEntity[Azure.Data.Tables.TableEntity]($E, [Azure.Data.Tables.TableUpdateMode]::Replace, [System.Threading.CancellationToken]::None) | Out-Null + } else { + (Get-PerfE2ECache $Ns)[$Key] = $Fields + } +} + +# Deterministic >64 KB payload; the harness recomputes it to compare length and hash. +function Get-PerfE2EBig([int]$Kb) { ('0123456789abcdef' * ($Kb * 64)) + 'END' } + +function Get-PerfE2EBatch([string]$Ns, [string]$Sink, [string]$Label, $Spec) { + $Count = [int]$Spec.tasks + @(for ($I = 0; $I -lt $Count; $I++) { + $T = @{ FunctionName = 'PerfE2E'; ns = $Ns; sink = $Sink; run = $Label; idx = $I; TenantFilter = "t$I" } + if ($Spec.holdms) { $T.holdms = [int]$Spec.holdms } + if ($Spec.task) { foreach ($K in $Spec.task.Keys) { $T[$K] = $Spec.task[$K] } } + if ($Spec.overrides -and $Spec.overrides["$I"]) { foreach ($K in $Spec.overrides["$I"].Keys) { $T[$K] = $Spec.overrides["$I"][$K] } } + $T + }) +} + +# Queue a child run from inside a task. via=start goes through Start-CraftOrchestrator; via=bridge calls +# QueueOrchestrationFromFile directly, so a 0-task batch or a collision-off skip reaches the bridge, which +# registers the child with its parent before the run is ever created. +function Start-PerfE2EChild($Item, $Ctx) { + $C = $Item.child + $Label = if ($C.label) { [string]$C.label } else { [string]$C.name } + $Batch = Get-PerfE2EBatch -Ns $Item.ns -Sink $Item.sink -Label $Label -Spec $C + if ([string]$C.via -eq 'bridge') { + $Path = Join-Path ([IO.Path]::GetTempPath()) "e2e-child-$([guid]::NewGuid().ToString('N')).jsonl" + [IO.File]::WriteAllLines($Path, [string[]]@(foreach ($B in $Batch) { ConvertTo-Json -InputObject $B -Compress -Depth 10 })) + $Parent = if ($Ctx.RunKey) { [string]$Ctx.RunKey } else { [string]$Ctx.RunName } + $Prio = if ($null -ne $C.priority) { [int]$C.priority } elseif ($null -ne $Ctx.Priority) { [int]$Ctx.Priority } else { 4 } + [Craft.Services.OrchestratorBridge]::QueueOrchestrationFromFile([string]$C.name, $Path, $Prio, $null, $null, $null, + $Parent, $false, ($C.allowCollision -ne $false)) + $Result = 'bridge' + } else { + $In = @{ OrchestratorName = [string]$C.name; Batch = $Batch } + foreach ($K in 'priority', 'sequential', 'allowCollision') { if ($C.ContainsKey($K)) { $In[$K] = $C[$K] } } + $Result = [string](Start-CraftOrchestrator -InputObject $In -WarningAction SilentlyContinue) + } + Write-PerfE2ERecord $Item.ns $Item.sink "Q|$Label|$([guid]::NewGuid().ToString('N').Substring(0, 8))" @{ + kind = 'Q'; run = $Label; parent = [string]$Item.run; result = $Result; ticks = [DateTime]::UtcNow.Ticks } +} + +# The task. Records start (and, unless it dies, end), optionally queues a child, holds, emits output, or throws. +function Push-PerfE2E { + param($Item) + $Start = [DateTime]::UtcNow.Ticks + $Ctx = Get-Variable -Name 'CraftOperationContext' -Scope Global -ValueOnly -ErrorAction SilentlyContinue + $Key = "T|$($Item.run)|$($Item.idx)|$([guid]::NewGuid().ToString('N').Substring(0, 8))" + $Rec = @{ + kind = 'T'; run = [string]$Item.run; idx = [int]$Item.idx; start = $Start; end = [long]0 + worker = [string]$Ctx.WorkerId; runName = [string]$Ctx.RunName; runKey = [string]$Ctx.RunKey + prio = $(if ($null -ne $Ctx.Priority) { [int]$Ctx.Priority } else { -1 }) + } + Write-PerfE2ERecord $Item.ns $Item.sink $Key $Rec + if ($Item.child) { Start-PerfE2EChild -Item $Item -Ctx $Ctx } + if ($Item.holdms -and [int]$Item.holdms -gt 0) { Start-Sleep -Milliseconds ([int]$Item.holdms) } + $Done = $Rec.Clone() + $Done['end'] = [DateTime]::UtcNow.Ticks + if ($Item.fail) { + $Done['failed'] = $true + Write-PerfE2ERecord $Item.ns $Item.sink $Key $Done + throw "e2e: task $($Item.run)/$($Item.idx) failed on purpose" + } + Write-PerfE2ERecord $Item.ns $Item.sink $Key $Done + $Out = if ($Item.outKb) { 'o' * ([int]$Item.outKb * 1024) } else { [string]$Item.out } + return @{ run = [string]$Item.run; idx = [int]$Item.idx; out = $Out } +} + +# The PostExecution. Records what arrived: the raw line count, each entry's run/idx and output length, and +# the Parameters (big payload length + hash, marker, nested object). followOn queues a run from here. +function Push-PerfE2EPost { + param($Item) + $Ticks = [DateTime]::UtcNow.Ticks + $P = $Item.Parameters + $Entries = [System.Collections.Generic.List[object]]::new() + foreach ($R in @($Item.Results)) { + if ($R -is [System.Collections.IList]) { foreach ($X in $R) { $Entries.Add($X) } } else { $Entries.Add($R) } + } + $Idxs = @(foreach ($E in $Entries) { if ($E -is [System.Collections.IDictionary]) { "$($E['run'])/$($E['idx'])" } else { "?$E" } }) + $Lens = @(foreach ($E in $Entries) { if ($E -is [System.Collections.IDictionary]) { ([string]$E['out']).Length } }) + $Big = [string]$P.big + $Sha = if ($Big) { [Convert]::ToHexString([Security.Cryptography.SHA256]::HashData([Text.Encoding]::UTF8.GetBytes($Big))) } else { '' } + $First = @($Item.Results)[0] + Write-PerfE2ERecord ([string]$P.ns) ([string]$P.sink) "P|$($P.run)|$([guid]::NewGuid().ToString('N').Substring(0, 8))" @{ + kind = 'P'; run = [string]$P.run; ticks = $Ticks; lines = @($Item.Results).Count; entries = $Entries.Count + idxs = ($Idxs -join ','); outLens = ($Lens -join ','); bigLen = $Big.Length; bigSha = $Sha + marker = [string]$P.marker; nested = $(if ($P.nested) { ConvertTo-Json -InputObject $P.nested -Compress -Depth 5 } else { '' }) + firstType = $(if ($null -ne $First) { $First.GetType().Name } else { '' }) + } + if ($P.followOn) { + $F = $P.followOn + $Batch = @(for ($I = 0; $I -lt [int]$F.tasks; $I++) { + @{ FunctionName = 'PerfE2E'; ns = [string]$P.ns; sink = [string]$P.sink; run = [string]$F.label; idx = $I; holdms = [int]$F.holdms; TenantFilter = "t$I" } }) + Start-CraftOrchestrator -InputObject @{ OrchestratorName = [string]$F.name; Batch = $Batch } | Out-Null + } +} + +# POST { ns, sink, runs: [ { name, label, tasks, task{}, overrides{idx:{}}, post{marker,bigKb,nested,followOn}, +# Priority, Sequential, AllowCollision, MaxConcurrency, StopOnFailure } ] }. Runs are queued in order in one +# invocation. Returns each Start-CraftOrchestrator result, its warnings, and the server tick it was called at. +function Invoke-PerfE2EStart { + param($Request, $TriggerMetadata) + $Spec = $Request.Body | ConvertTo-Json -Depth 30 -Compress | ConvertFrom-Json -AsHashtable + $Ns = [string]$Spec.ns + $Sink = [string]$Spec.sink + $Out = [System.Collections.Generic.List[object]]::new() + foreach ($R in @($Spec.runs)) { + $Label = if ($R.label) { [string]$R.label } else { [string]$R.name } + $In = @{ OrchestratorName = [string]$R.name; Batch = (Get-PerfE2EBatch -Ns $Ns -Sink $Sink -Label $Label -Spec $R) } + foreach ($K in 'Priority', 'Sequential', 'AllowCollision', 'MaxConcurrency', 'StopOnFailure', 'Reference') { + if ($R.ContainsKey($K)) { $In[$K] = $R[$K] } + } + if ($R.post) { + $Params = @{ ns = $Ns; sink = $Sink; run = $Label; marker = [string]$R.post.marker } + if ($R.post.nested) { $Params.nested = $R.post.nested } + if ($R.post.followOn) { $Params.followOn = $R.post.followOn } + if ($R.post.bigKb) { $Params.big = Get-PerfE2EBig ([int]$R.post.bigKb) } + $In.PostExecution = @{ FunctionName = 'PerfE2EPost'; Parameters = $Params } + } + $Ticks = [DateTime]::UtcNow.Ticks + $Warn = $null + $Res = Start-CraftOrchestrator -InputObject $In -WarningVariable Warn -WarningAction SilentlyContinue + $Out.Add(@{ name = [string]$R.name; label = $Label; result = [string]$Res; warning = (@($Warn) -join ' | '); enqueueTicks = $Ticks }) + } + return @{ StatusCode = 200; Body = @{ ok = $true; ns = $Ns; runs = @($Out) } } +} + +# GET ?ns=X[&source=table][&summary=1][&wipe=1] -> every record written under the namespace. +function Invoke-PerfE2EState { + param($Request, $TriggerMetadata) + $Ns = [string]$Request.Query.ns + $Rows = [System.Collections.Generic.List[object]]::new() + if ([string]$Request.Query.source -eq 'table') { + foreach ($E in (Get-PerfE2ETable).Query[Azure.Data.Tables.TableEntity]("PartitionKey eq '$Ns'")) { + $H = @{} + foreach ($K in $E.Keys) { if ($K -notin 'odata.etag', 'PartitionKey', 'RowKey', 'Timestamp') { $H[$K] = $E[$K] } } + $Rows.Add($H) + } + } else { + $C = Get-PerfE2ECache $Ns + [System.Threading.Monitor]::Enter($C.SyncRoot) + try { foreach ($V in $C.Values) { $Rows.Add($V) } } finally { [System.Threading.Monitor]::Exit($C.SyncRoot) } + if ($Request.Query['wipe']) { $C.Clear() } + } + if ($Request.Query['summary']) { + $By = @{} + foreach ($R in $Rows) { + $S = $By[$R.run] + if (-not $S) { $S = @{ run = $R.run; tasks = 0; ended = 0; posts = 0; minStart = [long]0; maxEnd = [long]0 }; $By[$R.run] = $S } + if ($R.kind -eq 'P') { $S.posts = $S.posts + 1; continue } + if ($R.kind -ne 'T') { continue } + $S.tasks = $S.tasks + 1 + if ($S.minStart -eq 0 -or $R.start -lt $S.minStart) { $S.minStart = $R.start } + if ($R.end -gt 0) { $S.ended = $S.ended + 1; if ($R.end -gt $S.maxEnd) { $S.maxEnd = $R.end } } + } + return @{ StatusCode = 200; Body = @{ ok = $true; ns = $Ns; count = $Rows.Count; runs = @($By.Values) } } + } + return @{ StatusCode = 200; Body = @{ ok = $true; ns = $Ns; count = $Rows.Count; rows = @($Rows) } } +} + +# GET ?name=X | ?prefix=X [&detail=1] -> run headers read straight from {TablePrefix}Work (the durable state); +# detail adds each task's D row (status, attempt, error) and the count of P/R/C rows still open. +function Invoke-PerfE2ERuns { + param($Request, $TriggerMetadata) + $Name = [string]$Request.Query.name + $Prefix = if ($Name) { "$Name~" } else { [string]$Request.Query.prefix } + $Upper = $Prefix.Substring(0, $Prefix.Length - 1) + [char]([int]$Prefix[-1] + 1) + $Tp = if ($env:App__Orchestrator__TablePrefix) { $env:App__Orchestrator__TablePrefix } else { 'E2EOrch' } + $Tc = [Azure.Data.Tables.TableClient]::new($env:AzureWebJobsStorage, "${Tp}Work") + $Runs = [System.Collections.Generic.List[object]]::new() + try { + foreach ($E in $Tc.Query[Azure.Data.Tables.TableEntity]("PartitionKey ge '$Prefix' and PartitionKey lt '$Upper' and RowKey eq '`$run'")) { + if ($Name -and [string]$E['Name'] -ne $Name) { continue } + $H = @{ + runKey = $E.PartitionKey; name = [string]$E['Name']; status = [string]$E['Status']; phase = [string]$E['Phase'] + priority = $E['Priority']; total = $E['Total']; done = $E['Done']; failed = $E['Failed']; cancelled = $E['Cancelled'] + postExecStatus = [string]$E['PostExecStatus']; sequential = $E['Sequential']; parentRunKey = [string]$E['ParentRunKey'] + startedTicks = $(if ($E['StartedUtc']) { ([DateTimeOffset]$E['StartedUtc']).UtcTicks } else { 0 }) + completedTicks = $(if ($E['CompletedUtc']) { ([DateTimeOffset]$E['CompletedUtc']).UtcTicks } else { 0 }) + } + if ([string]$Request.Query.detail -eq '1') { + $Tasks = [System.Collections.Generic.List[object]]::new() + $Open = @{ P = 0; R = 0; C = 0 } + foreach ($T in $Tc.Query[Azure.Data.Tables.TableEntity]("PartitionKey eq '$($E.PartitionKey)' and RowKey ge 'C|' and RowKey lt 'S'")) { + $State = $T.RowKey.Substring(0, 1) + if ($State -eq 'D') { + $Tasks.Add(@{ seq = [int]$T.RowKey.Substring(2); taskId = [string]$T['TaskId']; status = [string]$T['Status'] + attempt = $T['Attempt']; error = [string]$T['LastError'] }) + } elseif ($Open.ContainsKey($State)) { $Open[$State] = $Open[$State] + 1 } + } + $H.tasks = @($Tasks | Sort-Object { $_.seq }) + $H.open = $Open + } + $Runs.Add($H) + } + } catch { + return @{ StatusCode = 500; Body = @{ ok = $false; error = "$_" } } + } + return @{ StatusCode = 200; Body = @{ ok = $true; count = $Runs.Count; runs = @($Runs | Sort-Object { $_.startedTicks }) } } +} + +# GET ?op=caps|active|cancel|summaries|summary|jobs [&name=X][&status=S] -> the orchestration bridges. +function Invoke-PerfE2EBridge { + param($Request, $TriggerMetadata) + $Name = [string]$Request.Query.name + try { + switch ([string]$Request.Query.op) { + 'caps' { + $Longest = [Craft.Services.OrchestratorBridge].GetMethods() | Where-Object Name -eq 'QueueOrchestrationFromFile' | + Sort-Object { $_.GetParameters().Count } | Select-Object -Last 1 + $Names = @($Longest.GetParameters() | ForEach-Object Name) + $Def = (Get-Command Start-CraftOrchestrator).Definition + $Body = @{ queueFromFileParams = $Names.Count; paramNames = $Names + startHasMaxConcurrency = ($Def -match 'MaxConcurrency'); startHasStopOnFailure = ($Def -match 'StopOnFailure') } + } + 'active' { $Body = @{ active = [Craft.Services.OrchestratorBridge]::IsRunActive($Name) } } + 'cancel' { $Body = @{ cancelled = [Craft.Services.WorkerMetricsBridge]::CancelRun($Name) } } + 'summaries' { + $Body = @{ runs = @([Craft.Services.WorkerMetricsBridge]::GetRunSummaries() | Where-Object { -not $Name -or $_.Name -eq $Name } | ForEach-Object { + @{ name = $_.Name; priority = $_.Priority; total = $_.Total; queued = $_.Queued; running = $_.Running + completed = $_.Completed; failed = $_.Failed; completedUtc = $_.CompletedUtc } }) } + } + 'summary' { + $S = [Craft.Services.WorkerMetricsBridge]::GetSummary() + $M = [Craft.Services.WorkerMetricsBridge]::GetSnapshot().Memory + $Body = @{ jobsQueued = $S.JobsQueued; jobsQueuedLocal = $S.JobsQueuedLocal; jobsQueuedDurable = $S.JobsQueuedDurable + jobsActive = $S.JobsActive; bgBusy = $S.BgBusy; bgPoolSize = $S.BgPoolSize; limiterMax = $S.LimiterMax + heapMB = $M.HeapMB; rssMB = $M.RssMB; committedMB = $M.CommittedMB + workingSetMB = [math]::Round([System.Diagnostics.Process]::GetCurrentProcess().WorkingSet64 / 1MB, 1) } + } + 'jobs' { + $St = if ($Request.Query.status) { [string]$Request.Query.status } else { $null } + $Rn = if ($Name) { $Name } else { $null } + $Body = @{ jobs = @([Craft.Services.WorkerMetricsBridge]::GetJobDetails($Rn, $St, 500) | ForEach-Object { + @{ id = $_.Id; runName = $_.RunName; priority = $_.Priority; status = $_.Status } }) } + } + default { return @{ StatusCode = 400; Body = @{ ok = $false; error = 'unknown op' } } } + } + } catch { + return @{ StatusCode = 500; Body = @{ ok = $false; error = "$_" } } + } + $Body.ok = $true + return @{ StatusCode = 200; Body = $Body } +} + +# GET ?op=seed|list -> create (one row each) or list the previous orchestration design's tables, which the +# engine must drop at startup. +function Invoke-PerfE2ELegacy { + param($Request, $TriggerMetadata) + $Tp = if ($env:App__Orchestrator__TablePrefix) { $env:App__Orchestrator__TablePrefix } else { 'E2EOrch' } + $Legacy = @('Queue', 'QueueIndex', 'Tasks', 'Runs', 'Results' | ForEach-Object { "$Tp$_" }) + try { + $Svc = [Azure.Data.Tables.TableServiceClient]::new($env:AzureWebJobsStorage) + if ([string]$Request.Query.op -eq 'seed') { + foreach ($T in $Legacy) { + $Tc = $Svc.GetTableClient($T) + $Tc.CreateIfNotExists() | Out-Null + $E = [Azure.Data.Tables.TableEntity]::new('seed', 'row1') + $E['Note'] = 'legacy row seeded by the e2e' + $Tc.UpsertEntity[Azure.Data.Tables.TableEntity]($E, [Azure.Data.Tables.TableUpdateMode]::Replace, [System.Threading.CancellationToken]::None) | Out-Null + } + } + $Present = @($Svc.Query() | ForEach-Object Name | Where-Object { $_ -in $Legacy }) + return @{ StatusCode = 200; Body = @{ ok = $true; legacy = $Legacy; present = $Present } } + } catch { + return @{ StatusCode = 500; Body = @{ ok = $false; error = "$_" } } + } +} diff --git a/perf-harness/docker-compose.e2e-azure.yml b/perf-harness/docker-compose.e2e-azure.yml index f1b6770..abdd232 100644 --- a/perf-harness/docker-compose.e2e-azure.yml +++ b/perf-harness/docker-compose.e2e-azure.yml @@ -40,6 +40,12 @@ services: # Realtime SSE is opt-in (off by default) — turn it on so the realtime bridge test can run. - CRAFT_REALTIME_ENABLED=true - App__Orchestrator__TablePrefix=E2EOrch + # Orchestration checks: a 60s claim lease (the minimum) so the restart check reclaims interrupted tasks in + # minutes, and a fixed BG concurrency of the full pool (no ramp-up, no HTTP-pressure throttle from the + # harness's own polling) so concurrency and ordering assertions are deterministic. + - JobQueueLeaseSeconds=60 + - BackgroundBaseConcurrency=4 + - BackgroundHttpPressureThreshold=0 - App__RateLimit__Enabled=false # Scheduler: fast tick so the timer test doesn't wait long - App__Scheduler__ConfigFile=e2e-timers.json diff --git a/perf-harness/scripts/run-e2e-orchestration.ps1 b/perf-harness/scripts/run-e2e-orchestration.ps1 new file mode 100644 index 0000000..fe190c1 --- /dev/null +++ b/perf-harness/scripts/run-e2e-orchestration.ps1 @@ -0,0 +1,602 @@ +<# +.SYNOPSIS + Orchestration-engine section of the e2e regression. Dot-sourced by run-e2e.ps1 (needs $base, Info, + Add-Result and Add-Skip from it); not run on its own. + +.DESCRIPTION + Every check queues its own uniquely named runs into the SUT through PerfApi (/API/PerfE2EStart), lets the + PerfE2E task / PerfE2EPost PostExecution probes record what they saw (start/end ticks, worker, stamped + priority, received results and parameters), and asserts on those records plus the durable run state read + straight from the {TablePrefix}Work table (/API/PerfE2ERuns). All waits are bounded. + + Checks that need MaxConcurrency / StopOnFailure are capability-gated and reported SKIP on an image without + them. The restart check runs last: it restarts the SUT container. + + $OrchChecks (from run-e2e.ps1) limits the run to the named groups: + post fail child prio collide attrib cancel seq order status maxconc stopfail perf restart +#> + +# Perf gates. Calibrated from 3 full runs on the reference dev box (combined role, BgPoolSize=4, cpus=2, +# Azurite) on craft:orch-v2-a4d72cf: 2x the median. Short tasks are bounded by the pump (about BgPoolSize claims +# per 1s poll, so ~4 tasks/s here), which is what the fan-out numbers measure. +$OrchGates = @{ + Fanout1000Sec = 510 # 1,000 no-op tasks + PostExecution, enqueue -> PostExecution ran (median 255s) + ManyRuns300Sec = 160 # 300 single-task runs queued at once, enqueue -> all 300 Done (median 79s) + Ttfs5000Ms = 3800 # 5,000-task run on a warm pump, enqueue -> first task started (median 1.9s) + RssMB = 470 # SUT RSS after the perf runs (median 233MB) + IdleClaimMs = 12000 # run queued into an idle engine -> first start (idle poll cap 10s + slack); not calibrated +} + +$OrchSkipped = @{} +# pwsh -File hands "a,b" over as one string. +$OrchChecks = @($OrchChecks | ForEach-Object { "$_" -split ',' } | Where-Object { $_ }) +$OrchContainer = 'craft-e2e-az-sut' + +function New-OrchId { [guid]::NewGuid().ToString('N').Substring(0, 8) } + +function Invoke-OrchGet([string]$Path, [int]$TimeoutSec = 30) { Invoke-RestMethod "$base$Path" -TimeoutSec $TimeoutSec } + +function Start-OrchRuns([string]$Ns, $Runs, [string]$Sink = 'cache', [int]$TimeoutSec = 120) { + $Body = @{ ns = $Ns; sink = $Sink; runs = @($Runs) } | ConvertTo-Json -Depth 30 -Compress + $R = Invoke-RestMethod "$base/API/PerfE2EStart" -Method Post -ContentType 'application/json' -Body $Body -TimeoutSec $TimeoutSec + return , @($R.runs) +} + +function Get-OrchRecords([string]$Ns, [string]$Source = 'cache') { + return , @((Invoke-OrchGet "/API/PerfE2EState?ns=$Ns&source=$Source" 60).rows) +} + +# Per-run-label counts (tasks recorded, ended, posts, min start / max end ticks) without shipping every record. +function Get-OrchSummary([string]$Ns, [string]$Source = 'cache') { + $Map = @{} + foreach ($R in @((Invoke-OrchGet "/API/PerfE2EState?ns=$Ns&source=$Source&summary=1" 60).runs)) { $Map[$R.run] = $R } + return $Map +} + +function Get-OrchTasks($Rows, [string]$Run) { return , @($Rows.Where({ $_.kind -eq 'T' -and $_.run -eq $Run })) } +function Get-OrchPosts($Rows, [string]$Run) { return , @($Rows.Where({ $_.kind -eq 'P' -and $_.run -eq $Run })) } +function Get-OrchRuns([string]$Name, [switch]$Detail) { + return , @((Invoke-OrchGet "/API/PerfE2ERuns?name=$Name$(if ($Detail) { '&detail=1' })").runs) +} +function Get-OrchRunsByPrefix([string]$Prefix) { return , @((Invoke-OrchGet "/API/PerfE2ERuns?prefix=$Prefix" 60).runs) } +function Invoke-OrchBridge([string]$Op, [string]$Name = '') { Invoke-OrchGet "/API/PerfE2EBridge?op=$Op&name=$Name" } + +# Poll $Condition until it returns something truthy or the timeout lapses. Locals are prefixed so they cannot +# shadow the caller's variables the condition reads (scriptblocks resolve variables dynamically). +function Wait-Orch([scriptblock]$Condition, [int]$TimeoutSec, [int]$IntervalMs = 500) { + $WaitSw = [Diagnostics.Stopwatch]::StartNew() + while ($true) { + $WaitVal = try { & $Condition } catch { $null } + if ($WaitVal) { return [pscustomobject]@{ ok = $true; sec = [math]::Round($WaitSw.Elapsed.TotalSeconds, 1); value = $WaitVal } } + if ($WaitSw.Elapsed.TotalSeconds -ge $TimeoutSec) { break } + Start-Sleep -Milliseconds $IntervalMs + } + return [pscustomobject]@{ ok = $false; sec = [math]::Round($WaitSw.Elapsed.TotalSeconds, 1); value = $null } +} + +function Wait-OrchPosts([string]$Ns, [string[]]$Labels, [int]$TimeoutSec, [string]$Source = 'cache') { + Wait-Orch { + $S = Get-OrchSummary $Ns $Source + if (@($Labels.Where({ -not $S[$_] -or $S[$_].posts -lt 1 })).Count -eq 0) { $true } + } $TimeoutSec +} + +function Wait-OrchRunsDone([string]$Name, [int]$Count, [int]$TimeoutSec, [int]$IntervalMs = 1000) { + Wait-Orch { + $R = Get-OrchRuns $Name + if ($R.Count -ge $Count -and @($R.Where({ $_.phase -ne 'Done' })).Count -eq 0) { , $R } + } $TimeoutSec $IntervalMs +} + +function ConvertTo-OrchMs([long]$Ticks) { [math]::Round($Ticks / 10000) } + +# Highest number of intervals open at once. Ends sort before starts at the same tick. +function Get-OrchMaxOverlap($Tasks) { + $Events = [System.Collections.Generic.List[object]]::new() + foreach ($T in $Tasks) { + if ([long]$T.end -le 0) { continue } + $Events.Add([pscustomobject]@{ t = [long]$T.start; d = 1 }) + $Events.Add([pscustomobject]@{ t = [long]$T.end; d = -1 }) + } + $Max = 0; $Cur = 0 + foreach ($E in ($Events | Sort-Object t, d)) { $Cur = $Cur + $E.d; if ($Cur -gt $Max) { $Max = $Cur } } + return $Max +} + +function Get-OrchMin($Values) { ($Values | Measure-Object -Minimum).Minimum } +function Get-OrchMax($Values) { ($Values | Measure-Object -Maximum).Maximum } + +function Invoke-OrchCheck([string]$Group, [scriptblock]$Body) { + if ($OrchChecks -and $Group -notin $OrchChecks) { return } + if ($OrchSkipped[$Group]) { + foreach ($N in $OrchSkipped[$Group].names) { Add-Skip "orch-$Group" $N $OrchSkipped[$Group].reason } + return + } + Info "orchestration: $Group ..." + try { & $Body } + catch { Add-Result "orch-$Group" 'check-error' $false '-' "exception: $($_.Exception.Message) @ line $($_.InvocationInfo.ScriptLineNumber)" } +} + +function Get-OrchBig([int]$Kb) { ('0123456789abcdef' * ($Kb * 64)) + 'END' } + +# Capability gate for the features built after a4d72cf. +$OrchCaps = try { Invoke-OrchBridge 'caps' } catch { $null } +$OrchHasNew = $OrchCaps -and ([int]$OrchCaps.queueFromFileParams -ge 11) -and $OrchCaps.startHasMaxConcurrency -and $OrchCaps.startHasStopOnFailure +if (-not $OrchHasNew) { + $Why = "image lacks the feature (QueueOrchestrationFromFile has $($OrchCaps.queueFromFileParams) params; Start-CraftOrchestrator MaxConcurrency=$($OrchCaps.startHasMaxConcurrency) StopOnFailure=$($OrchCaps.startHasStopOnFailure))" + $OrchSkipped['maxconc'] = @{ reason = $Why; names = @('limit-2', 'unlimited-reaches-pool', 'limit-1-lets-p1-in', 'ignored-when-sequential') } + $OrchSkipped['stopfail'] = @{ reason = $Why; names = @('stops-at-failure', 'rest-cancelled', 'postexec-runs', 'default-runs-all') } +} +$OrchPool = [int](Invoke-OrchGet '/API/PerfAllocation').pool.bgTotal + +# -- 1. PostExecution contract ---------------------------------------------------------------------------- +Invoke-OrchCheck 'post' { + $Id = New-OrchId; $Ns = "post-$Id"; $Name = "E2EPost-$Id" + $null = Start-OrchRuns $Ns @(@{ + name = $Name; label = 'p'; tasks = 6; task = @{ holdms = 300; out = 'v' }; overrides = @{ '2' = @{ outKb = 80 } } + post = @{ marker = "mk-$Id"; bigKb = 150; nested = @{ a = 1; b = @('x', 'y'); c = @{ d = 'e' } } } + }) + $W = Wait-OrchPosts $Ns @('p') 90 + Start-Sleep -Seconds 3 # window in which a duplicate PostExecution would show up + $Rows = Get-OrchRecords $Ns + $T = Get-OrchTasks $Rows 'p'; $P = Get-OrchPosts $Rows 'p'; $H = (Get-OrchRuns $Name)[0] + if (-not $W.ok -or $P.Count -eq 0) { + Add-Result 'orch-post' 'postexec-ran' $false "$($W.sec)s" "no PostExecution within 90s; tasks=$($T.Count) status=$($H.status) phase=$($H.phase) post=$($H.postExecStatus)" + return + } + $Post = $P[0] + $Got = @($Post.idxs -split ',' | Sort-Object) -join ',' + $Exp = @(0..5 | ForEach-Object { "p/$_" } | Sort-Object) -join ',' + Add-Result 'orch-post' 'one-line-per-task' ($Post.lines -eq 6 -and $Got -eq $Exp) "$($W.sec)s" "lines=$($Post.lines) entries=$($Post.entries) idxs=$($Post.idxs) lineType=$($Post.firstType)" + $Lens = @($Post.outLens -split ',' | ForEach-Object { [int]$_ } | Sort-Object) -join ',' + Add-Result 'orch-post' 'task-output-over-64k' ($Lens -eq '1,1,1,1,1,81920') '80KB' "output lengths=$($Post.outLens) (task 2 returns 80 KB)" + $Big = Get-OrchBig 150 + $Sha = [Convert]::ToHexString([Security.Cryptography.SHA256]::HashData([Text.Encoding]::UTF8.GetBytes($Big))) + Add-Result 'orch-post' 'params-over-64k' ($Post.bigLen -eq $Big.Length -and $Post.bigSha -eq $Sha) "$([math]::Round($Big.Length / 1KB))KB" "received len=$($Post.bigLen) expected=$($Big.Length) sha-match=$($Post.bigSha -eq $Sha)" + $Nested = if ($Post.nested) { $Post.nested | ConvertFrom-Json } else { $null } + $NestOk = $Nested -and $Nested.a -eq 1 -and (@($Nested.b) -join ',') -eq 'x,y' -and $Nested.c.d -eq 'e' + Add-Result 'orch-post' 'params-intact' ($Post.marker -eq "mk-$Id" -and $NestOk) '-' "marker=$($Post.marker) nested=$($Post.nested)" + $MaxEnd = Get-OrchMax $T.end + $Gap = ConvertTo-OrchMs ($Post.ticks - $MaxEnd) + $OnceOk = $P.Count -eq 1 -and $Post.ticks -ge $MaxEnd -and $T.Count -eq 6 -and $H.status -eq 'Completed' -and $H.postExecStatus -eq 'Completed' + Add-Result 'orch-post' 'once-after-all-tasks' $OnceOk "+${Gap}ms" "posts=$($P.Count) tasks=$($T.Count) post-minus-last-task-end=${Gap}ms run=$($H.status)/$($H.postExecStatus)" +} + +# -- 2. A failing task ------------------------------------------------------------------------------------- +Invoke-OrchCheck 'fail' { + $Id = New-OrchId; $Ns = "fail-$Id"; $Name = "E2EFail-$Id" + $null = Start-OrchRuns $Ns @(@{ name = $Name; label = 'f'; tasks = 6; task = @{ holdms = 200 }; overrides = @{ '3' = @{ fail = $true } }; post = @{ marker = 'f' } }) + $W = Wait-OrchPosts $Ns @('f') 90 + Start-Sleep -Seconds 2 + $Rows = Get-OrchRecords $Ns + $T = Get-OrchTasks $Rows 'f'; $P = Get-OrchPosts $Rows 'f'; $H = (Get-OrchRuns $Name -Detail)[0] + $Failed = @($H.tasks.Where({ $_.status -eq 'Failed' })) + $RunOk = $H.phase -eq 'Done' -and $H.status -eq 'CompletedWithErrors' -and $H.failed -eq 1 -and $H.done -eq 6 -and $H.total -eq 6 -and + $Failed.Count -eq 1 -and $Failed[0].seq -eq 3 -and $Failed[0].error -match 'failed on purpose' + Add-Result 'orch-fail' 'run-completes-with-errors' $RunOk "$($W.sec)s" "status=$($H.status) phase=$($H.phase) done=$($H.done)/$($H.total) failed=$($H.failed) failedSeq=$($Failed.seq) err='$($Failed.error)'" + $Post = $P | Select-Object -First 1 + $PostOk = $P.Count -eq 1 -and $Post.lines -eq 5 -and ($Post.idxs -split ',') -notcontains 'f/3' -and $H.postExecStatus -eq 'Completed' + Add-Result 'orch-fail' 'postexec-still-runs' $PostOk '-' "posts=$($P.Count) lines=$($Post.lines) idxs=$($Post.idxs) postExec=$($H.postExecStatus)" + $PerIdx = @($T | Group-Object idx | ForEach-Object { "$($_.Name)x$($_.Count)" }) -join ',' + $Attempts = @($H.tasks | ForEach-Object { $_.attempt } | Sort-Object -Unique) -join ',' + $OnceOk = $T.Count -eq 6 -and @($T | Group-Object idx).Count -eq 6 -and $Attempts -eq '1' + Add-Result 'orch-fail' 'every-task-ran-once' $OnceOk '-' "starts per idx=$PerIdx attempts=$Attempts" +} + +# -- 3. Child-run gating ------------------------------------------------------------------------------------ +Invoke-OrchCheck 'child' { + $Id = New-OrchId; $Ns = "child-$Id" + # Blocker for the collision-off child: a run of the child's name that is still going when the child is queued. + $null = Start-OrchRuns $Ns @(@{ name = "E2EBlk-$Id"; label = 'blk'; tasks = 1; task = @{ holdms = 15000 } }) + $null = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['blk'].tasks -ge 1) { $true } } 30 + $null = Start-OrchRuns $Ns @( + @{ name = "E2EChP-$Id"; label = 'chP'; tasks = 2; task = @{ holdms = 200 }; post = @{ marker = 'chP' } + overrides = @{ '0' = @{ child = @{ name = "E2EChK-$Id"; label = 'chK'; tasks = 3; holdms = 3000 } } } } + @{ name = "E2EZ-$Id"; label = 'z'; tasks = 2; post = @{ marker = 'z' } + overrides = @{ '0' = @{ child = @{ name = "E2EZs-$Id"; label = 'zs'; tasks = 0 } } + '1' = @{ child = @{ name = "E2EZb-$Id"; label = 'zb'; tasks = 0; via = 'bridge' } } } } + @{ name = "E2ECs-$Id"; label = 'cs'; tasks = 1; post = @{ marker = 'cs' } + overrides = @{ '0' = @{ child = @{ name = "E2EBlk-$Id"; label = 'blk2'; tasks = 1; via = 'bridge'; allowCollision = $false } } } } + @{ name = "E2ESelf-$Id"; label = 'self'; tasks = 1; post = @{ marker = 'self' } + overrides = @{ '0' = @{ child = @{ name = "E2ESelf-$Id"; label = 'self2'; tasks = 1; holdms = 8000 } } } } + @{ name = "E2EFol-$Id"; label = 'fol'; tasks = 1 + post = @{ marker = 'fol'; followOn = @{ name = "E2EFolK-$Id"; label = 'folK'; tasks = 1; holdms = 8000 } } } + ) + $W = Wait-OrchPosts $Ns @('chP', 'z', 'cs', 'self', 'fol') 90 + $null = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['self2'].ended -ge 1 -and $S['folK'].ended -ge 1 -and $S['blk'].ended -ge 1) { $true } } 40 + $Rows = Get-OrchRecords $Ns + $PostOf = @{}; foreach ($L in 'chP', 'z', 'cs', 'self', 'fol') { $PostOf[$L] = (Get-OrchPosts $Rows $L) | Select-Object -First 1 } + + # a) a child holds its parent's PostExecution until the child's last task finished + $Kid = Get-OrchTasks $Rows 'chK'; $KidEnd = Get-OrchMax $Kid.end + $Par = (Get-OrchRuns "E2EChP-$Id")[0]; $KidRun = (Get-OrchRuns "E2EChK-$Id")[0] + $Gap = if ($PostOf['chP']) { ConvertTo-OrchMs ($PostOf['chP'].ticks - $KidEnd) } else { 'n/a' } + $Ok = $PostOf['chP'] -and $Kid.Count -eq 3 -and @($Kid.Where({ $_.end -le 0 })).Count -eq 0 -and $PostOf['chP'].ticks -gt $KidEnd -and + $Par.total -eq 3 -and $KidRun.parentRunKey -eq $Par.runKey + Add-Result 'orch-child' 'child-gates-parent' $Ok "+${Gap}ms" "parent post - child last end=${Gap}ms childTasks=$($Kid.Count) parentTotal=$($Par.total) (2 tasks+1 child) childParentKey-match=$($KidRun.parentRunKey -eq $Par.runKey)" + + # b) a child that is never created releases its parent: via Start-CraftOrchestrator (0 tasks -> NoTasks, never + # reaches the bridge) and via the bridge (registered with the parent, then abandoned when the batch is empty) + $ZRun = (Get-OrchRuns "E2EZ-$Id")[0] + $Zs = @($Rows.Where({ $_.kind -eq 'Q' -and $_.run -eq 'zs' })) | Select-Object -First 1 + $NoKids = (Get-OrchRuns "E2EZs-$Id").Count + (Get-OrchRuns "E2EZb-$Id").Count + $Ok = $PostOf['z'] -and $ZRun.phase -eq 'Done' -and $ZRun.status -eq 'Completed' -and $ZRun.total -eq 3 -and $ZRun.done -eq 3 -and + $Zs.result -like '*-NoTasks' -and $NoKids -eq 0 + Add-Result 'orch-child' 'zero-task-child-releases' $Ok '-' "post=$([bool]$PostOf['z']) status=$($ZRun.status) total/done=$($ZRun.total)/$($ZRun.done) (2 tasks + 1 bridge-registered child) startResult=$($Zs.result) childRunsCreated=$NoKids" + + # c) a collision-off child skipped because its name is active releases the parent without waiting for the blocker + $BlkEnd = Get-OrchMax (Get-OrchTasks $Rows 'blk').end + $BlkRuns = Get-OrchRuns "E2EBlk-$Id" + $Ok = $PostOf['cs'] -and $PostOf['cs'].ticks -lt $BlkEnd -and $BlkRuns.Count -eq 1 -and (Get-OrchTasks $Rows 'blk2').Count -eq 0 + $Lead = if ($PostOf['cs']) { ConvertTo-OrchMs ($BlkEnd - $PostOf['cs'].ticks) } else { 'n/a' } + Add-Result 'orch-child' 'skipped-child-releases' $Ok "-${Lead}ms" "parent post ran ${Lead}ms before the blocker ended; runs named E2EBlk=$($BlkRuns.Count) skippedChildTasks=$((Get-OrchTasks $Rows 'blk2').Count)" + + # d) a run re-queueing its own name is not its child + $Self2 = (Get-OrchTasks $Rows 'self2') | Select-Object -First 1 + $SelfRuns = Get-OrchRuns "E2ESelf-$Id" + $Ok = $PostOf['self'] -and $Self2 -and $PostOf['self'].ticks -lt $Self2.end -and $SelfRuns.Count -eq 2 -and @($SelfRuns.Where({ $_.parentRunKey })).Count -eq 0 + $Lead = if ($PostOf['self'] -and $Self2) { ConvertTo-OrchMs ($Self2.end - $PostOf['self'].ticks) } else { 'n/a' } + Add-Result 'orch-child' 'self-requeue-not-child' $Ok "-${Lead}ms" "parent post ${Lead}ms before the re-queued run ended; runs=$($SelfRuns.Count) withParent=$(@($SelfRuns.Where({ $_.parentRunKey })).Count)" + + # e) a run queued from a PostExecution is not a child (the parent is aggregating, not running tasks) + $Fol = (Get-OrchRuns "E2EFol-$Id")[0]; $FolK = (Get-OrchTasks $Rows 'folK') | Select-Object -First 1; $FolKRun = (Get-OrchRuns "E2EFolK-$Id")[0] + $Ok = $Fol.phase -eq 'Done' -and $FolK -and $Fol.completedTicks -lt $FolK.end -and -not $FolKRun.parentRunKey + $Lead = if ($FolK) { ConvertTo-OrchMs ($FolK.end - $Fol.completedTicks) } else { 'n/a' } + Add-Result 'orch-child' 'postexec-run-not-child' $Ok "-${Lead}ms" "parent finished ${Lead}ms before the follow-on ended; parent=$($Fol.status)/$($Fol.postExecStatus) followOnParent='$($FolKRun.parentRunKey)'" + if (-not $W.ok) { Add-Result 'orch-child' 'all-parents-finished' $false "$($W.sec)s" "not every parent PostExecution ran within 90s: $((@('chP','z','cs','self','fol').Where({ -not $PostOf[$_] })) -join ',') missing" } +} + +# -- 4. What a child inherits ------------------------------------------------------------------------------- +Invoke-OrchCheck 'prio' { + $Id = New-OrchId; $Ns = "prio-$Id" + $null = Start-OrchRuns $Ns @( + @{ name = "E2EPri-$Id"; label = 'pri'; Priority = 7; tasks = 2; post = @{ marker = 'pri' } + overrides = @{ '0' = @{ child = @{ name = "E2EPriK1-$Id"; label = 'k1'; tasks = 1 } } + '1' = @{ child = @{ name = "E2EPriK2-$Id"; label = 'k2'; tasks = 1; priority = 2 } } } } + @{ name = "E2ESq-$Id"; label = 'sq'; Sequential = $true; tasks = 2; task = @{ holdms = 200 }; post = @{ marker = 'sq' } + overrides = @{ '0' = @{ child = @{ name = "E2ESqK-$Id"; label = 'sqk'; tasks = 3; holdms = 2000 } } } } + @{ name = "E2ENc-$Id"; label = 'nc'; AllowCollision = $false; tasks = 2; post = @{ marker = 'nc' } + overrides = @{ '0' = @{ child = @{ name = "E2ENcK-$Id"; label = 'nck'; tasks = 1; holdms = 2000 } } + '1' = @{ child = @{ name = "E2ENcK-$Id"; label = 'nck'; tasks = 1; holdms = 2000 } } } } + ) + $W = Wait-OrchPosts $Ns @('pri', 'sq', 'nc') 90 + $Rows = Get-OrchRecords $Ns + $ParPrio = @((Get-OrchTasks $Rows 'pri').prio | Sort-Object -Unique) -join ',' + $K1 = (Get-OrchTasks $Rows 'k1') | Select-Object -First 1; $K1Run = (Get-OrchRuns "E2EPriK1-$Id")[0] + Add-Result 'orch-prio' 'child-inherits-priority' ($K1.prio -eq 7 -and $K1Run.priority -eq 7 -and $ParPrio -eq '7') "$($W.sec)s" "parent task prio=$ParPrio; child (no Priority) task ctx prio=$($K1.prio) run P=$($K1Run.priority)" + $K2 = (Get-OrchTasks $Rows 'k2') | Select-Object -First 1; $K2Run = (Get-OrchRuns "E2EPriK2-$Id")[0] + Add-Result 'orch-prio' 'explicit-child-priority-wins' ($K2.prio -eq 2 -and $K2Run.priority -eq 2) '-' "child Priority=2 under a P7 parent: task ctx prio=$($K2.prio) run P=$($K2Run.priority)" + $SqRun = (Get-OrchRuns "E2ESq-$Id")[0]; $SqkRun = (Get-OrchRuns "E2ESqK-$Id")[0]; $Sqk = Get-OrchTasks $Rows 'sqk' + $Ov = Get-OrchMaxOverlap $Sqk + Add-Result 'orch-prio' 'child-not-sequential' ($SqRun.sequential -eq 1 -and $SqkRun.sequential -eq 0 -and $Ov -ge 2) "overlap=$Ov" "parent sequential=$($SqRun.sequential) child sequential=$($SqkRun.sequential) child tasks max concurrent=$Ov (3 tasks x 2s)" + $NcRuns = Get-OrchRuns "E2ENcK-$Id" + $Ok = $NcRuns.Count -eq 2 -and @($NcRuns.Where({ $_.phase -ne 'Done' })).Count -eq 0 + Add-Result 'orch-prio' 'child-not-collision-off' $Ok '-' "collision-off parent queued two same-name children: runs created=$($NcRuns.Count) done=$(@($NcRuns.Where({ $_.phase -eq 'Done' })).Count)" +} + +# -- 5. Collisions -------------------------------------------------------------------------------------------- +Invoke-OrchCheck 'collide' { + $Id = New-OrchId; $Ns = "col-$Id" + $null = Start-OrchRuns $Ns @( + @{ name = "E2EStk-$Id"; label = 'stkA'; tasks = 3; task = @{ holdms = 1000 }; post = @{ marker = 'a' } } + @{ name = "E2EStk-$Id"; label = 'stkB'; tasks = 3; task = @{ holdms = 1000 }; post = @{ marker = 'b' } } + ) + $W = Wait-OrchPosts $Ns @('stkA', 'stkB') 60 + $Rows = Get-OrchRecords $Ns + $Stk = Get-OrchRuns "E2EStk-$Id" + $Ok = $W.ok -and $Stk.Count -eq 2 -and @($Stk.Where({ $_.status -eq 'Completed' })).Count -eq 2 -and + ((Get-OrchTasks $Rows 'stkA').Count + (Get-OrchTasks $Rows 'stkB').Count) -eq 6 + Add-Result 'orch-collide' 'default-stacks' $Ok "$($W.sec)s" "same-name runs=$($Stk.Count) completed=$(@($Stk.Where({ $_.status -eq 'Completed' })).Count) tasks=$((Get-OrchTasks $Rows 'stkA').Count)+$((Get-OrchTasks $Rows 'stkB').Count) posts=$((Get-OrchPosts $Rows 'stkA').Count)+$((Get-OrchPosts $Rows 'stkB').Count)" + + $Name = "E2ENoC-$Id" + $R1 = (Start-OrchRuns $Ns @(@{ name = $Name; label = 'noc1'; AllowCollision = $false; tasks = 2; task = @{ holdms = 4000 }; post = @{ marker = '1' } }))[0] + $Active = (Invoke-OrchBridge 'active' $Name).active + $R2 = (Start-OrchRuns $Ns @(@{ name = $Name; label = 'noc2'; AllowCollision = $false; tasks = 2; post = @{ marker = '2' } }))[0] + Add-Result 'orch-collide' 'active-while-running' ($Active -eq $true) '-' "IsRunActive=$Active right after queueing" + Add-Result 'orch-collide' 'collision-off-skips' ($R1.result -eq "Craft-$Name" -and $R2.result -eq "Craft-$Name-Skipped" -and $R2.warning) '-' "first=$($R1.result) second=$($R2.result) warning='$($R2.warning)'" + $null = Wait-OrchPosts $Ns @('noc1') 60 + $Idle = Wait-Orch { if ((Invoke-OrchBridge 'active' $Name).active -eq $false) { $true } } 15 + $R3 = (Start-OrchRuns $Ns @(@{ name = $Name; label = 'noc3'; AllowCollision = $false; tasks = 1; post = @{ marker = '3' } }))[0] + $W3 = Wait-OrchPosts $Ns @('noc3') 60 + $Rows = Get-OrchRecords $Ns + $Runs = Get-OrchRuns $Name + $Ok = $Idle.ok -and $R3.result -eq "Craft-$Name" -and $W3.ok -and $Runs.Count -eq 2 -and (Get-OrchTasks $Rows 'noc2').Count -eq 0 + Add-Result 'orch-collide' 'restarts-after-finish' $Ok "$($W3.sec)s" "inactive after finish=$($Idle.ok) third=$($R3.result) runs of name=$($Runs.Count) (skipped one never created; its tasks=$((Get-OrchTasks $Rows 'noc2').Count))" +} + +# -- 6. Parent attribution with overlapping same-name parents ---------------------------------------------- +Invoke-OrchCheck 'attrib' { + $Id = New-OrchId; $Ns = "att-$Id" + $null = Start-OrchRuns $Ns @( + @{ name = "E2EOvl-$Id"; label = 'ovlA'; tasks = 1; post = @{ marker = 'A' }; overrides = @{ '0' = @{ child = @{ name = "E2EOvlKA-$Id"; label = 'kA'; tasks = 1; holdms = 2000 } } } } + @{ name = "E2EOvl-$Id"; label = 'ovlB'; tasks = 1; post = @{ marker = 'B' }; overrides = @{ '0' = @{ child = @{ name = "E2EOvlKB-$Id"; label = 'kB'; tasks = 1; holdms = 9000 } } } } + ) + $W = Wait-OrchPosts $Ns @('ovlA', 'ovlB') 60 + $Rows = Get-OrchRecords $Ns + $PA = (Get-OrchPosts $Rows 'ovlA') | Select-Object -First 1; $PB = (Get-OrchPosts $Rows 'ovlB') | Select-Object -First 1 + $KA = (Get-OrchTasks $Rows 'kA') | Select-Object -First 1; $KB = (Get-OrchTasks $Rows 'kB') | Select-Object -First 1 + $KeyA = ((Get-OrchTasks $Rows 'ovlA') | Select-Object -First 1).runKey; $KeyB = ((Get-OrchTasks $Rows 'ovlB') | Select-Object -First 1).runKey + $KARun = (Get-OrchRuns "E2EOvlKA-$Id")[0]; $KBRun = (Get-OrchRuns "E2EOvlKB-$Id")[0] + $Ok = $PA -and $PB -and $PA.ticks -gt $KA.end -and $PB.ticks -gt $KB.end -and $PA.ticks -lt $KB.end + Add-Result 'orch-attrib' 'waits-for-own-child' $Ok "$($W.sec)s" ("A post-kA end={0}ms, B post-kB end={1}ms, A post before kB end by {2}ms" -f + $(if ($PA) { ConvertTo-OrchMs ($PA.ticks - $KA.end) }), $(if ($PB) { ConvertTo-OrchMs ($PB.ticks - $KB.end) }), $(if ($PA) { ConvertTo-OrchMs ($KB.end - $PA.ticks) })) + Add-Result 'orch-attrib' 'child-linked-to-exact-run' ($KeyA -ne $KeyB -and $KARun.parentRunKey -eq $KeyA -and $KBRun.parentRunKey -eq $KeyB) '-' "kA parent=$($KARun.parentRunKey) (A=$KeyA) kB parent=$($KBRun.parentRunKey) (B=$KeyB)" +} + +# -- 7. CancelRun by name ------------------------------------------------------------------------------------- +Invoke-OrchCheck 'cancel' { + $Id = New-OrchId; $Ns = "can-$Id"; $Name = "E2ECan-$Id" + $null = Start-OrchRuns $Ns @( + @{ name = $Name; label = 'canA'; tasks = 8; task = @{ holdms = 2500 }; post = @{ marker = 'A' } } + @{ name = $Name; label = 'canB'; tasks = 8; task = @{ holdms = 2500 }; post = @{ marker = 'B' } } + ) + $null = Wait-Orch { $S = Get-OrchSummary $Ns; if (($S['canA'].tasks + $S['canB'].tasks) -ge 3) { $true } } 30 250 + $Count = (Invoke-OrchBridge 'cancel' $Name).cancelled + $D = Wait-OrchRunsDone $Name 2 90 + $null = Wait-OrchPosts $Ns @('canA', 'canB') 30 + $Rows = Get-OrchRecords $Ns + $Runs = Get-OrchRuns $Name + $Desc = @($Runs | ForEach-Object { "$($_.status) done=$($_.done) cancelled=$($_.cancelled) failed=$($_.failed)" }) -join '; ' + $Ok = $D.ok -and $Count -gt 0 -and $Runs.Count -eq 2 -and @($Runs.Where({ $_.status -eq 'CompletedWithErrors' -and $_.cancelled -gt 0 })).Count -eq 2 + Add-Result 'orch-cancel' 'cancels-every-outing' $Ok "$($D.sec)s" "CancelRun returned $Count; $Desc" + # The run keys are in start order; the outings were queued A then B. + $PostOk = $true; $Parts = [System.Collections.Generic.List[string]]::new() + $I = 0 + foreach ($L in 'canA', 'canB') { + $P = Get-OrchPosts $Rows $L; $T = Get-OrchTasks $Rows $L; $Run = $Runs[$I]; $I = $I + 1 + $Ran = $Run.done - $Run.cancelled - $Run.failed + $One = $P.Count -eq 1 -and $P[0].lines -eq $Ran -and $T.Count -eq $Ran -and $Run.postExecStatus -eq 'Completed' + if (-not $One) { $PostOk = $false } + $Parts.Add("${L}: posts=$($P.Count) lines=$($P[0].lines) started=$($T.Count) ran=$Ran postExec=$($Run.postExecStatus)") + } + Add-Result 'orch-cancel' 'postexec-still-runs' $PostOk '-' ($Parts -join '; ') +} + +# -- 8. Sequential runs --------------------------------------------------------------------------------------- +Invoke-OrchCheck 'seq' { + $Enq = Invoke-OrchGet '/API/PerfSeqWorkerEnqueue?runs=4&steps=5&holdms=500' + $W = Wait-Orch { $R = Invoke-OrchGet '/API/PerfSeqWorkerResult'; if ($R.count -ge 20) { $R } } 90 + Start-Sleep -Milliseconds 500 + $Rows = @((Invoke-OrchGet '/API/PerfSeqWorkerResult').rows) + $OrderOk = $true; $PinOk = $true; $Workers = [System.Collections.Generic.List[string]]::new(); $Parts = [System.Collections.Generic.List[string]]::new() + foreach ($N in @($Enq.names)) { + $Steps = @($Rows.Where({ $_.run -eq $N }) | Sort-Object ticks) + $Order = @($Steps.idx) -join '' + $Ws = @($Steps.worker | Sort-Object -Unique) + if ($Order -ne '01234') { $OrderOk = $false } + if ($Ws.Count -ne 1) { $PinOk = $false } + foreach ($X in $Ws) { $Workers.Add($X) } + $Parts.Add("$($N.Substring(0, 4)):$Order@$($Ws -join '/')") + } + $Distinct = @($Workers | Sort-Object -Unique).Count + Add-Result 'orch-seq' 'payload-order' ($W.ok -and $OrderOk -and $Rows.Count -eq 20) "$($W.sec)s" "steps=$($Rows.Count) $($Parts -join ' ')" + Add-Result 'orch-seq' 'one-pinned-worker-per-run' ($PinOk -and $Distinct -eq [math]::Min(4, $OrchPool)) '-' "workers per run all 1=$PinOk; distinct workers across 4 concurrent runs=$Distinct" + + $Id = New-OrchId; $Ns = "seq-$Id"; $Name = "E2ESeqF-$Id" + $null = Start-OrchRuns $Ns @(@{ name = $Name; label = 'sf'; Sequential = $true; tasks = 5; task = @{ holdms = 300 }; overrides = @{ '2' = @{ fail = $true } }; post = @{ marker = 'sf' } }) + $W = Wait-OrchPosts $Ns @('sf') 60 + $Rows = Get-OrchRecords $Ns + $T = @((Get-OrchTasks $Rows 'sf') | Sort-Object start); $P = (Get-OrchPosts $Rows 'sf') | Select-Object -First 1; $H = (Get-OrchRuns $Name)[0] + $Ok = $T.Count -eq 5 -and (@($T.idx) -join '') -eq '01234' -and @($T.worker | Sort-Object -Unique).Count -eq 1 -and (Get-OrchMaxOverlap $T) -eq 1 -and + @($T.Where({ $_.end -le 0 })).Count -eq 0 -and $H.status -eq 'CompletedWithErrors' -and $H.failed -eq 1 -and $H.done -eq 5 + Add-Result 'orch-seq' 'failed-step-continues' $Ok "$($W.sec)s" "order=$(@($T.idx) -join '') workers=$(@($T.worker | Sort-Object -Unique) -join '/') overlap=$(Get-OrchMaxOverlap $T) status=$($H.status) done=$($H.done) failed=$($H.failed)" + # Every step that succeeded must reach the PostExecution, including the one right after the failed step. + $Got = @($P.idxs -split ',' | Sort-Object) -join ',' + Add-Result 'orch-seq' 'postexec-gets-every-success' ($P.lines -eq 4 -and $Got -eq 'sf/0,sf/1,sf/3,sf/4') '-' "expected sf/0,sf/1,sf/3,sf/4 (step 2 throws); PostExecution received lines=$($P.lines) idxs=$($P.idxs)" +} + +# -- 9. Priority and start-order ------------------------------------------------------------------------------ +# A hold run of pool+3 tasks fills every worker and leaves 3 claims in the local buffer (above the pump's low +# water mark of 2), so nothing else is claimed until the hold wave ends. The runs under test are queued in one +# call while that is the case; whatever is claimed first after it is down to the store's ordering. +function Start-OrchSaturation([string]$Ns, [string]$Name) { + $null = Start-OrchRuns $Ns @(@{ name = $Name; label = 'hold'; tasks = ($OrchPool + 3); task = @{ holdms = 6000 } }) + Wait-Orch { + $S = Get-OrchSummary $Ns + $A = Invoke-OrchGet '/API/PerfAllocation' + if ($S['hold'].tasks -ge $OrchPool -and [int]$A.jm.queued -ge 3) { $true } + } 30 250 +} + +Invoke-OrchCheck 'order' { + $Id = New-OrchId; $Ns = "ord-$Id" + $Sat = Start-OrchSaturation $Ns "E2EHold-$Id" + $null = Start-OrchRuns $Ns @( + @{ name = "E2EP9-$Id"; label = 'p9'; Priority = 9; tasks = 4; task = @{ holdms = 300 } } + @{ name = "E2EP1-$Id"; label = 'p1'; Priority = 1; tasks = 4; task = @{ holdms = 300 } } + ) + $W = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['p9'].ended -ge 4 -and $S['p1'].ended -ge 4) { $true } } 90 + $Rows = Get-OrchRecords $Ns + $P1 = Get-OrchTasks $Rows 'p1'; $P9 = Get-OrchTasks $Rows 'p9' + $Ok = $Sat.ok -and $W.ok -and (Get-OrchMax $P1.start) -lt (Get-OrchMin $P9.start) + Add-Result 'orch-order' 'lower-priority-first' $Ok ("{0}ms" -f (ConvertTo-OrchMs ((Get-OrchMin $P9.start) - (Get-OrchMax $P1.start)))) "saturated=$($Sat.ok); P9 queued first, then P1: last P1 start precedes first P9 start by $(ConvertTo-OrchMs ((Get-OrchMin $P9.start) - (Get-OrchMax $P1.start)))ms" + + $Id = New-OrchId; $Ns = "ordb-$Id" + $Sat = Start-OrchSaturation $Ns "E2EHold-$Id" + $null = Start-OrchRuns $Ns @( + @{ name = "E2EZulu-$Id"; label = 'zulu'; Priority = 6; tasks = 4; task = @{ holdms = 300 } } + @{ name = "E2EAlpha-$Id"; label = 'alpha'; Priority = 6; tasks = 4; task = @{ holdms = 300 } } + ) + $W = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['zulu'].ended -ge 4 -and $S['alpha'].ended -ge 4) { $true } } 90 + $S = Get-OrchSummary $Ns + $Lead = ConvertTo-OrchMs ($S['alpha'].minStart - $S['zulu'].minStart) + Add-Result 'orch-order' 'older-run-first-in-band' ($Sat.ok -and $W.ok -and $S['zulu'].minStart -lt $S['alpha'].minStart) "${Lead}ms" "saturated=$($Sat.ok); Zulu queued before Alpha (same band): Zulu first start leads Alpha by ${Lead}ms" +} + +# -- 11. Status consistency --------------------------------------------------------------------------------- +Invoke-OrchCheck 'status' { + $Id = New-OrchId; $Ns = "st-$Id"; $Name = "E2EStat-$Id" + $null = Start-OrchRuns $Ns @(@{ name = $Name; label = 'st'; tasks = 200; task = @{ holdms = 300 } }) + $Samples = 0; $Bad = [System.Collections.Generic.List[string]]::new(); $SumBad = [System.Collections.Generic.List[string]]::new() + $Sw = [Diagnostics.Stopwatch]::StartNew() + while ($Sw.Elapsed.TotalSeconds -lt 120) { + $H = (Get-OrchRuns $Name)[0] + if ($H.phase -eq 'Done') { break } + $Sm = (Invoke-OrchBridge 'summaries' $Name).runs | Select-Object -First 1 + $G = Invoke-OrchBridge 'summary' + if ($Sm -and ($Sm.queued + $Sm.running) -gt 0) { + $Samples = $Samples + 1 + if ($Sm.total -ne 200 -or ($Sm.completed + $Sm.queued + $Sm.running) -gt $Sm.total) { + $Bad.Add("t=$([math]::Round($Sw.Elapsed.TotalSeconds,1))s total=$($Sm.total) c=$($Sm.completed) q=$($Sm.queued) r=$($Sm.running)") + } + if (($G.jobsQueued + $G.jobsActive) -gt 200) { $SumBad.Add("q=$($G.jobsQueued) a=$($G.jobsActive)") } + } + Start-Sleep -Milliseconds 400 + } + $Done = (Get-OrchRuns $Name)[0] + Add-Result 'orch-status' 'summaries-consistent' ($Samples -ge 5 -and $Bad.Count -eq 0) "$Samples samples" "in-flight samples=$Samples violations=$($Bad.Count) $(@($Bad | Select-Object -First 3) -join ' | ')" + Add-Result 'orch-status' 'summary-bounded' ($Samples -ge 5 -and $SumBad.Count -eq 0) '-' "GetSummary queued+active > batch in $($SumBad.Count) samples $(@($SumBad | Select-Object -First 3) -join ' | ')" + $Left = Wait-Orch { + $Sm = (Invoke-OrchBridge 'summaries' $Name).runs | Select-Object -First 1 + if ((-not $Sm -or ($Sm.queued + $Sm.running) -eq 0) -and (Invoke-OrchBridge 'active' $Name).active -eq $false) { $true } + } 20 + $Sm = (Invoke-OrchBridge 'summaries' $Name).runs | Select-Object -First 1 + $S = Get-OrchSummary $Ns + $Ok = $Done.phase -eq 'Done' -and $Done.status -eq 'Completed' -and $Done.done -eq 200 -and $S['st'].tasks -eq 200 -and $Left.ok + Add-Result 'orch-status' 'leaves-active-set' $Ok "$([math]::Round($Sw.Elapsed.TotalSeconds,1))s" "run=$($Done.status) done=$($Done.done) recorded=$($S['st'].tasks); after: summary q=$($Sm.queued) r=$($Sm.running) c=$($Sm.completed) total=$($Sm.total) active=$(-not $Left.ok)" +} + +# -- 13. MaxConcurrency (gated) ------------------------------------------------------------------------------- +Invoke-OrchCheck 'maxconc' { + $Id = New-OrchId; $Ns = "mc-$Id" + $null = Start-OrchRuns $Ns @(@{ name = "E2EMc2-$Id"; label = 'mc2'; MaxConcurrency = 2; tasks = 12; task = @{ holdms = 1500 }; post = @{ marker = 'mc2' } }) + $W = Wait-OrchPosts $Ns @('mc2') 120 + $Ov = Get-OrchMaxOverlap (Get-OrchTasks (Get-OrchRecords $Ns) 'mc2') + Add-Result 'orch-maxconc' 'limit-2' ($W.ok -and $Ov -eq 2) "$($W.sec)s" "MaxConcurrency=2, 12 tasks x 1.5s: max concurrent=$Ov" + + $null = Start-OrchRuns $Ns @(@{ name = "E2EMc0-$Id"; label = 'mc0'; MaxConcurrency = 0; tasks = 12; task = @{ holdms = 1500 }; post = @{ marker = 'mc0' } }) + $W = Wait-OrchPosts $Ns @('mc0') 120 + $Ov = Get-OrchMaxOverlap (Get-OrchTasks (Get-OrchRecords $Ns) 'mc0') + Add-Result 'orch-maxconc' 'unlimited-reaches-pool' ($W.ok -and $Ov -eq $OrchPool) "$($W.sec)s" "MaxConcurrency=0: max concurrent=$Ov pool=$OrchPool" + + $null = Start-OrchRuns $Ns @(@{ name = "E2EMc1-$Id"; label = 'mc1'; MaxConcurrency = 1; tasks = 6; task = @{ holdms = 1000 }; post = @{ marker = 'mc1' } }) + $null = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['mc1'].tasks -ge 2) { $true } } 30 250 + $null = Start-OrchRuns $Ns @(@{ name = "E2EMcP1-$Id"; label = 'mcp1'; Priority = 1; tasks = 1 }) + $W = Wait-OrchPosts $Ns @('mc1') 60 + $Rows = Get-OrchRecords $Ns + $Mc1 = Get-OrchTasks $Rows 'mc1'; $P1 = (Get-OrchTasks $Rows 'mcp1') | Select-Object -First 1 + $Ok = $W.ok -and (Get-OrchMaxOverlap $Mc1) -eq 1 -and $P1 -and $P1.start -lt (Get-OrchMax $Mc1.start) + Add-Result 'orch-maxconc' 'limit-1-lets-p1-in' $Ok '-' "N=1 overlap=$(Get-OrchMaxOverlap $Mc1); P1 task started $(if ($P1) { ConvertTo-OrchMs ((Get-OrchMax $Mc1.start) - $P1.start) })ms before the N=1 run's last step" + + $R = (Start-OrchRuns $Ns @(@{ name = "E2EMcSq-$Id"; label = 'mcsq'; Sequential = $true; MaxConcurrency = 3; tasks = 3; task = @{ holdms = 300 }; post = @{ marker = 'mcsq' } }))[0] + $W = Wait-OrchPosts $Ns @('mcsq') 60 + $Sq = Get-OrchTasks (Get-OrchRecords $Ns) 'mcsq' + $Ok = $W.ok -and $R.warning -match 'MaxConcurrency' -and (Get-OrchMaxOverlap $Sq) -eq 1 -and $Sq.Count -eq 3 + Add-Result 'orch-maxconc' 'ignored-when-sequential' $Ok '-' "result=$($R.result) warning='$($R.warning)' overlap=$(Get-OrchMaxOverlap $Sq) steps=$($Sq.Count)" +} + +# -- 14. StopOnFailure (gated) -------------------------------------------------------------------------------- +Invoke-OrchCheck 'stopfail' { + $Id = New-OrchId; $Ns = "sof-$Id"; $Name = "E2ESof-$Id" + $null = Start-OrchRuns $Ns @(@{ name = $Name; label = 'sof'; Sequential = $true; StopOnFailure = $true; tasks = 5; task = @{ holdms = 200 }; overrides = @{ '2' = @{ fail = $true } }; post = @{ marker = 'sof' } }) + $W = Wait-OrchPosts $Ns @('sof') 60 + Start-Sleep -Seconds 2 + $Rows = Get-OrchRecords $Ns + $T = @((Get-OrchTasks $Rows 'sof') | Sort-Object start); $P = (Get-OrchPosts $Rows 'sof') | Select-Object -First 1; $H = (Get-OrchRuns $Name -Detail)[0] + Add-Result 'orch-stopfail' 'stops-at-failure' ((@($T.idx) -join '') -eq '012') "$($W.sec)s" "steps that ran=$(@($T.idx) -join ',')" + $Cancelled = @($H.tasks.Where({ $_.status -eq 'Cancelled' }).seq) -join ',' + Add-Result 'orch-stopfail' 'rest-cancelled' ($Cancelled -eq '3,4' -and $H.status -eq 'CompletedWithErrors') '-' "cancelled seqs=$Cancelled status=$($H.status) failed=$($H.failed) cancelled=$($H.cancelled)" + Add-Result 'orch-stopfail' 'postexec-runs' ($P -and $P.lines -eq 2 -and $H.postExecStatus -eq 'Completed') '-' "posts=$(@(Get-OrchPosts $Rows 'sof').Count) lines=$($P.lines) idxs=$($P.idxs) postExec=$($H.postExecStatus)" + $null = Start-OrchRuns $Ns @(@{ name = "E2ESofD-$Id"; label = 'sofd'; Sequential = $true; tasks = 5; task = @{ holdms = 200 }; overrides = @{ '2' = @{ fail = $true } }; post = @{ marker = 'sofd' } }) + $W = Wait-OrchPosts $Ns @('sofd') 60 + $T = Get-OrchTasks (Get-OrchRecords $Ns) 'sofd' + Add-Result 'orch-stopfail' 'default-runs-all' ($T.Count -eq 5) "$($W.sec)s" "without StopOnFailure steps that ran=$($T.Count)" +} + +# -- 12. Performance gates ------------------------------------------------------------------------------------- +Invoke-OrchCheck 'perf' { + $Id = New-OrchId; $Ns = "pf-$Id" + $Sw = [Diagnostics.Stopwatch]::StartNew() + $null = Start-OrchRuns $Ns @(@{ name = "E2EFan-$Id"; label = 'fan'; tasks = 1000; post = @{ marker = 'fan' } }) + $W = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['fan'].posts -ge 1) { $S['fan'] } } ($OrchGates.Fanout1000Sec + 60) 500 + $Sec = [math]::Round($Sw.Elapsed.TotalSeconds, 1) + $Rows = Get-OrchRecords $Ns + $T = Get-OrchTasks $Rows 'fan'; $P = (Get-OrchPosts $Rows 'fan') | Select-Object -First 1 + $Once = $T.Count -eq 1000 -and @($T | Group-Object idx).Count -eq 1000 -and $P.lines -eq 1000 + Add-Result 'orch-perf' 'fanout-1000' ($W.ok -and $Once -and $Sec -le $OrchGates.Fanout1000Sec) "${Sec}s" ("{0:N0} tasks/s; recorded={1} distinct={2} postLines={3}; gate {4}s" -f (1000 / [math]::Max(0.1, $Sec)), $T.Count, @($T | Group-Object idx).Count, $P.lines, $OrchGates.Fanout1000Sec) + $null = Invoke-OrchGet "/API/PerfE2EState?ns=$Ns&wipe=1" + + $Id = New-OrchId; $Ns = "pm-$Id" + $Runs = @(for ($I = 0; $I -lt 300; $I++) { @{ name = "E2EMany-$Id-$I"; label = "m$I"; tasks = 1 } }) + $Sw = [Diagnostics.Stopwatch]::StartNew() + $null = Start-OrchRuns $Ns $Runs -TimeoutSec 300 + $W = Wait-Orch { $R = Get-OrchRunsByPrefix "E2EMany-$Id-"; if ($R.Count -ge 300 -and @($R.Where({ $_.phase -ne 'Done' })).Count -eq 0) { , $R } } ($OrchGates.ManyRuns300Sec + 60) 1000 + $Sec = [math]::Round($Sw.Elapsed.TotalSeconds, 1) + $R = Get-OrchRunsByPrefix "E2EMany-$Id-" + $S = Get-OrchSummary $Ns + $Ran = @($S.Values.Where({ $_.tasks -eq 1 })).Count + Add-Result 'orch-perf' 'runs-300' ($W.ok -and $Ran -eq 300 -and $Sec -le $OrchGates.ManyRuns300Sec) "${Sec}s" "runs=$($R.Count) done=$(@($R.Where({ $_.phase -eq 'Done' })).Count) tasks ran once=$Ran; gate $($OrchGates.ManyRuns300Sec)s" + $null = Invoke-OrchGet "/API/PerfE2EState?ns=$Ns&wipe=1" + + # An idle pump backs off its poll to JobQueueIdlePollIntervalMs (10s default), so a run queued into an idle + # engine waits for the next poll. Pin that bound, then keep the pump warm (a held task in flight keeps it on + # the 1s poll) so the size comparison below measures claiming, not the idle backoff. + $Id = New-OrchId; $Ns = "pt-$Id" + Start-Sleep -Seconds 20 + $EIdle = (Start-OrchRuns $Ns @(@{ name = "E2ETfIdle-$Id"; label = 'tfidle'; tasks = 1 }))[0] + $WIdle = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['tfidle'].tasks -ge 1) { $S['tfidle'] } } 60 100 + $TtfsIdle = if ($WIdle.ok) { ConvertTo-OrchMs ($WIdle.value.minStart - $EIdle.enqueueTicks) } else { -1 } + Add-Result 'orch-perf' 'idle-claim-latency' ($WIdle.ok -and $TtfsIdle -le $OrchGates.IdleClaimMs) "${TtfsIdle}ms" "1-task run queued into an idle engine -> first start ${TtfsIdle}ms; gate $($OrchGates.IdleClaimMs)ms (idle poll cap 10s)" + $null = Start-OrchRuns $Ns @(@{ name = "E2ETfKeep-$Id"; label = 'keep'; tasks = 1; task = @{ holdms = 60000 } }) + $null = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['keep'].tasks -ge 1) { $true } } 30 250 + Start-Sleep -Seconds 2 + $E10 = (Start-OrchRuns $Ns @(@{ name = "E2ETf10-$Id"; label = 'tf10'; tasks = 10 }))[0] + $W10 = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['tf10'].tasks -ge 1) { $S['tf10'] } } 60 100 + $Ttfs10 = if ($W10.ok) { ConvertTo-OrchMs ($W10.value.minStart - $E10.enqueueTicks) } else { -1 } + $null = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['tf10'].ended -ge 10) { $true } } 60 + $E5k = (Start-OrchRuns $Ns @(@{ name = "E2ETf5k-$Id"; label = 'tf5k'; tasks = 5000 }) -TimeoutSec 300)[0] + $W5k = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['tf5k'].tasks -ge 1) { $S['tf5k'] } } 180 100 + $Ttfs5k = if ($W5k.ok) { ConvertTo-OrchMs ($W5k.value.minStart - $E5k.enqueueTicks) } else { -1 } + Add-Result 'orch-perf' 'ttfs-5000' ($W5k.ok -and $Ttfs5k -le $OrchGates.Ttfs5000Ms) "${Ttfs5k}ms" "enqueue->first task start: 5,000-task run ${Ttfs5k}ms vs 10-task run ${Ttfs10}ms (x$([math]::Round($Ttfs5k / [math]::Max(1, $Ttfs10), 1))); gate $($OrchGates.Ttfs5000Ms)ms" + $Cancelled = (Invoke-OrchBridge 'cancel' "E2ETf5k-$Id").cancelled + $D = Wait-OrchRunsDone "E2ETf5k-$Id" 1 240 2000 + $H = (Get-OrchRuns "E2ETf5k-$Id")[0] + Add-Result 'orch-perf' 'cancel-5000' ($D.ok -and $H.status -eq 'CompletedWithErrors' -and $H.done -eq 5000 -and $H.cancelled -gt 0) "$($D.sec)s" "CancelRun returned $Cancelled; run=$($H.status) done=$($H.done)/$($H.total) cancelled=$($H.cancelled)" + $null = Invoke-OrchGet "/API/PerfE2EState?ns=$Ns&wipe=1" + + Start-Sleep -Seconds 5 + $M = Invoke-OrchBridge 'summary' + Add-Result 'orch-perf' 'memory-after-perf' ($M.rssMB -le $OrchGates.RssMB) "$($M.rssMB)MB" "rss=$($M.rssMB)MB workingSet=$($M.workingSetMB)MB heap=$($M.heapMB)MB committed=$($M.committedMB)MB; gate rss $($OrchGates.RssMB)MB" +} + +# -- 10. Restart recovery + legacy table drop (restarts the SUT; keep last) ------------------------------------- +Invoke-OrchCheck 'restart' { + $Id = New-OrchId; $Ns = "rst-$Id"; $Name = "E2ERst-$Id" + $N = $OrchPool + 2 # pool running + 2 buffered claims for the graceful stop to hand back + # Earlier checks may still hold workers (the perf keeper task): start from an idle engine so the whole pool + # goes to this run and the running/buffered split is the intended one. + $Quiet = Wait-Orch { $G = Invoke-OrchBridge 'summary'; if ($G.jobsActive -eq 0 -and $G.jobsQueued -eq 0) { $true } } 120 1000 + $Seed = Invoke-OrchGet '/API/PerfE2ELegacy?op=seed' + $null = Start-OrchRuns $Ns @(@{ name = $Name; label = 'rst'; tasks = $N; task = @{ holdms = 20000 }; post = @{ marker = 'rst' } }) -Sink 'table' + $Up = Wait-Orch { $S = Get-OrchSummary $Ns 'table'; if ($S['rst'].tasks -ge $OrchPool) { $true } } 60 250 + Start-Sleep -Seconds 2 # let the pump buffer the remaining claims + $Pre = Get-OrchTasks (Get-OrchRecords $Ns 'table') 'rst' + $Running = @($Pre.Where({ $_.end -le 0 }).idx | Sort-Object -Unique) + $Unstarted = @((0..($N - 1)).Where({ $_ -notin $Pre.idx })) + Info "restart: docker restart $OrchContainer with $($Running.Count) tasks running, $($Unstarted.Count) not started ..." + $Sw = [Diagnostics.Stopwatch]::StartNew() + docker restart $OrchContainer 2>&1 | Out-Null + $Ready = Wait-Orch { $H = Invoke-OrchGet '/healthz' 5; if ($H.status -eq 'ready') { $true } } 180 1000 + $Legacy = Invoke-OrchGet '/API/PerfE2ELegacy?op=list' + Add-Result 'orch-restart' 'legacy-tables-dropped' ($Ready.ok -and @($Seed.present).Count -eq 5 -and @($Legacy.present).Count -eq 0) "$($Ready.sec)s" "seeded+present before=$(@($Seed.present).Count) present after restart=$(@($Legacy.present).Count) $(@($Legacy.present) -join ',')" + $D = Wait-OrchRunsDone $Name 1 480 3000 + $RestartSec = [math]::Round($Sw.Elapsed.TotalSeconds, 1) + $H = (Get-OrchRuns $Name -Detail)[0] + $Rows = Get-OrchRecords $Ns 'table' + $T = Get-OrchTasks $Rows 'rst'; $P = Get-OrchPosts $Rows 'rst' + $AttemptOf = @{}; foreach ($X in $H.tasks) { $AttemptOf[[int]$X.seq] = [int]$X.attempt } + $Attempts = @($H.tasks | ForEach-Object { "$($_.seq):$($_.attempt)" }) -join ',' + $Starts = @($T | Group-Object idx | Sort-Object { [int]$_.Name } | ForEach-Object { "$($_.Name)x$($_.Count)" }) -join ',' + $Ok = $Quiet.ok -and $Up.ok -and $D.ok -and $H.status -eq 'Completed' -and $H.done -eq $N -and @($H.tasks.Where({ $_.status -eq 'Completed' })).Count -eq $N + Add-Result 'orch-restart' 'run-completes' $Ok "${RestartSec}s" "restart->done ${RestartSec}s; run=$($H.status)/$($H.phase) done=$($H.done)/$($H.total) open=P$($H.open.P)/R$($H.open.R) (engine idle first=$($Quiet.ok))" + $Finished = @($T.Where({ $_.end -gt 0 }) | Group-Object idx) + $OnceOk = $Finished.Count -eq $N -and @($Finished.Where({ $_.Count -ne 1 })).Count -eq 0 + Add-Result 'orch-restart' 'each-task-completes-once' $OnceOk '-' "finished records per idx=$(@($Finished | ForEach-Object { "$($_.Name)x$($_.Count)" }) -join ','); starts per idx=$Starts; D-row attempts=$Attempts" + $ReBad = @($Running.Where({ $I = $_; $AttemptOf[[int]$I] -lt 2 -or @($T.Where({ $_.idx -eq $I })).Count -lt 2 })) + Add-Result 'orch-restart' 'interrupted-rerun' ($Running.Count -ge 1 -and $ReBad.Count -eq 0) '-' "running at restart=$($Running -join ','); their attempts=$(@($Running | ForEach-Object { "${_}:$($AttemptOf[[int]$_])" }) -join ',')" + $UnBad = @($Unstarted.Where({ $AttemptOf[[int]$_] -ne 1 })) + Add-Result 'orch-restart' 'unstarted-claims-released' ($Unstarted.Count -ge 1 -and $UnBad.Count -eq 0) '-' "not started at restart=$($Unstarted -join ','); attempts=$(@($Unstarted | ForEach-Object { "${_}:$($AttemptOf[[int]$_])" }) -join ',') (1 = handed back on graceful stop, 2 = lease lapsed)" + Add-Result 'orch-restart' 'postexec-once' ($P.Count -eq 1 -and $P[0].lines -eq $N -and $H.postExecStatus -eq 'Completed') '-' "posts=$($P.Count) lines=$($P[0].lines) postExec=$($H.postExecStatus)" +} + +Info ("orchestration gates: fanout-1000 <= {0}s, runs-300 <= {1}s, ttfs-5000 <= {2}ms, rss <= {3}MB, idle claim <= {4}ms" -f $OrchGates.Fanout1000Sec, $OrchGates.ManyRuns300Sec, $OrchGates.Ttfs5000Ms, $OrchGates.RssMB, $OrchGates.IdleClaimMs) diff --git a/perf-harness/scripts/run-e2e.ps1 b/perf-harness/scripts/run-e2e.ps1 index ce8b581..b3c4420 100644 --- a/perf-harness/scripts/run-e2e.ps1 +++ b/perf-harness/scripts/run-e2e.ps1 @@ -19,7 +19,10 @@ param( [int]$Port = 5399, [int]$ReadyTimeoutSec = 180, [switch]$Build, - [switch]$KeepUp + [switch]$KeepUp, + # Run only the orchestration section (skips the platform checks), optionally only some of its groups. + [switch]$OrchOnly, + [string[]]$OrchChecks ) $ErrorActionPreference = 'Stop' @@ -33,9 +36,14 @@ $results = New-Object System.Collections.ArrayList function Info($m) { Write-Host "[e2e] $m" -ForegroundColor Cyan } function Add-Result($area, $name, $pass, $perf, $detail) { - [void]$results.Add([pscustomobject]@{ area = $area; name = $name; pass = [bool]$pass; perf = $perf; detail = $detail }) + [void]$results.Add([pscustomobject]@{ area = $area; name = $name; pass = [bool]$pass; skip = $false; perf = $perf; detail = $detail }) $tag = if ($pass) { 'PASS' } else { 'FAIL' } - Write-Host (" [{0}] {1,-12} {2,-18} {3,-8} {4}" -f $tag, $area, $name, $perf, $detail) -ForegroundColor $(if ($pass) { 'Green' } else { 'Red' }) + Write-Host (" [{0}] {1,-14} {2,-28} {3,-10} {4}" -f $tag, $area, $name, $perf, $detail) -ForegroundColor $(if ($pass) { 'Green' } else { 'Red' }) +} +# A check this SUT cannot run (the feature is not in the image). Neither a pass nor a failure. +function Add-Skip($area, $name, $detail) { + [void]$results.Add([pscustomobject]@{ area = $area; name = $name; pass = $true; skip = $true; perf = '-'; detail = $detail }) + Write-Host (" [SKIP] {0,-14} {1,-28} {2,-10} {3}" -f $area, $name, '-', $detail) -ForegroundColor Yellow } function Api($path) { try { return Invoke-RestMethod "$base$path" -TimeoutSec 20 } catch { return $null } } # Same, but against an absolute URL — the throwaway containers further down run on their own ports. @@ -82,6 +90,8 @@ try { Add-Result 'health' 'readiness' $ready '-' "status=$($h.status)" Add-Result 'storage' 'azurite-ready' ($h.ready.storage -eq $true) '-' "storageReady=$($h.ready.storage)" + $suiteSw = [Diagnostics.Stopwatch]::StartNew() + if (-not $OrchOnly) { # ── API dispatch ──────────────────────────────────────────────────────────── $sw = [Diagnostics.Stopwatch]::StartNew(); $ping = Api '/API/PerfPing'; $sw.Stop() Add-Result 'api' 'dispatch-ping' ($ping.ok -eq $true) ("{0}ms" -f $sw.ElapsedMilliseconds) "endpoint=$($ping.endpoint)" @@ -274,11 +284,18 @@ try { $id = Fetch "$base/bundle.js" @('-H', 'Accept-Encoding: identity') Add-Result 'frontend' 'identity-content' ($id.Code -eq 200 -and $id.Body -match 'E2E_BUNDLE_MARKER') ("{0}ms {1}B" -f $id.TimeMs, $id.Size) "identity bundle content correct" + } + + # -- Orchestration engine (run-e2e-orchestration.ps1) ------------------------------ + if ($ready) { . (Join-Path $here 'run-e2e-orchestration.ps1') } + else { Add-Result 'orchestration' 'not-run' $false '-' 'SUT never became ready' } + # ── Summary ───────────────────────────────────────────────────────────────── $fail = @($results | Where-Object { -not $_.pass }) - $pass = @($results | Where-Object { $_.pass }) + $skip = @($results | Where-Object { $_.skip }) + $pass = @($results | Where-Object { $_.pass -and -not $_.skip }) Write-Host "" - Write-Host ("===== E2E: {0} passed, {1} failed =====" -f $pass.Count, $fail.Count) -ForegroundColor $(if ($fail.Count) { 'Red' } else { 'Green' }) + Write-Host ("===== E2E: {0} passed, {1} failed, {2} skipped ({3:N0}s) =====" -f $pass.Count, $fail.Count, $skip.Count, $suiteSw.Elapsed.TotalSeconds) -ForegroundColor $(if ($fail.Count) { 'Red' } else { 'Green' }) # GitHub Actions job summary — renders the PASS/FAIL table on the run page (no-op locally). Written # before the non-zero exit so a failing run still shows exactly which checks failed. @@ -289,7 +306,7 @@ try { [void]$md.AppendLine("| Result | Area | Check | Perf | Detail |") [void]$md.AppendLine("|:------:|------|-------|------|--------|") foreach ($r in $results) { - [void]$md.AppendLine("| $(if ($r.pass) { '✅' } else { '❌' }) | $($r.area) | $($r.name) | $($r.perf) | $($r.detail) |") + [void]$md.AppendLine("| $(if ($r.skip) { 'SKIP' } elseif ($r.pass) { '✅' } else { '❌' }) | $($r.area) | $($r.name) | $($r.perf) | $($r.detail) |") } Add-Content -Path $env:GITHUB_STEP_SUMMARY -Value $md.ToString() } From 97e87eca3c1f9b9dec49ab67563b8bcaabe356a3 Mon Sep 17 00:00:00 2001 From: Zacgoose <107489668+Zacgoose@users.noreply.github.com> Date: Tue, 6 Oct 2026 02:42:08 +0800 Subject: [PATCH 15/24] fix(status): derive every waiting-in-storage count from live holdings, and warn on ignored run options - the worker summary and metrics snapshot added the live local queue to the cached snapshot's unclaimed count, double-counting claims made since; all merged counts now use one helper - Start-CraftOrchestrator warns when MaxConcurrency is set on a sequential run or StopOnFailure on a non-sequential one - e2e status check bounds the durable queue rather than the global one --- Runtime/CraftRuntime/Start-CraftOrchestrator.ps1 | 6 ++++++ Services/Bridges/WorkerMetricsBridge.cs | 12 +++++++----- Services/Orchestration/JobQueueStatusReader.cs | 14 ++++++++++---- perf-harness/scripts/run-e2e-orchestration.ps1 | 6 ++++-- 4 files changed, 27 insertions(+), 11 deletions(-) diff --git a/Runtime/CraftRuntime/Start-CraftOrchestrator.ps1 b/Runtime/CraftRuntime/Start-CraftOrchestrator.ps1 index 72aa4ce..4a1ed1d 100644 --- a/Runtime/CraftRuntime/Start-CraftOrchestrator.ps1 +++ b/Runtime/CraftRuntime/Start-CraftOrchestrator.ps1 @@ -162,6 +162,12 @@ function Start-CraftOrchestrator { $Sequential = [bool]($InputObject.Sequential) $MaxConcurrency = [int]($InputObject.MaxConcurrency ?? 0) $StopOnFailure = [bool]($InputObject.StopOnFailure) + if ($Sequential -and $MaxConcurrency -gt 0) { + Write-Warning "Craft: MaxConcurrency is ignored for '$OrchestratorName': a sequential run already runs one step at a time" + } + if ($StopOnFailure -and -not $Sequential) { + Write-Warning "Craft: StopOnFailure is ignored for '$OrchestratorName': it applies to sequential runs only" + } Write-Information "Craft: Queuing orchestrator '$OrchestratorName' ($TaskCount tasks, P$Priority$(if ($Sequential) { ', Sequential' })$(if ($PostExecFunctionName) { ", PostExec: $PostExecFunctionName" })$(if ($ParentRunName) { ", Parent: $ParentRunName" }))" [Craft.Services.OrchestratorBridge]::QueueOrchestrationFromFile( diff --git a/Services/Bridges/WorkerMetricsBridge.cs b/Services/Bridges/WorkerMetricsBridge.cs index cbe09a0..464b92b 100644 --- a/Services/Bridges/WorkerMetricsBridge.cs +++ b/Services/Bridges/WorkerMetricsBridge.cs @@ -223,12 +223,13 @@ public static WorkerMetricsSnapshot GetSnapshot() // the durable queue table. GetCached never blocks: a stale/missing snapshot kicks off a // background refresh and this poll reports what is known now. var durable = s_queueReader?.GetCached(); + var waiting = durable == null ? 0 : s_queueReader!.WaitingInStorage(durable); snapshot.Jobs = new JobMetrics { - Queued = summary.Queued + (durable?.Unclaimed ?? 0), + Queued = summary.Queued + waiting, QueuedLocal = summary.Queued, - QueuedDurable = durable?.Unclaimed ?? 0, + QueuedDurable = waiting, Running = summary.Running, Completed = summary.Completed, Failed = summary.Failed, @@ -590,7 +591,8 @@ _ when name.Contains("PowerShell", StringComparison.OrdinalIgnoreCase) || /// Get a summary of just the busy/available counts. public static WorkerSummary GetSummary() { - var durable = s_queueReader?.GetCached(); + var snapshot = s_queueReader?.GetCached(); + var waiting = snapshot == null ? 0 : s_queueReader!.WaitingInStorage(snapshot); var localQueued = s_jobManager?.QueuedCount ?? 0; return new WorkerSummary @@ -605,9 +607,9 @@ public static WorkerSummary GetSummary() LimiterWaiting = s_limiter?.Waiting ?? 0, LimiterMax = s_limiter?.CurrentMax ?? 0, IsHttpThrottled = s_limiter?.IsHttpThrottled ?? false, - JobsQueued = localQueued + (durable?.Unclaimed ?? 0), + JobsQueued = localQueued + waiting, JobsQueuedLocal = localQueued, - JobsQueuedDurable = durable?.Unclaimed ?? 0, + JobsQueuedDurable = waiting, JobsActive = s_jobManager?.ActiveCount ?? 0, }; } diff --git a/Services/Orchestration/JobQueueStatusReader.cs b/Services/Orchestration/JobQueueStatusReader.cs index 3ea970a..6ec3c1b 100644 --- a/Services/Orchestration/JobQueueStatusReader.cs +++ b/Services/Orchestration/JobQueueStatusReader.cs @@ -140,16 +140,22 @@ public async Task GetSummaryAsync(CancellationToken ct = default) var snap = await GetAsync(ct: ct); if (snap == null) return summary; - // Waiting in storage = everything outstanding less what this process holds now (its queued and running - // orchestrator jobs). Using the snapshot's own subtraction would count claims made since it was taken twice. - var held = _jobs.GetJobs().Count(j => j.RunName != null && j.Status is "Queued" or "Running"); - summary.QueuedDurable = Math.Max(0, snap.Total - held); + summary.QueuedDurable = WaitingInStorage(snap); summary.Queued += summary.QueuedDurable; if (snap.OldestUnclaimedUtc is { } oldest && (summary.OldestQueuedUtc == null || oldest < summary.OldestQueuedUtc)) summary.OldestQueuedUtc = oldest; return summary; } + /// + /// Tasks waiting in storage, unclaimed: everything the snapshot found outstanding, less what this process holds + /// now (its queued and running orchestrator jobs). The snapshot's own + /// subtracted what was held when it was taken, so adding it to a live local count double-counts every claim + /// made since (seen live: 202 queued on a 200-task run). Every merged count goes through here. + /// + public int WaitingInStorage(QueueSnapshot snap) => + Math.Max(0, snap.Total - _jobs.GetJobs().Count(j => j.RunName != null && j.Status is "Queued" or "Running")); + /// Local job records plus the head of the durable queue, for the job listing. public async Task> GetJobDetailsAsync(string? runName = null, string? status = null, int limit = 100, CancellationToken ct = default) diff --git a/perf-harness/scripts/run-e2e-orchestration.ps1 b/perf-harness/scripts/run-e2e-orchestration.ps1 index fe190c1..f4001c3 100644 --- a/perf-harness/scripts/run-e2e-orchestration.ps1 +++ b/perf-harness/scripts/run-e2e-orchestration.ps1 @@ -439,13 +439,15 @@ Invoke-OrchCheck 'status' { if ($Sm.total -ne 200 -or ($Sm.completed + $Sm.queued + $Sm.running) -gt $Sm.total) { $Bad.Add("t=$([math]::Round($Sw.Elapsed.TotalSeconds,1))s total=$($Sm.total) c=$($Sm.completed) q=$($Sm.queued) r=$($Sm.running)") } - if (($G.jobsQueued + $G.jobsActive) -gt 200) { $SumBad.Add("q=$($G.jobsQueued) a=$($G.jobsActive)") } + # The global summary also counts unrelated local work (the e2e timer, leftovers of other checks), so bound + # the durable part: what is waiting in storage can never exceed this run's 200 tasks. + if ($G.jobsQueuedDurable -gt 200 -or ($G.jobsQueued - $G.jobsQueuedLocal) -ne $G.jobsQueuedDurable) { $SumBad.Add("q=$($G.jobsQueued) local=$($G.jobsQueuedLocal) durable=$($G.jobsQueuedDurable) a=$($G.jobsActive)") } } Start-Sleep -Milliseconds 400 } $Done = (Get-OrchRuns $Name)[0] Add-Result 'orch-status' 'summaries-consistent' ($Samples -ge 5 -and $Bad.Count -eq 0) "$Samples samples" "in-flight samples=$Samples violations=$($Bad.Count) $(@($Bad | Select-Object -First 3) -join ' | ')" - Add-Result 'orch-status' 'summary-bounded' ($Samples -ge 5 -and $SumBad.Count -eq 0) '-' "GetSummary queued+active > batch in $($SumBad.Count) samples $(@($SumBad | Select-Object -First 3) -join ' | ')" + Add-Result 'orch-status' 'summary-bounded' ($Samples -ge 5 -and $SumBad.Count -eq 0) '-' "GetSummary durable queue > batch in $($SumBad.Count) samples $(@($SumBad | Select-Object -First 3) -join ' | ')" $Left = Wait-Orch { $Sm = (Invoke-OrchBridge 'summaries' $Name).runs | Select-Object -First 1 if ((-not $Sm -or ($Sm.queued + $Sm.running) -eq 0) -and (Invoke-OrchBridge 'active' $Name).active -eq $false) { $true } From 1854b76591aad3a54033af0d92d27ccab081575c Mon Sep 17 00:00:00 2001 From: Zacgoose <107489668+Zacgoose@users.noreply.github.com> Date: Tue, 6 Oct 2026 02:58:51 +0800 Subject: [PATCH 16/24] fix(orchestration): make bulk cancel fast and exact, and tighten the e2e perf gates - the pump no longer claims tasks of a run being cancelled (it raced the cancel for the same rows and header); it cancels a page per visit only when no cancel is in progress, so an interrupted cancel still finishes - cancel reuses the rows it just read and keeps going until the pending range is empty; CancelRun reports what the run records - e2e perf gates recalibrated to the faster pump, plus a cancel-5000 gate --- Services/Orchestration/OrchestratorService.cs | 6 +- Services/Orchestration/WorkPump.cs | 11 ++++ Services/Storage/WorkStore.cs | 59 ++++++++++++++--- .../scripts/run-e2e-orchestration.ps1 | 20 ++++-- .../OrchestrationLifecycleTests.cs | 66 +++++++++++++++++++ 5 files changed, 144 insertions(+), 18 deletions(-) diff --git a/Services/Orchestration/OrchestratorService.cs b/Services/Orchestration/OrchestratorService.cs index ddd965c..19c038a 100644 --- a/Services/Orchestration/OrchestratorService.cs +++ b/Services/Orchestration/OrchestratorService.cs @@ -535,7 +535,7 @@ private Func BuildSequentialRunWork(RunHeader header, J if (current.StopOnFailure) { _logger.LogError(ex, "[Scheduler] Sequential task failed: {TaskId} — stopping the run", step.TaskId); - await _store.CancelPendingAsync(header.RunKey, WorkStore.StoppedReason(step.TaskId), jobCt); + await _store.CancelPendingAsync(header.RunKey, WorkStore.StoppedReason(step.TaskId), ct: jobCt); break; } _logger.LogError(ex, "[Scheduler] Sequential task failed: {TaskId} — continuing with the next step", step.TaskId); @@ -587,7 +587,9 @@ private static Dictionary TaskInvocation(Dictionary RefillAsync(CancellationToken ct) continue; } + // A run being cancelled: claiming its pending tasks would only race the cancel. Cancel a page of them instead, + // so a cancel interrupted part-way still finishes; its aggregation runs once it falls due. + if (header.CancelRequested && header.Phase == RunPhase.Tasks) + { + var (cancelled, _) = _store.IsCancelling(header.RunKey) + ? (0, null) + : await _store.CancelPendingAsync(header.RunKey, maxPages: 1, ct: ct); + if (cancelled == 0) _skip[header.RunKey] = (entry.Done, entry.Total, DateTime.MaxValue, 0); + continue; + } + IReadOnlyList claims; var probe = new WorkStore.ClaimProbe(); if (header.Sequential) diff --git a/Services/Storage/WorkStore.cs b/Services/Storage/WorkStore.cs index 5f67d80..2043aa4 100644 --- a/Services/Storage/WorkStore.cs +++ b/Services/Storage/WorkStore.cs @@ -394,7 +394,7 @@ public async Task> ClaimAsync(string runKey, int max, await FinishAsync(runKey, [new Finish(seq, "Failed", $"Interrupted {attempt - 1} times without completing")], 'R', ct); if (header.StopOnFailure) { - await CancelPendingAsync(runKey, StoppedReason(row.GetString("TaskId")), ct); + await CancelPendingAsync(runKey, StoppedReason(row.GetString("TaskId")), ct: ct); return null; } return await ClaimSequentialAsync(runKey, owner, lease, continuing, othersAreDead, ct); @@ -505,11 +505,14 @@ public sealed record FinishOutcome(RunHeader Header, bool ReachedBarrier, bool C return outcome; } + /// Rows the caller has just read (by seq), used on the first attempt instead of reading each + /// one again; the transaction is still guarded by their ETags, and a retry reads afresh. private async Task FinishChunkAsync(string runKey, IReadOnlyList chunk, char? fromState, - CancellationToken ct) + CancellationToken ct, Dictionary? known = null) { for (var attempt = 0; attempt < ConflictRetries; attempt++) { + if (attempt > 0) known = null; var headerRow = await _store.GetAsync(_work, runKey, HeaderKey, ct); if (headerRow == null) return null; var header = RunHeader.FromRow(headerRow); @@ -535,8 +538,8 @@ public sealed record FinishOutcome(RunHeader Header, bool ReachedBarrier, bool C continue; } - StoreRow? row = null; - foreach (var state in fromState is { } s ? [s] : new[] { 'R', 'P' }) + StoreRow? row = known != null && known.TryGetValue(f.Seq, out var seenRow) ? seenRow : null; + foreach (var state in row != null ? [] : fromState is { } s ? [s] : new[] { 'R', 'P' }) { row = await _store.GetAsync(_work, runKey, Key(state, f.Seq), ct); if (row != null) break; @@ -703,22 +706,60 @@ public async Task AddChildAsync(string parentKey, string childKey, Cancell // ── cancel ── /// Cancel every pending task of a run. Running tasks finish; the barrier then fires as usual. + private readonly ConcurrentDictionary _cancelling = new(StringComparer.Ordinal); + + /// Whether a full cancel of the run is in progress in this process (the scheduler leaves it alone). + public bool IsCancelling(string runKey) => _cancelling.ContainsKey(runKey); + + /// Pages of up to 49 to cancel before returning; the scheduler does one per visit to finish + /// a cancel that was interrupted. A full cancel (no bound) marks the run as being cancelled while it works. public async Task<(int Cancelled, FinishOutcome? Outcome)> CancelPendingAsync(string runKey, - string reason = "Cancelled by user", CancellationToken ct = default) + string reason = "Cancelled by user", int maxPages = int.MaxValue, CancellationToken ct = default) + { + if (maxPages != int.MaxValue) return await CancelPagesAsync(runKey, reason, maxPages, ct); + _cancelling.AddOrUpdate(runKey, 1, (_, n) => n + 1); + try + { + return await CancelPagesAsync(runKey, reason, maxPages, ct); + } + finally + { + if (_cancelling.AddOrUpdate(runKey, 0, (_, n) => n - 1) <= 0) _cancelling.TryRemove(runKey, out _); + } + } + + private async Task<(int Cancelled, FinishOutcome? Outcome)> CancelPagesAsync(string runKey, string reason, int maxPages, + CancellationToken ct) { var cancelled = 0; + var idle = 0; FinishOutcome? outcome = null; - while (true) + for (var pages = 0; pages < maxPages; pages++) { var page = new List(); + var rows = new Dictionary(); await foreach (var r in Range(runKey, 'P', MaxPerTransaction, ct: ct)) - if (SeqOf(r.RowKey) != AggregateSeq) page.Add(new Finish(SeqOf(r.RowKey), "Cancelled", reason)); + { + var seq = SeqOf(r.RowKey); + if (seq == AggregateSeq) continue; + page.Add(new Finish(seq, "Cancelled", reason)); + rows[seq] = r; + } if (page.Count == 0) return (cancelled, outcome); - var result = await FinishAsync(runKey, page, 'P', ct); - if (result == null || result.Applied == 0) return (cancelled, outcome); + var result = await FinishChunkAsync(runKey, page, 'P', ct, rows); + if (result == null) return (cancelled, outcome); + // Nothing applied means another writer cancelled (or claimed) this page first, not that none is left: + // only an empty pending range ends the loop. Bounded, so rows that somehow cannot be cancelled do not spin. + if (result.Applied == 0) + { + if (++idle >= 3) return (cancelled, outcome); + continue; + } + idle = 0; cancelled += result.Applied; outcome = result; } + return (cancelled, outcome); } // ── the instance lock ── diff --git a/perf-harness/scripts/run-e2e-orchestration.ps1 b/perf-harness/scripts/run-e2e-orchestration.ps1 index f4001c3..30dc6c7 100644 --- a/perf-harness/scripts/run-e2e-orchestration.ps1 +++ b/perf-harness/scripts/run-e2e-orchestration.ps1 @@ -20,11 +20,14 @@ # Azurite) on craft:orch-v2-a4d72cf: 2x the median. Short tasks are bounded by the pump (about BgPoolSize claims # per 1s poll, so ~4 tasks/s here), which is what the fan-out numbers measure. $OrchGates = @{ - Fanout1000Sec = 510 # 1,000 no-op tasks + PostExecution, enqueue -> PostExecution ran (median 255s) - ManyRuns300Sec = 160 # 300 single-task runs queued at once, enqueue -> all 300 Done (median 79s) - Ttfs5000Ms = 3800 # 5,000-task run on a warm pump, enqueue -> first task started (median 1.9s) + # Provisional after the pump wake-up and bulk-cancel fixes (2026-10-06, one run: 15.6s / 9s / 1371ms / 238MB / + # 84ms; cancel-5000 measured after the fix). About 2.5x the measured value; re-calibrate from 3 runs. + Fanout1000Sec = 40 # 1,000 no-op tasks + PostExecution, enqueue -> PostExecution ran + ManyRuns300Sec = 25 # 300 single-task runs queued at once, enqueue -> all 300 Done + Ttfs5000Ms = 3500 # 5,000-task run on a warm pump, enqueue -> first task started RssMB = 470 # SUT RSS after the perf runs (median 233MB) - IdleClaimMs = 12000 # run queued into an idle engine -> first start (idle poll cap 10s + slack); not calibrated + IdleClaimMs = 1500 # run queued into an idle engine -> first start (the pump wakes on a new run) + Cancel5000Sec = 30 # CancelRun on a 5,000-task run -> run finished with every pending task cancelled } $OrchSkipped = @{} @@ -550,10 +553,13 @@ Invoke-OrchCheck 'perf' { $W5k = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['tf5k'].tasks -ge 1) { $S['tf5k'] } } 180 100 $Ttfs5k = if ($W5k.ok) { ConvertTo-OrchMs ($W5k.value.minStart - $E5k.enqueueTicks) } else { -1 } Add-Result 'orch-perf' 'ttfs-5000' ($W5k.ok -and $Ttfs5k -le $OrchGates.Ttfs5000Ms) "${Ttfs5k}ms" "enqueue->first task start: 5,000-task run ${Ttfs5k}ms vs 10-task run ${Ttfs10}ms (x$([math]::Round($Ttfs5k / [math]::Max(1, $Ttfs10), 1))); gate $($OrchGates.Ttfs5000Ms)ms" + $CancelSw = [Diagnostics.Stopwatch]::StartNew() $Cancelled = (Invoke-OrchBridge 'cancel' "E2ETf5k-$Id").cancelled - $D = Wait-OrchRunsDone "E2ETf5k-$Id" 1 240 2000 + $D = Wait-OrchRunsDone "E2ETf5k-$Id" 1 240 500 + $CancelSec = [math]::Round($CancelSw.Elapsed.TotalSeconds, 1) $H = (Get-OrchRuns "E2ETf5k-$Id")[0] - Add-Result 'orch-perf' 'cancel-5000' ($D.ok -and $H.status -eq 'CompletedWithErrors' -and $H.done -eq 5000 -and $H.cancelled -gt 0) "$($D.sec)s" "CancelRun returned $Cancelled; run=$($H.status) done=$($H.done)/$($H.total) cancelled=$($H.cancelled)" + # CancelRun reports what the run records, so it must match the header (it once returned 0 while 4,995 were cancelled). + Add-Result 'orch-perf' 'cancel-5000' ($D.ok -and $H.status -eq 'CompletedWithErrors' -and $H.done -eq 5000 -and $H.cancelled -gt 0 -and [int]$Cancelled -eq [int]$H.cancelled -and $CancelSec -le $OrchGates.Cancel5000Sec) "${CancelSec}s" "CancelRun returned $Cancelled; run=$($H.status) done=$($H.done)/$($H.total) cancelled=$($H.cancelled); gate $($OrchGates.Cancel5000Sec)s" $null = Invoke-OrchGet "/API/PerfE2EState?ns=$Ns&wipe=1" Start-Sleep -Seconds 5 @@ -601,4 +607,4 @@ Invoke-OrchCheck 'restart' { Add-Result 'orch-restart' 'postexec-once' ($P.Count -eq 1 -and $P[0].lines -eq $N -and $H.postExecStatus -eq 'Completed') '-' "posts=$($P.Count) lines=$($P[0].lines) postExec=$($H.postExecStatus)" } -Info ("orchestration gates: fanout-1000 <= {0}s, runs-300 <= {1}s, ttfs-5000 <= {2}ms, rss <= {3}MB, idle claim <= {4}ms" -f $OrchGates.Fanout1000Sec, $OrchGates.ManyRuns300Sec, $OrchGates.Ttfs5000Ms, $OrchGates.RssMB, $OrchGates.IdleClaimMs) +Info ("orchestration gates: fanout-1000 <= {0}s, runs-300 <= {1}s, ttfs-5000 <= {2}ms, rss <= {3}MB, idle claim <= {4}ms, cancel-5000 <= {5}s" -f $OrchGates.Fanout1000Sec, $OrchGates.ManyRuns300Sec, $OrchGates.Ttfs5000Ms, $OrchGates.RssMB, $OrchGates.IdleClaimMs, $OrchGates.Cancel5000Sec) diff --git a/tests/Craft.Tests/OrchestrationLifecycleTests.cs b/tests/Craft.Tests/OrchestrationLifecycleTests.cs index 969ff6e..134be91 100644 --- a/tests/Craft.Tests/OrchestrationLifecycleTests.cs +++ b/tests/Craft.Tests/OrchestrationLifecycleTests.cs @@ -167,6 +167,72 @@ public async Task AReadyEntryWhoseRunIsGone_IsDroppedByThePump() Assert.Empty(await h.ReadyNamesAsync()); } + // ── bulk cancel ── + + /// + /// Cancelling a big backlog while the pump is working. The pump once kept claiming the run's tasks during the + /// cancel (each cancelled page woke it), so the two fought over the same rows and the run header: the call took + /// tens of seconds, under-reported, and could give up on lost races. Now the pump leaves a run being + /// cancelled alone, and the cancel reuses the rows it has just read. + /// + [Fact] + public async Task CancellingABigBacklogWithThePumpRunning_CancelsEveryPendingTask_AndReportsTheTrueCount() + { + await using var h = await OrchestrationHarness.CreateAsync(poolSize: 4); + h.Svc.HoldMs = 200; + h.Svc.MarkRecoveryDone(); + await h.Pump.StartAsync(CancellationToken.None); + try + { + Assert.True(await h.Start("Backlog", Batch(5000, "b"))); + Assert.True(await h.DriveUntil(() => Task.FromResult(h.Svc.Started.Count >= 4))); + + var sw = System.Diagnostics.Stopwatch.StartNew(); + var (found, cancelled) = await h.Svc.CancelRunAsync("Backlog"); + sw.Stop(); + + var run = (await h.Store.GetRunByNameAsync("Backlog"))!; + Assert.True(found); + Assert.Equal(run.Cancelled, cancelled); + Assert.Empty(await h.Store.GetTasksAsync(run.RunKey, 'P')); + Assert.True(cancelled >= 5000 - h.Svc.Started.Count - 8, $"cancelled {cancelled}, started {h.Svc.Started.Count}"); + Assert.True(sw.Elapsed < TimeSpan.FromSeconds(20), $"cancelling 5,000 took {sw.Elapsed.TotalSeconds:F1}s"); + } + finally + { + await h.Pump.StopAsync(CancellationToken.None); + } + } + + [Fact] + public async Task ACancelInterruptedAfterItsFlag_IsFinishedByThePump() + { + await using var h = await OrchestrationHarness.CreateAsync(poolSize: 4); + Assert.True(await h.Start("Interrupted", Batch(300, "i"), "Agg")); + var run = (await h.Store.GetRunByNameAsync("Interrupted"))!; + await h.Store.RequestCancelAsync(run.RunKey); // the process died before cancelling anything + + Assert.True(await h.DriveUntilFinished("Interrupted", 30_000)); + var after = (await h.Store.GetRunByNameAsync("Interrupted"))!; + Assert.Equal(300, after.Cancelled); + Assert.Empty(h.Svc.Started); + Assert.Single(h.Svc.PostExecs); + } + + [Fact] + public async Task ABulkCancel_ReusesTheRowsItReads_RatherThanReadingEachAgain() + { + var count = new CountingTableStore(new MemoryTableStore()); + await using var h = await OrchestrationHarness.CreateAsync(tables: count); + Assert.True(await h.Start("Cheap", Batch(4900, "c"))); + count.Reset(); + + Assert.Equal(4900, (await h.Svc.CancelRunAsync("Cheap")).cancelledCount); + + // 100 pages: the header before and after each, never one read per cancelled task. + Assert.InRange(count.For("OrchestratorWork").PointReads, 0, 400); + } + // ── startup and cleanup ── [Fact] From bbdf0bb794cf89872923e3e8997e0a7c48ac178a Mon Sep 17 00:00:00 2001 From: Zacgoose <107489668+Zacgoose@users.noreply.github.com> Date: Tue, 6 Oct 2026 12:03:33 +0800 Subject: [PATCH 17/24] fix(orchestration): never run a claim handed back at shutdown, and log the lock release - WithdrawJob takes a queued job out under the record lock, so it cannot start after being handed back - Report the instance lock release (or why it lapses) at shutdown --- Services/Orchestration/JobManager.cs | 63 ++++++++++++++----- Services/Orchestration/WorkPump.cs | 19 ++++-- Services/Storage/WorkStore.cs | 8 +-- .../OrchestrationLifecycleTests.cs | 25 ++++++++ 4 files changed, 90 insertions(+), 25 deletions(-) diff --git a/Services/Orchestration/JobManager.cs b/Services/Orchestration/JobManager.cs index 06d33d1..c412f66 100644 --- a/Services/Orchestration/JobManager.cs +++ b/Services/Orchestration/JobManager.cs @@ -378,18 +378,26 @@ private async Task RunJobAsync(QueuedJob job, CancellationToken ct) }; opScope = OperationContext.Set(parentInvocation); - // Cancelled after the dispatcher's own check but before the job got going. - if (_cancelledJobIds.TryRemove(job.Record.Id, out var notified)) + // Cancelled or withdrawn after the dispatcher's own check but before the job got going. Under the + // record's lock so WithdrawJob either stops it here or sees it Running, never both. + bool skip, notified; + lock (job.Record) + { + skip = _cancelledJobIds.TryRemove(job.Record.Id, out notified); + if (!skip) + { + job.Record.Status = "Running"; + job.Record.StartedUtc = DateTime.UtcNow; + } + } + if (skip) { if (!notified && job.Descriptor is { } cancelled) NotifyStateWriter(w => w.Cancelled(cancelled), "cancellation", job.Record.Name); return; } - job.Record.Status = "Running"; - job.Record.StartedUtc = DateTime.UtcNow; - - var queueTime = job.Record.StartedUtc.Value - job.Record.QueuedUtc; + var queueTime = job.Record.StartedUtc!.Value - job.Record.QueuedUtc; if (queueTime.TotalSeconds > 1) { _logger.LogInformation( @@ -556,19 +564,22 @@ public override void Dispose() public bool CancelJob(string jobId) { if (!_jobs.TryGetValue(jobId, out var record)) return false; - if (record.Status != "Queued") return false; - - record.Status = "Cancelled"; - record.CompletedUtc = DateTime.UtcNow; - record.LastError = "Cancelled by user"; - - // Under the queue lock, so the dispatcher either still finds the entry (and we tell the state writer - // here) or has already dequeued it (and tells it when it skips the job). JobDescriptor? descriptor; - lock (_queueLock) + lock (record) { - descriptor = FindLiveEntry(record)?.Descriptor; - _cancelledJobIds[jobId] = descriptor != null; + if (record.Status != "Queued") return false; + + record.Status = "Cancelled"; + record.CompletedUtc = DateTime.UtcNow; + record.LastError = "Cancelled by user"; + + // Under the queue lock, so the dispatcher either still finds the entry (and we tell the state writer + // here) or has already dequeued it (and tells it when it skips the job). + lock (_queueLock) + { + descriptor = FindLiveEntry(record)?.Descriptor; + _cancelledJobIds[jobId] = descriptor != null; + } } if (descriptor is { } d) NotifyStateWriter(w => w.Cancelled(d), "cancellation", record.Name); @@ -577,6 +588,24 @@ public bool CancelJob(string jobId) return true; } + /// + /// Take a queued job back without running it and without recording an outcome, so its owner can hand the + /// work to another process. False once it has started. + /// + public bool WithdrawJob(string jobId) + { + if (!_jobs.TryGetValue(jobId, out var record)) return false; + lock (record) + { + if (record.Status != "Queued") return false; + record.Status = "Cancelled"; + record.CompletedUtc = DateTime.UtcNow; + record.LastError = "Withdrawn at shutdown"; + _cancelledJobIds[jobId] = true; + } + return true; + } + /// Cancel all queued jobs in a run group. public int CancelRun(string runName) { diff --git a/Services/Orchestration/WorkPump.cs b/Services/Orchestration/WorkPump.cs index dc7dd0e..62f2a70 100644 --- a/Services/Orchestration/WorkPump.cs +++ b/Services/Orchestration/WorkPump.cs @@ -381,9 +381,10 @@ internal async Task RenewAsync(CancellationToken ct) public override async Task StopAsync(CancellationToken cancellationToken) { await base.StopAsync(cancellationToken); + _logger.LogInformation("[WorkPump] Stopping with {Count} claim(s) in flight; lock held={Held}", _inFlight.Count, _holdsLock); foreach (var (id, v) in _inFlight.ToList()) { - if (_jobs.GetJobs().FirstOrDefault(j => j.Id == id) is not { Status: "Queued" }) continue; + if (!_jobs.WithdrawJob(id)) continue; try { await _store.ReleaseAsync(v.Claim.RunKey, v.Claim.Seq, _owner, refundAttempt: true, cancellationToken); } catch (Exception ex) { _logger.LogDebug(ex, "[WorkPump] Could not release {Job} on shutdown", id); } } @@ -399,9 +400,19 @@ public override async Task StopAsync(CancellationToken cancellationToken) await Task.Delay(250, cancellationToken); } } - catch (OperationCanceledException) { return; } - try { await _store.ReleaseInstanceLockAsync(_owner, cancellationToken); } - catch (Exception ex) { _logger.LogDebug(ex, "[WorkPump] Could not release the instance lock on shutdown"); } + catch (OperationCanceledException) + { + _logger.LogWarning("[WorkPump] Shutdown cut short with tasks still running; the instance lock lapses at its lease"); + return; + } + try + { + if (await _store.ReleaseInstanceLockAsync(_owner, cancellationToken)) + _logger.LogInformation("[WorkPump] Released the instance lock"); + else + _logger.LogWarning("[WorkPump] Could not release the instance lock on shutdown; it lapses at its lease"); + } + catch (Exception ex) { _logger.LogWarning(ex, "[WorkPump] Could not release the instance lock on shutdown; it lapses at its lease"); } _holdsLock = false; } } diff --git a/Services/Storage/WorkStore.cs b/Services/Storage/WorkStore.cs index 2043aa4..3139d59 100644 --- a/Services/Storage/WorkStore.cs +++ b/Services/Storage/WorkStore.cs @@ -805,12 +805,12 @@ public sealed record InstanceLock(string Owner, DateTimeOffset LeaseUntil, DateT return ok ? (true, new InstanceLock(owner, now.Add(lease), acquired)) : (false, await GetInstanceLockAsync(ct)); } - /// Give the instance lock up, if still holds it, so a successor starts at once. - public async Task ReleaseInstanceLockAsync(string owner, CancellationToken ct = default) + /// Give the instance lock up, if still holds it, so a successor starts at once. + /// False when still held it and the delete lost a race. + public async Task ReleaseInstanceLockAsync(string owner, CancellationToken ct = default) { var row = await _store.GetAsync(_names, LockPartition, LockKey, ct); - if (row?.GetString("Owner") == owner) - await _store.TrySubmitAsync(_names, LockPartition, [StoreOp.Delete(row)], ct); + return row?.GetString("Owner") != owner || await _store.TrySubmitAsync(_names, LockPartition, [StoreOp.Delete(row)], ct); } // ── index repair ── diff --git a/tests/Craft.Tests/OrchestrationLifecycleTests.cs b/tests/Craft.Tests/OrchestrationLifecycleTests.cs index 134be91..c7d8007 100644 --- a/tests/Craft.Tests/OrchestrationLifecycleTests.cs +++ b/tests/Craft.Tests/OrchestrationLifecycleTests.cs @@ -90,6 +90,31 @@ public async Task OnShutdown_ClaimsThatNeverStarted_AreHandedBack_WithTheirAttem Assert.All(pending, t => Assert.Equal(0, t.Attempt)); } + /// + /// The job manager stops after the pump, so it can still dispatch during shutdown. A claim handed back must + /// also be taken out of its queue, or this process runs the task while its successor runs it again. + /// + [Fact] + public async Task OnShutdown_AHandedBackClaim_IsNeverRunByThisProcess() + { + var (store, pump, jobs) = NewIdlePump(); + var ran = 0; + jobs.SetWorkResolver((_, _) => Task.FromResult?>(_ => + { + Interlocked.Increment(ref ran); + return Task.CompletedTask; + })); + await CreateAsync(store, "Handback", 4); + Assert.Equal(4, await pump.RefillAsync(CancellationToken.None)); + + await pump.StopAsync(CancellationToken.None); + _ = Task.Run(() => jobs.StartAsync(CancellationToken.None)); + await Task.Delay(500); + + Assert.Equal(0, ran); + await jobs.StopAsync(CancellationToken.None); + } + /// /// Depending on timing, the cancel lands while the job is still queued or just after the dispatcher has /// dequeued it. The second case once dropped the state-writer notification: the claim was never finished, From f584d4b307e0e834c1a76522a02c53e59bce7585 Mon Sep 17 00:00:00 2001 From: Zacgoose <107489668+Zacgoose@users.noreply.github.com> Date: Tue, 6 Oct 2026 12:03:33 +0800 Subject: [PATCH 18/24] test(perf): add orchestration benchmark for comparing engines on Azurite or a storage account --- perf-harness/docker-compose.bench.yml | 7 + perf-harness/scripts/run-orch-bench.ps1 | 238 ++++++++++++++++++++++++ 2 files changed, 245 insertions(+) create mode 100644 perf-harness/docker-compose.bench.yml create mode 100644 perf-harness/scripts/run-orch-bench.ps1 diff --git a/perf-harness/docker-compose.bench.yml b/perf-harness/docker-compose.bench.yml new file mode 100644 index 0000000..1864c53 --- /dev/null +++ b/perf-harness/docker-compose.bench.yml @@ -0,0 +1,7 @@ +# Override for scripts/run-orch-bench.ps1: point the SUT at another storage account and an isolated table prefix, +# so one engine's run never reads another's tables. Layered on docker-compose.e2e-azure.yml. +services: + sut: + environment: + - AzureWebJobsStorage=${BENCH_STORAGE} + - App__Orchestrator__TablePrefix=${BENCH_PREFIX} diff --git a/perf-harness/scripts/run-orch-bench.ps1 b/perf-harness/scripts/run-orch-bench.ps1 new file mode 100644 index 0000000..f361982 --- /dev/null +++ b/perf-harness/scripts/run-orch-bench.ps1 @@ -0,0 +1,238 @@ +<# +.SYNOPSIS + Orchestration benchmark: the same workloads against any Craft image, on Azurite or a real storage account, so + engines can be compared like for like. + +.DESCRIPTION + Brings up docker-compose.e2e-azure.yml (Azurite + the SUT with PerfApi, BgPoolSize=4, 2 CPUs), optionally pointing + the SUT at a real account (-Storage azure, connection from CRAFT_TEST_TABLE_CONNECTION) under a unique table + prefix, runs the scenarios below one after another, writes one JSON result file, and tears the stack down. + Uses only Start-CraftOrchestrator, the PerfE2E recording task/PostExecution and WorkerMetricsBridge.CancelRun, + so it runs unchanged on engines before and after the storage-first redesign. Timings come from task start/end + ticks recorded inside the SUT. + +.EXAMPLE + pwsh scripts/run-orch-bench.ps1 -SutImage craft:pre-v2-c98a091 -Storage azurite -Label pre-azurite +#> +[CmdletBinding()] +param( + [Parameter(Mandatory)][string]$SutImage, + [ValidateSet('azurite', 'azure')][string]$Storage = 'azurite', + [string]$Label = 'bench', + [int]$Port = 5399, + [string]$OutDir = (Join-Path ([System.IO.Path]::GetTempPath()) 'craft-bench'), + [string[]]$Only +) + +$ErrorActionPreference = 'Stop' +# pwsh -File passes "a,b" as one string. +$Only = @($Only | ForEach-Object { $_ -split ',' } | Where-Object { $_ }) +$here = Split-Path -Parent $MyInvocation.MyCommand.Path +$root = Split-Path -Parent $here +$compose = @('-f', (Join-Path $root 'docker-compose.e2e-azure.yml'), '-f', (Join-Path $root 'docker-compose.bench.yml')) +$base = "http://127.0.0.1:$Port" +$Azurite = 'DefaultEndpointsProtocol=http;AccountName=devstoreaccount1;AccountKey=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==;BlobEndpoint=http://azurite:10000/devstoreaccount1;QueueEndpoint=http://azurite:10001/devstoreaccount1;TableEndpoint=http://azurite:10002/devstoreaccount1;' + +$Prefix = 'Bench' + [guid]::NewGuid().ToString('N').Substring(0, 8) +$env:SUT_IMAGE = $SutImage +$env:SUT_PORT = "$Port" +$env:BENCH_PREFIX = $Prefix +$env:BENCH_STORAGE = if ($Storage -eq 'azure') { + if (-not $env:CRAFT_TEST_TABLE_CONNECTION) { throw 'CRAFT_TEST_TABLE_CONNECTION is not set' } + $env:CRAFT_TEST_TABLE_CONNECTION +} else { $Azurite } + +$Results = [ordered]@{ + label = $Label; image = $SutImage; storage = $Storage; prefix = $Prefix + account = if ($Storage -eq 'azure') { [regex]::Match($env:CRAFT_TEST_TABLE_CONNECTION, 'AccountName=([^;]+)').Groups[1].Value } else { 'azurite' } + startedUtc = [DateTime]::UtcNow.ToString('o'); scenarios = [ordered]@{} +} + +function Info($m) { Write-Host "[bench $Label] $m" -ForegroundColor Cyan } +function Get-Api([string]$Path, [int]$TimeoutSec = 60) { Invoke-RestMethod "$base$Path" -TimeoutSec $TimeoutSec } +function New-Id { [guid]::NewGuid().ToString('N').Substring(0, 6) } +function Ms([long]$Ticks) { [math]::Round($Ticks / 10000.0) } + +function Start-Runs([string]$Ns, $Runs, [string]$Sink = 'cache', [int]$TimeoutSec = 600) { + $Body = @{ ns = $Ns; sink = $Sink; runs = @($Runs) } | ConvertTo-Json -Depth 30 -Compress + $Sw = [Diagnostics.Stopwatch]::StartNew() + $R = Invoke-RestMethod "$base/API/PerfE2EStart" -Method Post -ContentType 'application/json' -Body $Body -TimeoutSec $TimeoutSec + return [pscustomobject]@{ runs = @($R.runs); callMs = $Sw.ElapsedMilliseconds } +} + +function Get-Summary([string]$Ns, [string]$Source = 'cache') { + $Map = @{} + foreach ($R in @((Get-Api "/API/PerfE2EState?ns=$Ns&source=$Source&summary=1").runs)) { $Map[$R.run] = $R } + return $Map +} + +function Wait-Until([scriptblock]$Condition, [int]$TimeoutSec, [int]$IntervalMs = 250) { + $Sw = [Diagnostics.Stopwatch]::StartNew() + while ($Sw.Elapsed.TotalSeconds -lt $TimeoutSec) { + $V = try { & $Condition } catch { $null } + if ($V) { return [pscustomobject]@{ ok = $true; value = $V; sec = $Sw.Elapsed.TotalSeconds } } + Start-Sleep -Milliseconds $IntervalMs + } + return [pscustomobject]@{ ok = $false; value = $null; sec = $Sw.Elapsed.TotalSeconds } +} + +function Wait-Ready([int]$TimeoutSec = 240) { + $W = Wait-Until { $H = Get-Api '/healthz' 10; if ($H.status -eq 'ready') { $true } } $TimeoutSec 1000 + if (-not $W.ok) { throw 'SUT never became ready' } +} + +function Add-Scenario([string]$Name, [hashtable]$Data) { + $Results.scenarios[$Name] = $Data + Info ("{0,-22} {1}" -f $Name, (($Data.GetEnumerator() | ForEach-Object { "$($_.Key)=$($_.Value)" }) -join ' ')) +} + +function Invoke-Scenario([string]$Name, [scriptblock]$Body) { + if ($Only -and $Name -notin $Only) { return } + try { & $Body } + catch { Add-Scenario $Name @{ error = $_.Exception.Message } } +} + +Info "compose up: $SutImage on $Storage (prefix $Prefix)" +docker compose @compose up -d | Out-Host +if ($LASTEXITCODE -ne 0) { throw 'compose up failed' } + +try { + Wait-Ready + Start-Sleep -Seconds 5 + + # 1. Throughput of short tasks, one run, with an aggregation. + Invoke-Scenario 'fanout-1000' { + $Ns = "f1k-$(New-Id)" + $S = Start-Runs $Ns @(@{ name = "BenchFan-$Ns"; label = 'fan'; tasks = 1000; post = @{ marker = 'm' } }) + $T0 = $S.runs[0].enqueueTicks + $W = Wait-Until { $M = Get-Summary $Ns; if ($M['fan'].posts -ge 1) { $M['fan'] } } 1800 500 + $Rows = @((Get-Api "/API/PerfE2EState?ns=$Ns").rows) + $Post = @($Rows.Where({ $_.kind -eq 'P' }))[0] + $Sec = if ($Post) { (Ms ($Post.ticks - $T0)) / 1000.0 } else { -1 } + Add-Scenario 'fanout-1000' @{ ok = $W.ok; createMs = $S.callMs; endToEndSec = [math]::Round($Sec, 1); tasksPerSec = [math]::Round(1000 / [math]::Max(0.1, $Sec), 1); ran = $W.value.tasks } + } + + # 2. Efficiency with real work: 200 tasks of 250 ms on 4 workers (ideal 12.5 s). + Invoke-Scenario 'fanout-200x250ms' { + $Ns = "f200-$(New-Id)" + $S = Start-Runs $Ns @(@{ name = "BenchWork-$Ns"; label = 'w'; tasks = 200; task = @{ holdms = 250 }; post = @{ marker = 'm' } }) + $T0 = $S.runs[0].enqueueTicks + $W = Wait-Until { $M = Get-Summary $Ns; if ($M['w'].posts -ge 1) { $M['w'] } } 1800 500 + $Sec = if ($W.ok) { (Ms ($W.value.maxEnd - $T0)) / 1000.0 } else { -1 } + Add-Scenario 'fanout-200x250ms' @{ ok = $W.ok; tasksDoneSec = [math]::Round($Sec, 1); efficiencyPct = [math]::Round(12.5 / [math]::Max(0.1, $Sec) * 100); idealSec = 12.5 } + } + + # 3./4. Many small runs queued at once (one task each, no aggregation). + foreach ($N in 300, 1000) { + Invoke-Scenario "runs-$N" { + $Ns = "r$N-$(New-Id)" + $Specs = @(for ($I = 0; $I -lt $N; $I++) { @{ name = "BenchMany-$Ns-$I"; label = "r$I"; tasks = 1 } }) + $S = Start-Runs $Ns $Specs 'cache' 1800 + $T0 = $S.runs[0].enqueueTicks + $W = Wait-Until { $M = Get-Summary $Ns; $Done = @($M.Values.Where({ $_.ended -ge 1 })).Count; if ($Done -ge $N) { $M } } 1800 1000 + $Last = if ($W.ok) { ($W.value.Values | Measure-Object -Property maxEnd -Maximum).Maximum } else { 0 } + $Sec = if ($W.ok) { (Ms ($Last - $T0)) / 1000.0 } else { -1 } + Add-Scenario "runs-$N" @{ ok = $W.ok; createMs = $S.callMs; allDoneSec = [math]::Round($Sec, 1); runsPerSec = [math]::Round($N / [math]::Max(0.1, $Sec), 1) } + } + } + + # 5. Creating a big run and how soon its first task starts. + Invoke-Scenario 'ttfs-5000' { + $Ns = "t5k-$(New-Id)" + $S = Start-Runs $Ns @(@{ name = "BenchBig-$Ns"; label = 'big'; tasks = 5000; task = @{ holdms = 50 } }) 'cache' 900 + $T0 = $S.runs[0].enqueueTicks + $W = Wait-Until { $M = Get-Summary $Ns; if ($M['big'].tasks -ge 1) { $M['big'] } } 600 50 + $First = if ($W.ok) { Ms ($W.value.minStart - $T0) } else { -1 } + $Cancel = Get-Api "/API/PerfE2EBridge?op=cancel&name=BenchBig-$Ns" 900 + Add-Scenario 'ttfs-5000' @{ ok = $W.ok; createMs = $S.callMs; firstStartMs = $First } + Start-Sleep -Seconds 5 + } + + # 6. An idle engine picking up a new run. + Invoke-Scenario 'idle-claim' { + Start-Sleep -Seconds 20 + $Ns = "idle-$(New-Id)" + $S = Start-Runs $Ns @(@{ name = "BenchIdle-$Ns"; label = 'i'; tasks = 1 }) + $T0 = $S.runs[0].enqueueTicks + $W = Wait-Until { $M = Get-Summary $Ns; if ($M['i'].tasks -ge 1) { $M['i'] } } 120 20 + Add-Scenario 'idle-claim' @{ ok = $W.ok; firstStartMs = $(if ($W.ok) { Ms ($W.value.minStart - $T0) } else { -1 }) } + } + + # 7. Per-step overhead of a sequential run. + Invoke-Scenario 'sequential-50' { + $Ns = "seq-$(New-Id)" + $S = Start-Runs $Ns @(@{ name = "BenchSeq-$Ns"; label = 's'; tasks = 50; Sequential = $true; post = @{ marker = 'm' } }) + $T0 = $S.runs[0].enqueueTicks + $W = Wait-Until { $M = Get-Summary $Ns; if ($M['s'].posts -ge 1) { $M['s'] } } 900 250 + $Sec = if ($W.ok) { (Ms ($W.value.maxEnd - $T0)) / 1000.0 } else { -1 } + Add-Scenario 'sequential-50' @{ ok = $W.ok; stepsDoneSec = [math]::Round($Sec, 1); msPerStep = [math]::Round($Sec * 1000 / 50) } + } + + # 8. A high-priority run arriving behind a big backlog in a lower band. + Invoke-Scenario 'priority-jump' { + $Ns = "pj-$(New-Id)" + $null = Start-Runs $Ns @(@{ name = "BenchBacklog-$Ns"; label = 'bl'; tasks = 3000; task = @{ holdms = 100 }; Priority = 6 }) 'cache' 900 + $null = Wait-Until { $M = Get-Summary $Ns; if ($M['bl'].tasks -ge 8) { $true } } 300 100 + $S = Start-Runs $Ns @(@{ name = "BenchUrgent-$Ns"; label = 'u'; tasks = 4; Priority = 1 }) + $T0 = $S.runs[0].enqueueTicks + $W = Wait-Until { $M = Get-Summary $Ns; if ($M['u'].ended -ge 4) { $M['u'] } } 600 50 + $null = Get-Api "/API/PerfE2EBridge?op=cancel&name=BenchBacklog-$Ns" 900 + Add-Scenario 'priority-jump' @{ ok = $W.ok; firstStartMs = $(if ($W.ok) { Ms ($W.value.minStart - $T0) } else { -1 }); allDoneMs = $(if ($W.ok) { Ms ($W.value.maxEnd - $T0) } else { -1 }) } + Start-Sleep -Seconds 5 + } + + # 9. Cancelling a large backlog while it runs, to the run finalising (its aggregation running). + Invoke-Scenario 'cancel-5000' { + $Ns = "cx-$(New-Id)" + $null = Start-Runs $Ns @(@{ name = "BenchCancel-$Ns"; label = 'c'; tasks = 5000; task = @{ holdms = 200 }; post = @{ marker = 'm' } }) 'cache' 900 + $null = Wait-Until { $M = Get-Summary $Ns; if ($M['c'].tasks -ge 8) { $true } } 300 100 + $Sw = [Diagnostics.Stopwatch]::StartNew() + $Cancel = Get-Api "/API/PerfE2EBridge?op=cancel&name=BenchCancel-$Ns" 1800 + $CallMs = $Sw.ElapsedMilliseconds + $W = Wait-Until { $M = Get-Summary $Ns; if ($M['c'].posts -ge 1) { $M['c'] } } 1800 250 + Add-Scenario 'cancel-5000' @{ ok = $W.ok; callMs = $CallMs; finalisedSec = [math]::Round($Sw.Elapsed.TotalSeconds, 1); reported = $Cancel.cancelled; tasksRan = $W.value.tasks } + } + + # 10. Memory after the work above. + Invoke-Scenario 'memory' { + $M = Get-Api '/API/PerfE2EBridge?op=summary' + Add-Scenario 'memory' @{ rssMB = $M.rssMB; heapMB = $M.heapMB; committedMB = $M.committedMB } + } + + # 11./12. Recovery mid-run (200 x 1 s tasks): a graceful recycle (SIGTERM, 30 s grace) and a crash (SIGKILL). + # Time from the stop to every task done and the aggregation run; executions > 200 means tasks ran twice. + foreach ($Mode in 'graceful', 'crash') { + Invoke-Scenario "restart-$Mode" { + $Ns = "rs-$(New-Id)" + $null = Start-Runs $Ns @(@{ name = "BenchRestart-$Ns"; label = 'rs'; tasks = 200; task = @{ holdms = 1000 }; post = @{ marker = 'm' } }) 'table' + $null = Wait-Until { $M = Get-Summary $Ns 'table'; if ($M['rs'].tasks -ge 20) { $true } } 300 500 + $Sw = [Diagnostics.Stopwatch]::StartNew() + if ($Mode -eq 'graceful') { docker stop -t 30 craft-e2e-az-sut | Out-Null } else { docker kill craft-e2e-az-sut | Out-Null } + $Exit = docker inspect craft-e2e-az-sut --format '{{.State.ExitCode}}' + $StopSec = $Sw.Elapsed.TotalSeconds + docker start craft-e2e-az-sut | Out-Null + Wait-Ready + $ReadySec = $Sw.Elapsed.TotalSeconds + $W = Wait-Until { $M = Get-Summary $Ns 'table'; if ($M['rs'].posts -ge 1) { $M['rs'] } } 1800 2000 + $Rows = @((Get-Api "/API/PerfE2EState?ns=$Ns&source=table" 120).rows.Where({ $_.kind -eq 'T' })) + $Distinct = @($Rows | Group-Object idx).Count + Add-Scenario "restart-$Mode" @{ ok = $W.ok; stopSec = [math]::Round($StopSec, 1); exitCode = $Exit; readySec = [math]::Round($ReadySec, 1); recoveredSec = [math]::Round($Sw.Elapsed.TotalSeconds, 1); distinctTasks = $Distinct; executions = $Rows.Count } + } + } +} +finally { + $Results.endedUtc = [DateTime]::UtcNow.ToString('o') + $null = New-Item -ItemType Directory -Force -Path $OutDir + $File = Join-Path $OutDir "$Label.json" + $Results | ConvertTo-Json -Depth 10 | Set-Content -Path $File -Encoding utf8 + Info "results: $File" + docker compose @compose logs --no-color sut 2>$null | Set-Content -Path (Join-Path $OutDir "$Label.sut.log") -Encoding utf8 + docker compose @compose down -v | Out-Null + if ($Storage -eq 'azure') { + # Only this run's tables: every one starts with its unique prefix. The connection string stays in the environment. + Info "dropping $Prefix* tables from $($Results.account)" + $Tables = @(az storage table list --connection-string $env:CRAFT_TEST_TABLE_CONNECTION --query "[?starts_with(name, '$Prefix')].name" -o tsv 2>$null) + foreach ($T in $Tables) { if ($T) { az storage table delete --name $T --connection-string $env:CRAFT_TEST_TABLE_CONNECTION -o none 2>$null } } + Info "dropped $($Tables.Count) table(s)" + } +} From f400c8747064993c3e5579e3b515f77f957fa0c7 Mon Sep 17 00:00:00 2001 From: Zacgoose <107489668+Zacgoose@users.noreply.github.com> Date: Tue, 6 Oct 2026 12:47:55 +0800 Subject: [PATCH 19/24] perf(orchestration): finish a sequential step and claim the next in one transaction - Header, step and next pending row read together; the aggregation is claimed directly at the barrier - Falls back to the batched finish and a separate claim if the combined write fails - e2e gate on per-step cost --- Services/Orchestration/OrchestratorService.cs | 36 ++++++-- Services/Storage/WorkStore.cs | 87 +++++++++++++++++- .../scripts/run-e2e-orchestration.ps1 | 11 ++- tests/Craft.Tests/OrchestrationCostTests.cs | 27 ++++++ tests/Craft.Tests/OrchestrationModeTests.cs | 39 ++++++++ tests/Craft.Tests/WorkStoreTests.cs | 89 ++++++++++++++++++- 6 files changed, 278 insertions(+), 11 deletions(-) diff --git a/Services/Orchestration/OrchestratorService.cs b/Services/Orchestration/OrchestratorService.cs index 19c038a..b0b4278 100644 --- a/Services/Orchestration/OrchestratorService.cs +++ b/Services/Orchestration/OrchestratorService.cs @@ -496,12 +496,15 @@ private Func BuildSequentialRunWork(RunHeader header, J PowerShellWorker? worker = null; var faulted = false; var step = new WorkStore.ClaimedTask(header.RunKey, first.Seq, first.TaskId, first.Attempt); + RunHeader? known = null; + Dictionary? nextPayload = null; try { worker = CheckoutSequentialWorker(jobCt); while (true) { - var current = await _store.GetRunAsync(header.RunKey, jobCt); + var current = known ?? await _store.GetRunAsync(header.RunKey, jobCt); + known = null; if (current == null || current.IsFinished) break; if (current.CancelRequested) { @@ -516,14 +519,16 @@ private Func BuildSequentialRunWork(RunHeader header, J break; } - var parameters = await _store.GetPayloadAsync(header.RunKey, step.Seq, jobCt); + var parameters = nextPayload ?? await _store.GetPayloadAsync(header.RunKey, step.Seq, jobCt); + nextPayload = null; + WorkStore.Finish finish; try { if (parameters == null) throw new InvalidOperationException("The task's payload row is missing"); var output = await RunScriptAsync(taskPath, TaskInvocation(parameters), current.HasPostExec, worker); if (current.HasPostExec && !string.IsNullOrEmpty(output)) await _results.StoreResultAsync(header.RunKey, step.TaskId, output); - await FinishAsync(header, step.Seq, "Completed"); + finish = new WorkStore.Finish(step.Seq, "Completed", Owner: Owner); } catch (OperationCanceledException) when (jobCt.IsCancellationRequested) { @@ -531,18 +536,39 @@ private Func BuildSequentialRunWork(RunHeader header, J } catch (Exception ex) { - await FinishAsync(header, step.Seq, "Failed", ex.Message); if (current.StopOnFailure) { + await FinishAsync(header, step.Seq, "Failed", ex.Message); _logger.LogError(ex, "[Scheduler] Sequential task failed: {TaskId} — stopping the run", step.TaskId); await _store.CancelPendingAsync(header.RunKey, WorkStore.StoppedReason(step.TaskId), ct: jobCt); break; } _logger.LogError(ex, "[Scheduler] Sequential task failed: {TaskId} — continuing with the next step", step.TaskId); + finish = new WorkStore.Finish(step.Seq, "Failed", ex.Message, Owner); } - if (await _store.ClaimSequentialAsync(header.RunKey, Owner, Lease, continuing: true, ct: jobCt) is not { } next) break; + // Finish this step and claim the next in one transaction; if that cannot be written, the batcher + // keeps retrying the finish and the next step is claimed on its own. + WorkStore.StepResult result; + try { result = await _store.FinishStepAsync(header.RunKey, finish, Owner, Lease, jobCt); } + catch (Exception ex) when (ex is not OperationCanceledException) + { + _logger.LogWarning(ex, "[Scheduler] Could not finish step {TaskId} of {Run} with its successor; recording it on its own", + step.TaskId, header.Name); + await _finisher.FinishAsync(header.RunKey, finish); + if (await _store.ClaimSequentialAsync(header.RunKey, Owner, Lease, continuing: true, ct: jobCt) is not { } claimed) break; + step = claimed; + continue; + } + if (result.Next is not { } next) + { + if (result.Outcome?.Header is { CancelRequested: true, IsFinished: false }) + await _store.CancelPendingAsync(header.RunKey, ct: jobCt); + break; + } step = next; + known = result.Outcome?.Header; + nextPayload = result.Payload; } } catch (OperationCanceledException) when (jobCt.IsCancellationRequested) diff --git a/Services/Storage/WorkStore.cs b/Services/Storage/WorkStore.cs index 3139d59..f894864 100644 --- a/Services/Storage/WorkStore.cs +++ b/Services/Storage/WorkStore.cs @@ -505,15 +505,63 @@ public sealed record FinishOutcome(RunHeader Header, bool ReachedBarrier, bool C return outcome; } + /// What did: the finish, the step it claimed next (null when there is + /// none, the run was asked to cancel, or another driver holds it) and that step's payload when it could be read. + public sealed record StepResult(FinishOutcome? Outcome, ClaimedTask? Next, Dictionary? Payload); + + private sealed class StepClaim(string owner, TimeSpan lease) + { + public string Owner { get; } = owner; + public TimeSpan Lease { get; } = lease; + public ClaimedTask? Claimed { get; set; } + } + + /// + /// Finish a sequential run's current step and claim its next one in the same transaction, renewing the + /// driver lease: the next pending step, or the aggregation task when this step completes the run's tasks. + /// One round trip of concurrent reads, one transaction, then the Ready update alongside the next payload read. + /// + public async Task FinishStepAsync(string runKey, Finish finish, string owner, TimeSpan lease, + CancellationToken ct = default) + { + var claim = new StepClaim(owner, lease); + Task?>? payload = null; + var outcome = await FinishChunkAsync(runKey, [finish], 'R', ct, claim: claim, + afterSubmit: () => payload = claim.Claimed is { Seq: not AggregateSeq } next ? TryGetPayloadAsync(runKey, next.Seq, ct) : null); + return new StepResult(outcome, claim.Claimed, payload == null ? null : await payload); + } + + private async Task?> TryGetPayloadAsync(string runKey, int seq, CancellationToken ct) + { + try { return await GetPayloadAsync(runKey, seq, ct); } + catch (Exception ex) when (ex is not OperationCanceledException) { return null; } + } + + private async Task FirstAsync(char state, string runKey, CancellationToken ct) + { + await foreach (var r in Range(runKey, state, 1, ct)) return r; + return null; + } + /// Rows the caller has just read (by seq), used on the first attempt instead of reading each /// one again; the transaction is still guarded by their ETags, and a retry reads afresh. private async Task FinishChunkAsync(string runKey, IReadOnlyList chunk, char? fromState, - CancellationToken ct, Dictionary? known = null) + CancellationToken ct, Dictionary? known = null, StepClaim? claim = null, Action? afterSubmit = null) { for (var attempt = 0; attempt < ConflictRetries; attempt++) { if (attempt > 0) known = null; - var headerRow = await _store.GetAsync(_work, runKey, HeaderKey, ct); + StoreRow? headerRow, nextRow = null; + if (claim != null) + { + var headerRead = _store.GetAsync(_work, runKey, HeaderKey, ct); + var stepRead = _store.GetAsync(_work, runKey, Key('R', chunk[0].Seq), ct); + var nextRead = FirstAsync('P', runKey, ct); + await Task.WhenAll(headerRead, stepRead, nextRead); + (headerRow, nextRow) = (headerRead.Result, nextRead.Result); + known = stepRead.Result is { } stepRow ? new() { [chunk[0].Seq] = stepRow } : []; + } + else headerRow = await _store.GetAsync(_work, runKey, HeaderKey, ct); if (headerRow == null) return null; var header = RunHeader.FromRow(headerRow); var ops = new List(); @@ -575,6 +623,12 @@ public sealed record FinishOutcome(RunHeader Header, bool ReachedBarrier, bool C if (applied == 0) return new FinishOutcome(header, false, false, 0); + var now = DateTimeOffset.UtcNow; + var claimable = claim != null && !header.CancelRequested + && !(header.DriverOwner is { } driver && driver != claim.Owner && header.DriverLease > now); + var until = now.Add(claim?.Lease ?? TimeSpan.Zero); + ClaimedTask? next = null; + if (!completed && header.Phase == RunPhase.Tasks && header.Done >= header.Total) { barrier = true; @@ -582,10 +636,29 @@ public sealed record FinishOutcome(RunHeader Header, bool ReachedBarrier, bool C { header.Phase = RunPhase.Aggregate; header.PostExecStatus = "Pending"; - ops.Add(StoreOp.Insert(PendingRow(runKey, AggregateSeq, "PostExecution", 0))); + if (claimable) + { + ops.Add(StoreOp.Insert(RunningRow(runKey, AggregateSeq, "PostExecution", 1, claim!.Owner, until))); + next = new ClaimedTask(runKey, AggregateSeq, "PostExecution", 1); + } + else ops.Add(StoreOp.Insert(PendingRow(runKey, AggregateSeq, "PostExecution", 0))); } else completed = true; } + else if (claimable && !completed && nextRow != null) + { + var seq = SeqOf(nextRow.RowKey); + var taskId = nextRow.GetString("TaskId")!; + var nextAttempt = (nextRow.GetInt32("Attempt") ?? 0) + 1; + ops.Add(StoreOp.Delete(nextRow)); + ops.Add(StoreOp.Insert(RunningRow(runKey, seq, taskId, nextAttempt, claim!.Owner, until))); + next = new ClaimedTask(runKey, seq, taskId, nextAttempt); + } + if (next != null) + { + header.DriverOwner = claim!.Owner; + header.DriverLease = until; + } if (completed) { header.Phase = RunPhase.Done; @@ -597,7 +670,13 @@ public sealed record FinishOutcome(RunHeader Header, bool ReachedBarrier, bool C await _rate.TakeAsync(runKey, ops.Count, ct); if (await _store.TrySubmitAsync(_work, runKey, ops, ct)) { - var after = (await GetRunAsync(runKey, ct)) ?? header; + if (claim != null) + { + claim.Claimed = next; + afterSubmit?.Invoke(); + } + // The step path wrote the header under its ETag, so what it wrote is the state after it. + var after = claim != null ? header : (await GetRunAsync(runKey, ct)) ?? header; if (completed) await RetireAsync(after, ct); else await IndexAsync("update its Ready counts", runKey, () => PublishReadyAsync(after, ct), ct); Changed(runKey); diff --git a/perf-harness/scripts/run-e2e-orchestration.ps1 b/perf-harness/scripts/run-e2e-orchestration.ps1 index 30dc6c7..daccf71 100644 --- a/perf-harness/scripts/run-e2e-orchestration.ps1 +++ b/perf-harness/scripts/run-e2e-orchestration.ps1 @@ -28,6 +28,7 @@ $OrchGates = @{ RssMB = 470 # SUT RSS after the perf runs (median 233MB) IdleClaimMs = 1500 # run queued into an idle engine -> first start (the pump wakes on a new run) Cancel5000Sec = 30 # CancelRun on a 5,000-task run -> run finished with every pending task cancelled + SeqStepMs = 20 # sequential no-op steps, first start -> last end per step (9.6ms; 32.6ms before finish+claim shared a txn) } $OrchSkipped = @{} @@ -533,6 +534,14 @@ Invoke-OrchCheck 'perf' { Add-Result 'orch-perf' 'runs-300' ($W.ok -and $Ran -eq 300 -and $Sec -le $OrchGates.ManyRuns300Sec) "${Sec}s" "runs=$($R.Count) done=$(@($R.Where({ $_.phase -eq 'Done' })).Count) tasks ran once=$Ran; gate $($OrchGates.ManyRuns300Sec)s" $null = Invoke-OrchGet "/API/PerfE2EState?ns=$Ns&wipe=1" + # Storage cost between sequential steps: each step's finish and the next step's claim share one transaction. + $Id = New-OrchId; $Ns = "ps-$Id" + $null = Start-OrchRuns $Ns @(@{ name = "E2ESeqPerf-$Id"; label = 'sq'; Sequential = $true; tasks = 100 }) + $W = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['sq'].ended -ge 100) { $S['sq'] } } 120 250 + $StepMs = if ($W.ok) { [math]::Round((ConvertTo-OrchMs ($W.value.maxEnd - $W.value.minStart)) / 99, 1) } else { -1 } + Add-Result 'orch-perf' 'sequential-step' ($W.ok -and $W.value.tasks -eq 100 -and $StepMs -le $OrchGates.SeqStepMs) "${StepMs}ms" "100 steps, first start -> last end per step ${StepMs}ms; gate $($OrchGates.SeqStepMs)ms" + $null = Invoke-OrchGet "/API/PerfE2EState?ns=$Ns&wipe=1" + # An idle pump backs off its poll to JobQueueIdlePollIntervalMs (10s default), so a run queued into an idle # engine waits for the next poll. Pin that bound, then keep the pump warm (a held task in flight keeps it on # the 1s poll) so the size comparison below measures claiming, not the idle backoff. @@ -607,4 +616,4 @@ Invoke-OrchCheck 'restart' { Add-Result 'orch-restart' 'postexec-once' ($P.Count -eq 1 -and $P[0].lines -eq $N -and $H.postExecStatus -eq 'Completed') '-' "posts=$($P.Count) lines=$($P[0].lines) postExec=$($H.postExecStatus)" } -Info ("orchestration gates: fanout-1000 <= {0}s, runs-300 <= {1}s, ttfs-5000 <= {2}ms, rss <= {3}MB, idle claim <= {4}ms, cancel-5000 <= {5}s" -f $OrchGates.Fanout1000Sec, $OrchGates.ManyRuns300Sec, $OrchGates.Ttfs5000Ms, $OrchGates.RssMB, $OrchGates.IdleClaimMs, $OrchGates.Cancel5000Sec) +Info ("orchestration gates: fanout-1000 <= {0}s, runs-300 <= {1}s, ttfs-5000 <= {2}ms, rss <= {3}MB, idle claim <= {4}ms, cancel-5000 <= {5}s, sequential step <= {6}ms" -f $OrchGates.Fanout1000Sec, $OrchGates.ManyRuns300Sec, $OrchGates.Ttfs5000Ms, $OrchGates.RssMB, $OrchGates.IdleClaimMs, $OrchGates.Cancel5000Sec, $OrchGates.SeqStepMs) diff --git a/tests/Craft.Tests/OrchestrationCostTests.cs b/tests/Craft.Tests/OrchestrationCostTests.cs index 74e839d..bc8fd4c 100644 --- a/tests/Craft.Tests/OrchestrationCostTests.cs +++ b/tests/Craft.Tests/OrchestrationCostTests.cs @@ -125,6 +125,33 @@ public async Task ARunAtItsConcurrencyLimit_CostsNoStorageReads_AndThePumpNeverH Assert.Equal(5, c.For(Ready).Queries); } + [Fact] + public async Task ASequentialStep_FinishesAndClaimsTheNext_InOneTransaction_FromAnyRunSize() + { + var (s, c) = NewStore(); + var started = NextStart(); + var run = await s.CreateRunAsync(new RunHeader + { + RunKey = WorkStore.RunKeyFor("Steps", started), + Name = "Steps", + StartedUtc = started, + TaskScriptName = "Invoke-CraftTask", + Sequential = true, + }, Enumerable.Range(0, 10_000).Select(i => new WorkStore.NewTask($"t{i}", new() { ["i"] = i })).ToList()); + var step = (await s.ClaimSequentialAsync(run.RunKey, "w", Lease))!; + c.Reset(); + + var r = await s.FinishStepAsync(run.RunKey, new WorkStore.Finish(step.Seq, "Completed", Owner: "w"), "w", Lease); + + Assert.NotNull(r.Payload); + var w = c.For(Work); + output.WriteLine($"step: work {w} ready {c.For(Ready)}"); + Assert.Equal(3, w.PointReads); // header and step (with the next-step range, read together), payload + Assert.Equal((1, 1, 0), (w.Queries, w.Rows, w.UnboundedRanges)); + Assert.Equal(1, w.Submits); + Assert.Equal(1, c.For(Ready).Upserts); + } + [Fact] public async Task FinishingABatch_IsOneTransaction_AndOneReadyUpdate() { diff --git a/tests/Craft.Tests/OrchestrationModeTests.cs b/tests/Craft.Tests/OrchestrationModeTests.cs index f771f3d..e51771a 100644 --- a/tests/Craft.Tests/OrchestrationModeTests.cs +++ b/tests/Craft.Tests/OrchestrationModeTests.cs @@ -160,6 +160,45 @@ public async Task ASequentialRun_CarriesOnPastAFailedStep_ByDefault() Assert.Equal(3, Assert.Single(h.Svc.PostExecs).Lines.Length); } + [Fact] + public async Task CancellingASequentialRun_StopsItAfterTheStepInHand() + { + await using var h = await OrchestrationHarness.CreateAsync(); + h.Svc.BeforeRun = async t => + { + if (FakeOrchestrator.IdOf(t) == "s1") await h.Svc.CancelRunAsync("SeqCancel"); + }; + Assert.True(await h.Start("SeqCancel", Batch(5, "s"), "Agg", sequential: true)); + + Assert.True(await h.DriveUntilFinished("SeqCancel")); + Assert.Equal(["s0", "s1"], h.Svc.Started); + var run = (await h.Store.GetRunByNameAsync("SeqCancel"))!; + Assert.Equal((0, 3, "CompletedWithErrors"), (run.Failed, run.Cancelled, run.Status)); + } + + /// + /// A step's finish and the claim of the next share one transaction. If that cannot be written the finish + /// goes the ordinary way (retried in the background) and the next step is claimed on its own. + /// + [Fact] + public async Task ASequentialStepWhoseCombinedWriteFails_IsRecordedOnItsOwn_AndTheRunCarriesOn() + { + var faulty = new FaultyTableStore(new MemoryTableStore()); + var failures = 0; + faulty.FailSubmit = (_, ops) => + ops.Count(o => o.Row.RowKey.StartsWith("R|", StringComparison.Ordinal)) == 2 + && Interlocked.CompareExchange(ref failures, 1, 0) == 0; + await using var h = await OrchestrationHarness.CreateAsync(tables: faulty); + Assert.True(await h.Start("SeqFallback", Batch(4, "s"), "Agg", sequential: true)); + + Assert.True(await h.DriveUntilFinished("SeqFallback")); + Assert.Equal(1, failures); + Assert.Equal(["s0", "s1", "s2", "s3"], h.Svc.Started); + var run = (await h.Store.GetRunByNameAsync("SeqFallback"))!; + Assert.Equal(("Completed", 4), (run.Status, run.Done)); + Assert.Contains(h.Log.Lines, l => l.Message.Contains("with its successor; recording it on its own", StringComparison.Ordinal)); + } + [Fact] public async Task StopOnFailure_CancelsTheStepsAfterTheFirstFailure_AndStillAggregates() { diff --git a/tests/Craft.Tests/WorkStoreTests.cs b/tests/Craft.Tests/WorkStoreTests.cs index 2fb5ea2..53b0292 100644 --- a/tests/Craft.Tests/WorkStoreTests.cs +++ b/tests/Craft.Tests/WorkStoreTests.cs @@ -120,7 +120,7 @@ public async Task ATaskThatKeepsDying_IsFailedAfterThreeAttempts() } private static Task CreateModeAsync(WorkStore s, string name, int tasks, bool sequential = false, - bool stopOnFailure = false) + bool stopOnFailure = false, string? postExec = null) { var h = Header(name); return s.CreateRunAsync(new RunHeader @@ -131,9 +131,96 @@ private static Task CreateModeAsync(WorkStore s, string name, int tas TaskScriptName = h.TaskScriptName, Sequential = sequential, StopOnFailure = stopOnFailure, + PostExecFunctionName = postExec, }, Tasks(tasks)); } + // ── a sequential step: finish it and claim the next in one transaction ── + + [Fact] + public async Task FinishingASequentialStep_ClaimsTheNextOne_WithItsPayload_UnderTheDriverLease() + { + var (s, _) = New(); + var run = await CreateModeAsync(s, "Step", 3, sequential: true); + var first = (await s.ClaimSequentialAsync(run.RunKey, "w", Lease))!; + + var r = await s.FinishStepAsync(run.RunKey, new WorkStore.Finish(first.Seq, "Completed", Owner: "w"), "w", Lease); + + Assert.Equal((1, "t1", 1), (r.Next!.Seq, r.Next.TaskId, r.Next.Attempt)); + Assert.Equal("1", r.Payload!["i"].ToString()); + Assert.Equal("Completed", Assert.Single(await s.GetTasksAsync(run.RunKey, 'D')).Status); + Assert.Equal("w", Assert.Single(await s.GetTasksAsync(run.RunKey, 'R')).Owner); + var header = (await s.GetRunAsync(run.RunKey))!; + Assert.Equal(("w", 1), (header.DriverOwner, header.Done)); + Assert.Equal(header.Done, r.Outcome!.Header.Done); + } + + [Fact] + public async Task FinishingTheLastSequentialStep_ClaimsTheAggregationDirectly() + { + var (s, _) = New(); + var run = await CreateModeAsync(s, "StepAgg", 2, sequential: true, postExec: "Agg"); + var step = (await s.ClaimSequentialAsync(run.RunKey, "w", Lease))!; + step = (await s.FinishStepAsync(run.RunKey, new WorkStore.Finish(step.Seq, "Completed", Owner: "w"), "w", Lease)).Next!; + + var r = await s.FinishStepAsync(run.RunKey, new WorkStore.Finish(step.Seq, "Failed", "x", "w"), "w", Lease); + + Assert.Equal(WorkStore.AggregateSeq, r.Next!.Seq); + Assert.Null(r.Payload); + Assert.True(r.Outcome!.ReachedBarrier); + Assert.Empty(await s.GetTasksAsync(run.RunKey, 'P')); + Assert.Equal(WorkStore.AggregateSeq, Assert.Single(await s.GetTasksAsync(run.RunKey, 'R')).Seq); + var header = (await s.GetRunAsync(run.RunKey))!; + Assert.Equal((RunPhase.Aggregate, "Pending", 1), (header.Phase, header.PostExecStatus, header.Failed)); + } + + [Fact] + public async Task FinishingTheLastSequentialStep_WithoutAggregation_CompletesTheRun_AndClaimsNothing() + { + var (s, _) = New(); + var run = await CreateModeAsync(s, "StepLast", 1, sequential: true); + var step = (await s.ClaimSequentialAsync(run.RunKey, "w", Lease))!; + + var r = await s.FinishStepAsync(run.RunKey, new WorkStore.Finish(step.Seq, "Completed", Owner: "w"), "w", Lease); + + Assert.Null(r.Next); + Assert.True(r.Outcome!.Completed); + Assert.Equal("Completed", (await s.GetRunAsync(run.RunKey))!.Status); + } + + [Fact] + public async Task ACancelRequest_StopsTheNextStepBeingClaimed() + { + var (s, _) = New(); + var run = await CreateModeAsync(s, "StepCancel", 3, sequential: true); + var step = (await s.ClaimSequentialAsync(run.RunKey, "w", Lease))!; + await s.RequestCancelAsync(run.RunKey); + + var r = await s.FinishStepAsync(run.RunKey, new WorkStore.Finish(step.Seq, "Completed", Owner: "w"), "w", Lease); + + Assert.Null(r.Next); + Assert.True(r.Outcome!.Header.CancelRequested); + Assert.Equal(2, (await s.GetTasksAsync(run.RunKey, 'P')).Count); + Assert.Empty(await s.GetTasksAsync(run.RunKey, 'R')); + } + + [Fact] + public async Task AStepTakenOverByAnotherDriver_IsNeitherFinishedNorFollowed_ByItsFormerOwner() + { + var (s, _) = New(); + var run = await CreateModeAsync(s, "StepLost", 3, sequential: true); + Assert.NotNull(await s.ClaimSequentialAsync(run.RunKey, "old", TimeSpan.FromMilliseconds(1))); + await Task.Delay(20); + var taken = (await s.ClaimSequentialAsync(run.RunKey, "new", Lease))!; + + var r = await s.FinishStepAsync(run.RunKey, new WorkStore.Finish(taken.Seq, "Completed", Owner: "old"), "old", Lease); + + Assert.Null(r.Next); + Assert.Equal(0, r.Outcome!.Applied); + Assert.Equal("new", Assert.Single(await s.GetTasksAsync(run.RunKey, 'R')).Owner); + Assert.Equal("new", (await s.GetRunAsync(run.RunKey))!.DriverOwner); + } + [Fact] public async Task ASequentialStepThatKeepsDying_StopsAStopOnFailureRun() { From 28e742bc4ebac111913756c741c5b685a3622c85 Mon Sep 17 00:00:00 2001 From: Zacgoose <107489668+Zacgoose@users.noreply.github.com> Date: Tue, 6 Oct 2026 13:29:22 +0800 Subject: [PATCH 20/24] fix(orchestration): treat queue-id suffixed run names as one name for collisions Runs named Name-{guid} (a queue id appended per outing) collide with each other and with Name when collisions are off. --- Services/Bridges/OrchestratorBridge.cs | 11 ++++--- Services/Orchestration/OrchestratorService.cs | 14 ++++---- Services/Storage/WorkStore.cs | 23 +++++++++++++ .../Craft.Tests/OrchestrationContractTests.cs | 32 +++++++++++++++++++ .../OrchestratorBridgeLineageTests.cs | 8 +++++ 5 files changed, 77 insertions(+), 11 deletions(-) diff --git a/Services/Bridges/OrchestratorBridge.cs b/Services/Bridges/OrchestratorBridge.cs index 815b445..c32cc07 100644 --- a/Services/Bridges/OrchestratorBridge.cs +++ b/Services/Bridges/OrchestratorBridge.cs @@ -113,15 +113,16 @@ private static (string ParentRunKey, string ChildKey)? RegisterPendingChild(stri string.IsNullOrEmpty(parentRunName) ? null : s_service?.RegisterPendingChild(parentRunName, childName); /// - /// Whether a run of this name is unfinished or already queued here to start. Lets a caller that does not - /// want overlapping runs (allowCollision: false) skip, and say so, before building the batch. The - /// start itself checks again, so a run that appears in between is still skipped. + /// Whether a run of this name is unfinished or already queued here to start, counting outings that carry a + /// queue id suffix (Name-{guid}) as the same name. Lets a caller that does not want overlapping runs + /// (allowCollision: false) skip, and say so, before building the batch. The start itself checks + /// again, so a run that appears in between is still skipped. /// PS usage: [Craft.Services.OrchestratorBridge]::IsRunActive($name). /// public static bool IsRunActive(string name) { - name = TableKeys.Sanitize(name); - if (s_pending.Any(p => p.Name == name)) return true; + var family = WorkStore.CollisionFamily(TableKeys.Sanitize(name)); + if (s_pending.Any(p => WorkStore.CollisionFamily(TableKeys.Sanitize(p.Name)) == family)) return true; return s_service != null && Task.Run(() => s_service.IsRunActiveAsync(name)).GetAwaiter().GetResult(); } diff --git a/Services/Orchestration/OrchestratorService.cs b/Services/Orchestration/OrchestratorService.cs index b0b4278..72c9172 100644 --- a/Services/Orchestration/OrchestratorService.cs +++ b/Services/Orchestration/OrchestratorService.cs @@ -169,14 +169,15 @@ public async Task StartFromBatchAsync(string name, string batchJson, int p bool sequential = false, string? parentRunKey = null, string? childKey = null, bool allowCollision = true, int maxConcurrency = 0, bool stopOnFailure = false) { - var gated = false; + string? gate = null; try { name = TableKeys.Sanitize(name); if (!allowCollision) { - gated = _activePlanners.TryAdd(name, true); - if (!gated || await IsActiveAsync(name, ct)) + var family = WorkStore.CollisionFamily(name); + if (_activePlanners.TryAdd(family, true)) gate = family; + if (gate == null || await _store.IsFamilyActiveAsync(family, ct)) { _logger.LogWarning("[Orchestrator] Run {Name} skipped: a run of that name is still active and collisions are off", name); return false; @@ -207,7 +208,7 @@ await CreateAsync(name, tasks, priority, genericTaskFunc, postExecFunctionName, } finally { - if (gated) _activePlanners.TryRemove(name, out _); + if (gate != null) _activePlanners.TryRemove(gate, out _); if (!string.IsNullOrEmpty(batchFilePath)) { try { if (File.Exists(batchFilePath)) File.Delete(batchFilePath); } @@ -219,8 +220,9 @@ await CreateAsync(name, tasks, priority, genericTaskFunc, postExecFunctionName, private async Task IsActiveAsync(string name, CancellationToken ct) => (await _store.GetActiveRunsAsync(name, ct)).Count > 0; - /// Whether any run with this name is unfinished. - public Task IsRunActiveAsync(string name, CancellationToken ct = default) => IsActiveAsync(TableKeys.Sanitize(name), ct); + /// Whether any run of this name's collision family () is unfinished. + public Task IsRunActiveAsync(string name, CancellationToken ct = default) => + _store.IsFamilyActiveAsync(WorkStore.CollisionFamily(TableKeys.Sanitize(name)), ct); /// The runs an operator action names: the run with that key, or every unfinished run of that name. private async Task> TargetRunsAsync(string keyOrName) diff --git a/Services/Storage/WorkStore.cs b/Services/Storage/WorkStore.cs index f894864..fe36def 100644 --- a/Services/Storage/WorkStore.cs +++ b/Services/Storage/WorkStore.cs @@ -194,6 +194,29 @@ public async Task> GetActiveRunsAsync(string name, CancellationT return runs; } + /// + /// The name runs collide on: the run name without a trailing -{guid}, the queue id a caller appends so + /// its queue page can find each outing (Cache-{queueId} and Cache are one family). + /// + public static string CollisionFamily(string name) => + name.Length > 37 && name[^37] == '-' && Guid.TryParseExact(name.AsSpan(name.Length - 36), "D", out _) + ? name[..^37] + : name; + + /// Whether any run of this collision family is unfinished: one range read for the plain name and one + /// for {family}-.... + public async Task IsFamilyActiveAsync(string family, CancellationToken ct = default) + { + if ((await GetActiveRunsAsync(family, ct)).Count > 0) return true; + await foreach (var row in _store.QueryRowKeyRangeAsync(_names, ActivePartition, $"{family}-", $"{family}.", ct: ct)) + { + if (row.GetString("Name") is { } name && CollisionFamily(name) == family + && await GetRunAsync(row.RowKey, ct) is { IsFinished: false }) + return true; + } + return false; + } + /// A run by its key, or else the newest unfinished run with that name. public async Task ResolveRunAsync(string keyOrName, CancellationToken ct = default) => (keyOrName.Contains('~') ? await GetRunAsync(keyOrName, ct) : null) diff --git a/tests/Craft.Tests/OrchestrationContractTests.cs b/tests/Craft.Tests/OrchestrationContractTests.cs index 2c7f4af..78e1e70 100644 --- a/tests/Craft.Tests/OrchestrationContractTests.cs +++ b/tests/Craft.Tests/OrchestrationContractTests.cs @@ -104,6 +104,38 @@ public async Task WithoutCollisions_ARunNameStillGoing_IsNotStartedAgain_AndItsB Assert.True(await Start(h, "Recurring", Batch(1), allowCollision: false)); } + /// + /// Callers append a queue id to a run's name so their queue page can find each outing. Without collisions, + /// Cache-{queueId} must still collide with yesterday's Cache-{otherQueueId} and with plain + /// Cache, or the flag never skips anything. + /// + [Fact] + public async Task WithoutCollisions_OutingsThatCarryAQueueId_CollideWithEachOther_AndWithThePlainName() + { + await using var h = await NewAsync(); + var first = $"Cache-{Guid.NewGuid()}"; + Assert.True(await Start(h, first, Batch(1), allowCollision: false)); + + Assert.False(await Start(h, $"Cache-{Guid.NewGuid()}", Batch(1), allowCollision: false)); + Assert.False(await Start(h, "Cache", Batch(1), allowCollision: false)); + Assert.True(await Start(h, $"CacheMore-{Guid.NewGuid()}", Batch(1), allowCollision: false)); + Assert.True(await Start(h, "Cache-weekly", Batch(1), allowCollision: false)); + Assert.True(await Start(h, $"Cache-{Guid.NewGuid()}", Batch(1))); + + Assert.True(await h.DriveUntilAllFinished()); + Assert.True(await Start(h, $"Cache-{Guid.NewGuid()}", Batch(1), allowCollision: false)); + } + + [Fact] + public async Task ANameSuffixThatIsNotAQueueId_IsItsOwnName_ForCollisions() + { + await using var h = await NewAsync(); + Assert.True(await Start(h, "Report-weekly", Batch(1), allowCollision: false)); + + Assert.True(await Start(h, "Report", Batch(1), allowCollision: false)); + Assert.False(await Start(h, $"Report-{Guid.NewGuid()}", Batch(1), allowCollision: false)); + } + [Fact] public async Task ByDefault_RunsOfOneNameStackUp_AndEachRunsEveryTask() { diff --git a/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs b/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs index 6b79e9b..cfbf5fb 100644 --- a/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs +++ b/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs @@ -306,6 +306,14 @@ public async Task IsRunActive_SeesUnfinishedRunsAndRunsQueuedToStart_SoACallerCa OrchestratorBridge.QueueOrchestration("LineageQueuedRun", "[]", 4); Assert.True(OrchestratorBridge.IsRunActive("LineageQueuedRun")); Assert.NotNull(TakePending("LineageQueuedRun")); + + await CreateRunAsync(store, $"LineageFamily-{Guid.NewGuid()}"); + Assert.True(OrchestratorBridge.IsRunActive($"LineageFamily-{Guid.NewGuid()}")); + Assert.True(OrchestratorBridge.IsRunActive("LineageFamily")); + var queued = $"LineageQueuedFamily-{Guid.NewGuid()}"; + OrchestratorBridge.QueueOrchestration(queued, "[]", 4); + Assert.True(OrchestratorBridge.IsRunActive($"LineageQueuedFamily-{Guid.NewGuid()}")); + Assert.NotNull(TakePending(queued)); } finally { From e04011c240d0f41d64906841a682638f1593059b Mon Sep 17 00:00:00 2001 From: Zacgoose <107489668+Zacgoose@users.noreply.github.com> Date: Tue, 6 Oct 2026 14:21:42 +0800 Subject: [PATCH 21/24] fix(orchestration): show every step of a sequential run in worker stats, job lists and the queue page - The driving job carries each step's name and keeps a finished record per step; the last step's outcome is the job's - Run summaries and task lists cover the latest outing of a name only - e2e check orch-seq each-step-visible --- Services/Bridges/JobRecord.cs | 3 + Services/Bridges/QueueStatusBridge.cs | 2 +- Services/Orchestration/JobManager.cs | 57 +++++++++++++++++- Services/Orchestration/OrchestratorService.cs | 25 +++++++- Services/Orchestration/WorkPump.cs | 9 ++- .../API/Modules/PerfApi/PerfApi.psm1 | 5 ++ .../scripts/run-e2e-orchestration.ps1 | 13 ++++ tests/Craft.Tests/OrchestrationHarness.cs | 5 ++ tests/Craft.Tests/OrchestrationModeTests.cs | 60 +++++++++++++++++++ 9 files changed, 172 insertions(+), 7 deletions(-) diff --git a/Services/Bridges/JobRecord.cs b/Services/Bridges/JobRecord.cs index 29f0771..5ae776e 100644 --- a/Services/Bridges/JobRecord.cs +++ b/Services/Bridges/JobRecord.cs @@ -13,6 +13,9 @@ public class JobRecord public string Id { get; set; } = string.Empty; public string Name { get; set; } = string.Empty; public string? RunName { get; set; } + + /// The outing of the run this job belongs to, when it came from a stored run. + public string? RunKey { get; set; } public int Priority { get; set; } public string Status { get; set; } = "Queued"; public DateTime QueuedUtc { get; set; } diff --git a/Services/Bridges/QueueStatusBridge.cs b/Services/Bridges/QueueStatusBridge.cs index f3bd14b..47db101 100644 --- a/Services/Bridges/QueueStatusBridge.cs +++ b/Services/Bridges/QueueStatusBridge.cs @@ -190,7 +190,7 @@ private static List GetTaskDetails(string runName) { if (s_jobManager == null) return []; - var jobs = s_jobManager.GetJobs(runName, limit: 100); + var jobs = s_jobManager.GetRunJobs(runName, limit: 100); return jobs.Select(j => new TaskDetail { Timestamp = (j.CompletedUtc ?? j.StartedUtc ?? j.QueuedUtc).ToString("O"), diff --git a/Services/Orchestration/JobManager.cs b/Services/Orchestration/JobManager.cs index c412f66..55762e6 100644 --- a/Services/Orchestration/JobManager.cs +++ b/Services/Orchestration/JobManager.cs @@ -199,6 +199,7 @@ public string Enqueue(JobDescriptor descriptor, string name, string? id = null) Id = jobId, Name = name, RunName = descriptor.RunName, + RunKey = descriptor.RunKey, Priority = descriptor.Priority, Status = "Queued", QueuedUtc = DateTime.UtcNow @@ -420,7 +421,9 @@ private async Task RunJobAsync(QueuedJob job, CancellationToken ct) await work(ct); - job.Record.Status = "Completed"; + var (status, error) = _lastStep.TryRemove(job.Record.Id, out var last) ? last : ("Completed", null); + job.Record.Status = status; + job.Record.LastError = error; job.Record.CompletedUtc = DateTime.UtcNow; } catch (OperationCanceledException) when (ct.IsCancellationRequested) @@ -438,6 +441,7 @@ private async Task RunJobAsync(QueuedJob job, CancellationToken ct) } finally { + _lastStep.TryRemove(job.Record.Id, out _); // Accounting first, context teardown second: ReleaseSlot cannot throw, Dispose can. Interlocked.Decrement(ref _activeCount); Interlocked.Increment(ref _totalProcessed); @@ -474,7 +478,7 @@ public List GetRunSummaries() .GroupBy(j => j.RunName!) .Select(g => { - var jobs = g.ToList(); + var jobs = LatestOuting(g); return new JobRunSummary { Name = g.Key, @@ -495,6 +499,18 @@ public List GetRunSummaries() .ToList(); } + /// A run's jobs, newest first, from its latest outing only: earlier runs of the same name are left out. + public List GetRunJobs(string runName, int limit) => + LatestOuting(_jobs.Values.Where(j => string.Equals(j.RunName, runName, StringComparison.OrdinalIgnoreCase))) + .OrderByDescending(j => j.QueuedUtc).Take(limit).ToList(); + + private static List LatestOuting(IEnumerable jobs) + { + var list = jobs.ToList(); + var latest = list.Where(j => j.RunKey != null).MaxBy(j => j.QueuedUtc)?.RunKey; + return latest == null ? list : list.Where(j => j.RunKey == latest).ToList(); + } + public List GetJobs(string? runName = null, string? status = null, int? limit = null) { var query = _jobs.Values.AsEnumerable(); @@ -588,6 +604,43 @@ public bool CancelJob(string jobId) return true; } + /// + /// A job that runs a sequential run's steps one after another has finished a step. The step is kept as its + /// own finished record and the job carries on under ; when there is no next step, + /// this step's outcome becomes the job's. Job lists, run summaries and queue pages then show every step, + /// not just the one the job was dispatched for. + /// + public void AdvanceStep(string jobId, string status, string? error, string? nextName) + { + if (!_jobs.TryGetValue(jobId, out var record)) return; + if (nextName == null) + { + _lastStep[jobId] = (status, error); + return; + } + var now = DateTime.UtcNow; + var stepId = $"{jobId}#{Interlocked.Increment(ref _stepRecords)}"; + _jobs[stepId] = new JobRecord + { + Id = stepId, + Name = record.Name, + RunName = record.RunName, + RunKey = record.RunKey, + Priority = record.Priority, + Status = status, + QueuedUtc = record.QueuedUtc, + StartedUtc = record.StartedUtc, + CompletedUtc = now, + LastError = error, + }; + record.Name = nextName; + record.QueuedUtc = now; + record.StartedUtc = now; + } + + private readonly ConcurrentDictionary _lastStep = new(); + private long _stepRecords; + /// /// Take a queued job back without running it and without recording an outcome, so its owner can hand the /// work to another process. False once it has started. diff --git a/Services/Orchestration/OrchestratorService.cs b/Services/Orchestration/OrchestratorService.cs index 72c9172..838dec4 100644 --- a/Services/Orchestration/OrchestratorService.cs +++ b/Services/Orchestration/OrchestratorService.cs @@ -2,6 +2,7 @@ using System.Text.Json; using System.Text.Json.Serialization; using Craft.Configuration; +using Craft.Hosting; using Craft.PowerShellHost; using Craft.Services; using Craft.Storage; @@ -498,6 +499,10 @@ private Func BuildSequentialRunWork(RunHeader header, J PowerShellWorker? worker = null; var faulted = false; var step = new WorkStore.ClaimedTask(header.RunKey, first.Seq, first.TaskId, first.Attempt); + // The job this driver runs in carries each step's name in turn, and keeps a record of each step it finishes. + var jobId = WorkPump.JobId(step); + void StepDone(string status, string? error, WorkStore.ClaimedTask? next) => + _jobManager.AdvanceStep(jobId, status, error, next is { } n ? WorkPump.JobName(header.Name, n) : null); RunHeader? known = null; Dictionary? nextPayload = null; try @@ -511,6 +516,7 @@ private Func BuildSequentialRunWork(RunHeader header, J if (current.CancelRequested) { await FinishAsync(header, step.Seq, "Cancelled", "Cancelled by user"); + StepDone("Cancelled", "Cancelled by user", null); await _store.CancelPendingAsync(header.RunKey, ct: jobCt); break; } @@ -527,7 +533,18 @@ private Func BuildSequentialRunWork(RunHeader header, J try { if (parameters == null) throw new InvalidOperationException("The task's payload row is missing"); - var output = await RunScriptAsync(taskPath, TaskInvocation(parameters), current.HasPostExec, worker); + var job = OperationContext.Current; + string output; + using (OperationContext.Set(new OperationContext.Invocation(WorkPump.JobName(header.Name, step)) + { + WorkerId = job?.WorkerId, + RunName = job?.RunName, + RunKey = job?.RunKey, + Priority = job?.Priority, + })) + { + output = await RunScriptAsync(taskPath, TaskInvocation(parameters), current.HasPostExec, worker); + } if (current.HasPostExec && !string.IsNullOrEmpty(output)) await _results.StoreResultAsync(header.RunKey, step.TaskId, output); finish = new WorkStore.Finish(step.Seq, "Completed", Owner: Owner); @@ -541,6 +558,7 @@ private Func BuildSequentialRunWork(RunHeader header, J if (current.StopOnFailure) { await FinishAsync(header, step.Seq, "Failed", ex.Message); + StepDone("Failed", ex.Message, null); _logger.LogError(ex, "[Scheduler] Sequential task failed: {TaskId} — stopping the run", step.TaskId); await _store.CancelPendingAsync(header.RunKey, WorkStore.StoppedReason(step.TaskId), ct: jobCt); break; @@ -558,10 +576,13 @@ private Func BuildSequentialRunWork(RunHeader header, J _logger.LogWarning(ex, "[Scheduler] Could not finish step {TaskId} of {Run} with its successor; recording it on its own", step.TaskId, header.Name); await _finisher.FinishAsync(header.RunKey, finish); - if (await _store.ClaimSequentialAsync(header.RunKey, Owner, Lease, continuing: true, ct: jobCt) is not { } claimed) break; + var claimed = await _store.ClaimSequentialAsync(header.RunKey, Owner, Lease, continuing: true, ct: jobCt); + StepDone(finish.Status, finish.Error, claimed); + if (claimed == null) break; step = claimed; continue; } + StepDone(finish.Status, finish.Error, result.Next); if (result.Next is not { } next) { if (result.Outcome?.Header is { CancelRequested: true, IsFinished: false }) diff --git a/Services/Orchestration/WorkPump.cs b/Services/Orchestration/WorkPump.cs index 62f2a70..ff886a5 100644 --- a/Services/Orchestration/WorkPump.cs +++ b/Services/Orchestration/WorkPump.cs @@ -161,6 +161,12 @@ protected override async Task ExecuteAsync(CancellationToken stoppingToken) internal void ForgetBackoff() => _skip.Clear(); + /// The job a claimed task runs as: its id, and the name job lists and worker stats show. + internal static string JobId(WorkStore.ClaimedTask c) => $"{c.RunKey}|{c.Seq}"; + + internal static string JobName(string runName, WorkStore.ClaimedTask c) => + c.Seq == WorkStore.AggregateSeq ? $"{runName}-PostExec" : $"{runName}-{c.TaskId}"; + /// /// Wait until this process holds the instance lock. A predecessor that shut down cleanly released it, so this /// is immediate after a normal recycle; one that crashed holds it until its lease runs out. @@ -348,9 +354,8 @@ internal async Task RefillAsync(CancellationToken ct) foreach (var c in claims) { - var name = c.Seq == WorkStore.AggregateSeq ? $"{header.Name}-PostExec" : $"{header.Name}-{c.TaskId}"; var descriptor = new JobDescriptor(header.Name, c.TaskId, header.Priority) { RunKey = c.RunKey, Seq = c.Seq, Attempt = c.Attempt }; - var jobId = _jobs.Enqueue(descriptor, name, id: $"{c.RunKey}|{c.Seq}"); + var jobId = _jobs.Enqueue(descriptor, JobName(header.Name, c), id: JobId(c)); _inFlight[jobId] = (c, now + _lease); if (held != null) held[c.RunKey] = held.GetValueOrDefault(c.RunKey) + 1; } diff --git a/perf-harness/api-harness/API/Modules/PerfApi/PerfApi.psm1 b/perf-harness/api-harness/API/Modules/PerfApi/PerfApi.psm1 index 2601ca0..cb82228 100644 --- a/perf-harness/api-harness/API/Modules/PerfApi/PerfApi.psm1 +++ b/perf-harness/api-harness/API/Modules/PerfApi/PerfApi.psm1 @@ -800,6 +800,11 @@ function Invoke-PerfE2EBridge { startHasMaxConcurrency = ($Def -match 'MaxConcurrency'); startHasStopOnFailure = ($Def -match 'StopOnFailure') } } 'active' { $Body = @{ active = [Craft.Services.OrchestratorBridge]::IsRunActive($Name) } } + 'queue' { $Body = @{ entries = @([Craft.Services.QueueStatusBridge]::GetRunStatus($null, $Name) | ConvertFrom-Json) } } + 'workers' { + $Body = @{ busy = @([Craft.Services.WorkerMetricsBridge]::GetSnapshot().BgPool.Workers | Where-Object IsBusy | + ForEach-Object { @{ id = $_.WorkerId; fn = $_.CurrentFunction } }) } + } 'cancel' { $Body = @{ cancelled = [Craft.Services.WorkerMetricsBridge]::CancelRun($Name) } } 'summaries' { $Body = @{ runs = @([Craft.Services.WorkerMetricsBridge]::GetRunSummaries() | Where-Object { -not $Name -or $_.Name -eq $Name } | ForEach-Object { diff --git a/perf-harness/scripts/run-e2e-orchestration.ps1 b/perf-harness/scripts/run-e2e-orchestration.ps1 index daccf71..be0772e 100644 --- a/perf-harness/scripts/run-e2e-orchestration.ps1 +++ b/perf-harness/scripts/run-e2e-orchestration.ps1 @@ -387,6 +387,19 @@ Invoke-OrchCheck 'seq' { # Every step that succeeded must reach the PostExecution, including the one right after the failed step. $Got = @($P.idxs -split ',' | Sort-Object) -join ',' Add-Result 'orch-seq' 'postexec-gets-every-success' ($P.lines -eq 4 -and $Got -eq 'sf/0,sf/1,sf/3,sf/4') '-' "expected sf/0,sf/1,sf/3,sf/4 (step 2 throws); PostExecution received lines=$($P.lines) idxs=$($P.idxs)" + + # Every step runs inside the one job that drives the run, but worker stats and the queue page must show + # each step on its own: the worker under the step it is on, the queue with one task per step. + $Id = New-OrchId; $Ns = "seqv-$Id"; $Name = "E2ESeqV-$Id" + $null = Start-OrchRuns $Ns @(@{ name = $Name; label = 'sv'; Sequential = $true; tasks = 4; task = @{ holdms = 1500 } }) + $Labels = [System.Collections.Generic.HashSet[string]]::new() + $null = Wait-Orch { + foreach ($B in @((Invoke-OrchBridge 'workers').busy)) { if ($B.fn -like "$Name-*") { [void]$Labels.Add($B.fn) } } + $S = Get-OrchSummary $Ns; if ($S['sv'].ended -ge 4) { $true } + } 60 250 + $Q = Wait-Orch { $E = @((Invoke-OrchBridge 'queue' $Name).entries)[0]; if ($E.Status -eq 'Completed') { $E } } 30 + $E = $Q.value + Add-Result 'orch-seq' 'each-step-visible' ($Labels.Count -ge 3 -and $E.TotalTasks -eq 4 -and $E.CompletedTasks -eq 4 -and @($E.Tasks).Count -eq 4) '-' "worker labels seen=$($Labels.Count) (of 4 steps); queue total=$($E.TotalTasks) completed=$($E.CompletedTasks) tasks listed=$(@($E.Tasks).Count)" } # -- 9. Priority and start-order ------------------------------------------------------------------------------ diff --git a/tests/Craft.Tests/OrchestrationHarness.cs b/tests/Craft.Tests/OrchestrationHarness.cs index 5a643ae..c8a281a 100644 --- a/tests/Craft.Tests/OrchestrationHarness.cs +++ b/tests/Craft.Tests/OrchestrationHarness.cs @@ -1,6 +1,7 @@ using System.Collections.Concurrent; using System.Text.Json; using Craft.Configuration; +using Craft.Hosting; using Craft.Orchestration; using Craft.PowerShellHost; using Craft.Storage; @@ -23,6 +24,9 @@ internal sealed class FakeOrchestrator(JobManager jobs, WorkStore store, ResultS public readonly ConcurrentQueue> Tasks = new(); public readonly ConcurrentQueue Started = new(); + + /// The job name each task ran under, as worker stats show it. + public readonly ConcurrentQueue RanAs = new(); public readonly ConcurrentQueue<(Dictionary Parameters, string[] Lines)> PostExecs = new(); /// The task's output; throw to fail it. @@ -55,6 +59,7 @@ internal override async Task RunScriptAsync(string path, Dictionary>((string)parameters["TaskJson"])!; Tasks.Enqueue(task); Started.Enqueue(IdOf(task)); + RanAs.Enqueue(OperationContext.Current?.Function); var now = Interlocked.Increment(ref _active); for (var seen = _maxActive; now > seen; seen = _maxActive) if (Interlocked.CompareExchange(ref _maxActive, now, seen) == seen) break; diff --git a/tests/Craft.Tests/OrchestrationModeTests.cs b/tests/Craft.Tests/OrchestrationModeTests.cs index e51771a..afc6f8a 100644 --- a/tests/Craft.Tests/OrchestrationModeTests.cs +++ b/tests/Craft.Tests/OrchestrationModeTests.cs @@ -160,6 +160,66 @@ public async Task ASequentialRun_CarriesOnPastAFailedStep_ByDefault() Assert.Equal(3, Assert.Single(h.Svc.PostExecs).Lines.Length); } + /// + /// A sequential run's steps all run inside the one job that drives them. Each step must still show up as + /// a job of its own: under its own name on the worker, and as its own record in job lists and run + /// summaries, or the queue page reports one task for the whole run. + /// + [Fact] + public async Task EveryStepOfASequentialRun_ShowsUpAsItsOwnJob() + { + await using var h = await OrchestrationHarness.CreateAsync(); + h.Svc.Body = t => FakeOrchestrator.IdOf(t) == "s1" ? throw new InvalidOperationException("step down") : "{}"; + Assert.True(await h.Start("SeqJobs", Batch(4, "s"), sequential: true)); + + Assert.True(await h.DriveUntilFinished("SeqJobs")); + Assert.Equal(["SeqJobs-Job_s0", "SeqJobs-Job_s1", "SeqJobs-Job_s2", "SeqJobs-Job_s3"], h.Svc.RanAs); + Assert.True(await h.DriveUntil(() => Task.FromResult(h.Jobs.GetJobs("SeqJobs").All(j => j.Status is not ("Queued" or "Running"))))); + var jobs = h.Jobs.GetJobs("SeqJobs").ToDictionary(j => j.Name, j => j.Status); + Assert.Equal(new Dictionary + { + ["SeqJobs-Job_s0"] = "Completed", + ["SeqJobs-Job_s1"] = "Failed", + ["SeqJobs-Job_s2"] = "Completed", + ["SeqJobs-Job_s3"] = "Completed", + }, jobs); + var summary = Assert.Single(h.Jobs.GetRunSummaries(), s => s.Name == "SeqJobs"); + Assert.Equal((4, 3, 1), (summary.Total, summary.Completed, summary.Failed)); + } + + [Fact] + public async Task ARunsSummaryAndTaskList_CoverItsLatestOuting_NotEveryRunOfThatName() + { + await using var h = await OrchestrationHarness.CreateAsync(); + for (var outing = 0; outing < 2; outing++) + { + Assert.True(await h.Start("SeqTwice", Batch(3, "s"), sequential: true)); + Assert.True(await h.DriveUntilFinished("SeqTwice")); + Assert.True(await h.DriveUntil(() => Task.FromResult(h.Jobs.GetJobs("SeqTwice").All(j => j.Status is not ("Queued" or "Running"))))); + } + + Assert.Equal(6, h.Jobs.GetJobs("SeqTwice").Count); + Assert.Equal(3, h.Jobs.GetRunJobs("SeqTwice", 100).Count); + var summary = Assert.Single(h.Jobs.GetRunSummaries(), s => s.Name == "SeqTwice"); + Assert.Equal((3, 3), (summary.Total, summary.Completed)); + } + + [Fact] + public async Task ASequentialRunsLastStep_DecidesItsJobsOutcome_AndItsAggregationIsAJobToo() + { + await using var h = await OrchestrationHarness.CreateAsync(); + h.Svc.Body = t => FakeOrchestrator.IdOf(t) == "s1" ? throw new InvalidOperationException("last down") : "{}"; + Assert.True(await h.Start("SeqLast", Batch(2, "s"), "Agg", sequential: true)); + + Assert.True(await h.DriveUntilFinished("SeqLast")); + Assert.True(await h.DriveUntil(() => Task.FromResult(h.Jobs.GetJobs("SeqLast").All(j => j.Status is not ("Queued" or "Running"))))); + var jobs = h.Jobs.GetJobs("SeqLast").ToDictionary(j => j.Name, j => j.Status); + Assert.Equal("Completed", jobs["SeqLast-Job_s0"]); + Assert.Equal("Failed", jobs["SeqLast-Job_s1"]); + Assert.Equal("Completed", jobs["SeqLast-PostExec"]); + Assert.Equal(3, jobs.Count); + } + [Fact] public async Task CancellingASequentialRun_StopsItAfterTheStepInHand() { From c296552b4ffe6b588fab57bab7f238445f8b9644 Mon Sep 17 00:00:00 2001 From: Zacgoose <107489668+Zacgoose@users.noreply.github.com> Date: Tue, 6 Oct 2026 16:00:00 +0800 Subject: [PATCH 22/24] feat(auth): keep the token's claims on the normalised principal - app-only and signed-in principals both carry the validated token's claims (azp, scp, ...) - pin the app-only vs user split and the rewrite with middleware tests --- Services/Auth/EasyAuthPrincipal.cs | 20 +++ Services/Hosting/CraftAuthMiddleware.cs | 27 ++-- tests/Craft.Tests/CraftAuthMiddlewareTests.cs | 152 ++++++++++++++++++ tests/Craft.Tests/EasyAuthPrincipalTests.cs | 37 +++++ 4 files changed, 218 insertions(+), 18 deletions(-) create mode 100644 tests/Craft.Tests/CraftAuthMiddlewareTests.cs diff --git a/Services/Auth/EasyAuthPrincipal.cs b/Services/Auth/EasyAuthPrincipal.cs index 8be18f2..9eb4f3a 100644 --- a/Services/Auth/EasyAuthPrincipal.cs +++ b/Services/Auth/EasyAuthPrincipal.cs @@ -73,6 +73,26 @@ public static JsonDocument Decode(string headerValue) => public static string Encode(T principal) => Convert.ToBase64String(Encoding.UTF8.GetBytes(JsonSerializer.Serialize(principal))); + /// + /// Encodes the normalised SWA-format principal for an EasyAuth , keeping the + /// token's original claims so the hosted app can read any of them (e.g. azp, scp). + /// + /// + /// The claims are those of the token EasyAuth validated, and userRoles marks the result as + /// already normalised (), so it is never transformed twice. + /// + public static string EncodeNormalised( + JsonElement source, string identityProvider, string userId, string userDetails, IReadOnlyList userRoles) + { + var claims = source.ValueKind == JsonValueKind.Object && + source.TryGetProperty("claims", out var sourceClaims) && + sourceClaims.ValueKind == JsonValueKind.Array + ? sourceClaims.Clone() + : JsonDocument.Parse("[]").RootElement.Clone(); + + return Encode(new { identityProvider, userId, userDetails, userRoles, claims }); + } + /// /// Pulls the identity claims out of an EasyAuth principal. /// diff --git a/Services/Hosting/CraftAuthMiddleware.cs b/Services/Hosting/CraftAuthMiddleware.cs index 8a3cb6c..c24152c 100644 --- a/Services/Hosting/CraftAuthMiddleware.cs +++ b/Services/Hosting/CraftAuthMiddleware.cs @@ -51,7 +51,7 @@ public static WebApplication UseCraftAuth( if (hasPrincipal) { // Returns false when the caller was rejected and a response has already been written. - if (!await TryNormalisePrincipalAsync(context, authService, logger, existingHeader.ToString())) + if (!await TryNormalisePrincipalAsync(context, ids => authService.GetUserRoles(ids), logger, existingHeader.ToString())) return; } else if (isDevelopment) @@ -68,9 +68,10 @@ public static WebApplication UseCraftAuth( return app; } + /// Looks a signed-in user up in the allowedUsers table by its identifiers. /// if the request was rejected and the response is already written. - private static async Task TryNormalisePrincipalAsync( - HttpContext context, AuthService authService, ILogger logger, string headerValue) + internal static async Task TryNormalisePrincipalAsync( + HttpContext context, Func, Task> resolveUserRoles, ILogger logger, string headerValue) { try { @@ -91,13 +92,8 @@ private static async Task TryNormalisePrincipalAsync( // Service principal. The idp header MUST stay "aad": the hosted app keys off it to treat // the caller as an API client and resolve its name from the ApiClients table. The real // provider goes in identityProvider for audit only. - context.Request.Headers["x-ms-client-principal"] = EasyAuthPrincipal.Encode(new - { - identityProvider = realIdp, - userId = claims.ObjectId ?? claims.AppId, - userDetails = claims.AppId, - userRoles = Array.Empty(), - }); + context.Request.Headers["x-ms-client-principal"] = EasyAuthPrincipal.EncodeNormalised( + root, realIdp, claims.ObjectId ?? claims.AppId, claims.AppId, Array.Empty()); context.Request.Headers["x-ms-client-principal-idp"] = "aad"; context.Request.Headers["x-ms-client-principal-name"] = claims.AppId; return true; @@ -114,7 +110,7 @@ private static async Task TryNormalisePrincipalAsync( // Resolve roles by the display name AND the stable object id (a GitHub user's numeric id, // an Entra user's oid), so an allowedUsers row keyed on either one grants the user its roles. - var roles = await authService.GetUserRoles(new[] { userName, claims.ObjectId }); + var roles = await resolveUserRoles(new[] { userName, claims.ObjectId }); if (roles is null) { // Authenticated by the platform but not authorised here. Strip the header so nothing @@ -125,13 +121,8 @@ private static async Task TryNormalisePrincipalAsync( return false; } - context.Request.Headers["x-ms-client-principal"] = EasyAuthPrincipal.Encode(new - { - identityProvider = realIdp, - userId = claims.ObjectId ?? userName, - userDetails = userName, - userRoles = roles, - }); + context.Request.Headers["x-ms-client-principal"] = EasyAuthPrincipal.EncodeNormalised( + root, realIdp, claims.ObjectId ?? userName, userName, roles); context.Request.Headers["x-ms-client-principal-idp"] = "azureStaticWebApps"; context.Request.Headers["x-ms-client-principal-name"] = userName; diff --git a/tests/Craft.Tests/CraftAuthMiddlewareTests.cs b/tests/Craft.Tests/CraftAuthMiddlewareTests.cs new file mode 100644 index 0000000..4690d86 --- /dev/null +++ b/tests/Craft.Tests/CraftAuthMiddlewareTests.cs @@ -0,0 +1,152 @@ +using System.Text; +using System.Text.Json; +using Craft.Auth; +using Craft.Hosting; +using Microsoft.AspNetCore.Http; +using Microsoft.Extensions.Logging.Abstractions; + +namespace Craft.Tests; + +/// +/// Pins how an EasyAuth principal is rewritten for the hosted app. An app-only token must become an API +/// client (idp "aad", name = app id, no allowedUsers lookup); a signed-in user must be authorised against +/// allowedUsers and keep its own identity even when the token names the app it signed in through. Both +/// keep the token's claims. Getting the split wrong either locks out every API client or runs a user as +/// an app. +/// +public class CraftAuthMiddlewareTests +{ + private const string OidClaim = "http://schemas.microsoft.com/identity/claims/objectidentifier"; + private static readonly string[] AdminRoles = ["admin"]; + private static readonly string?[] UserIdentifiers = ["a@b.com", "user-oid"]; + + private static string EasyAuthHeader(params (string Typ, string Val)[] claims) + { + var items = string.Join(",", claims.Select(c => $$"""{"typ":"{{c.Typ}}","val":"{{c.Val}}"}""")); + return Convert.ToBase64String(Encoding.UTF8.GetBytes($$"""{"auth_typ":"aad","claims":[{{items}}]}""")); + } + + private static DefaultHttpContext Request(string principalHeader) + { + var context = new DefaultHttpContext(); + context.Request.Headers["x-ms-client-principal"] = principalHeader; + context.Request.Headers["x-ms-client-principal-idp"] = "aad"; + context.Response.Body = new MemoryStream(); + return context; + } + + private sealed class RoleLookup(string[]? roles) + { + public List Calls { get; } = []; + + public Task Resolve(IEnumerable ids) + { + Calls.Add([.. ids]); + return Task.FromResult(roles); + } + } + + private static async Task Normalise(HttpContext context, RoleLookup lookup) => + await CraftAuthMiddleware.TryNormalisePrincipalAsync( + context, lookup.Resolve, NullLogger.Instance, context.Request.Headers["x-ms-client-principal"].ToString()); + + private static JsonElement Principal(HttpContext context) + { + using var document = EasyAuthPrincipal.Decode(context.Request.Headers["x-ms-client-principal"].ToString()); + return document.RootElement.Clone(); + } + + private static string? Claim(JsonElement principal, string typ) => + principal.GetProperty("claims").EnumerateArray() + .Where(c => c.GetProperty("typ").GetString() == typ) + .Select(c => c.GetProperty("val").GetString()) + .FirstOrDefault(); + + [Fact] + public async Task AppOnlyToken_BecomesAnApiClient() + { + var lookup = new RoleLookup(AdminRoles); + var context = Request(EasyAuthHeader(("appid", "app-1"), (OidClaim, "sp-oid"), ("idtyp", "app"), ("azpacr", "1"))); + + Assert.True(await Normalise(context, lookup)); + + Assert.Equal("aad", context.Request.Headers["x-ms-client-principal-idp"].ToString()); + Assert.Equal("app-1", context.Request.Headers["x-ms-client-principal-name"].ToString()); + var principal = Principal(context); + Assert.Equal("app-1", principal.GetProperty("userDetails").GetString()); + Assert.Equal("sp-oid", principal.GetProperty("userId").GetString()); + Assert.Equal(0, principal.GetProperty("userRoles").GetArrayLength()); + Assert.Equal("1", Claim(principal, "azpacr")); + Assert.Empty(lookup.Calls); // an API client is never looked up in allowedUsers + } + + [Fact] + public async Task SignedInUser_IsAuthorisedAndKeepsItsOwnIdentity() + { + var lookup = new RoleLookup(AdminRoles); + var context = Request(EasyAuthHeader( + ("upn", "a@b.com"), (OidClaim, "user-oid"), ("azp", "client-abc"), ("scp", "user_impersonation"))); + + Assert.True(await Normalise(context, lookup)); + + // The token names the app the user signed in through, but the caller is the user, not that app. + Assert.Equal("azureStaticWebApps", context.Request.Headers["x-ms-client-principal-idp"].ToString()); + Assert.Equal("a@b.com", context.Request.Headers["x-ms-client-principal-name"].ToString()); + var principal = Principal(context); + Assert.Equal("a@b.com", principal.GetProperty("userDetails").GetString()); + Assert.Equal("user-oid", principal.GetProperty("userId").GetString()); + Assert.Equal("admin", principal.GetProperty("userRoles")[0].GetString()); + Assert.Equal("client-abc", Claim(principal, "azp")); + Assert.Equal("user_impersonation", Claim(principal, "scp")); + Assert.Equal(UserIdentifiers, Assert.Single(lookup.Calls)); + } + + [Fact] + public async Task UserNotInAllowedUsers_IsRejectedAndThePrincipalStripped() + { + var context = Request(EasyAuthHeader(("upn", "a@b.com"), ("azp", "client-abc"))); + + Assert.False(await Normalise(context, new RoleLookup(null))); + + Assert.Equal(StatusCodes.Status401Unauthorized, context.Response.StatusCode); + Assert.False(context.Request.Headers.ContainsKey("x-ms-client-principal")); + } + + [Fact] + public async Task NormalisedPrincipal_PassesThroughUntouched() + { + var normalised = EasyAuthPrincipal.EncodeNormalised( + JsonDocument.Parse("""{"claims":[{"typ":"upn","val":"a@b.com"}]}""").RootElement, + "aad", "user-oid", "a@b.com", AdminRoles); + var lookup = new RoleLookup(null); + var context = Request(normalised); + + Assert.True(await Normalise(context, lookup)); + + Assert.Equal(normalised, context.Request.Headers["x-ms-client-principal"].ToString()); + Assert.Empty(lookup.Calls); + } + + [Fact] + public async Task PrincipalWithNoIdentity_PassesThroughUntouched() + { + var header = EasyAuthHeader(("scp", "user_impersonation")); + var lookup = new RoleLookup(AdminRoles); + var context = Request(header); + + Assert.True(await Normalise(context, lookup)); + + Assert.Equal(header, context.Request.Headers["x-ms-client-principal"].ToString()); + Assert.Empty(lookup.Calls); + } + + [Fact] + public async Task UnparseablePrincipal_PassesThroughUntouched() + { + var context = Request("not-base64!!"); + + Assert.True(await Normalise(context, new RoleLookup(AdminRoles))); + + Assert.Equal("not-base64!!", context.Request.Headers["x-ms-client-principal"].ToString()); + } +} diff --git a/tests/Craft.Tests/EasyAuthPrincipalTests.cs b/tests/Craft.Tests/EasyAuthPrincipalTests.cs index 146d2d5..a04f7c8 100644 --- a/tests/Craft.Tests/EasyAuthPrincipalTests.cs +++ b/tests/Craft.Tests/EasyAuthPrincipalTests.cs @@ -210,4 +210,41 @@ public void Decode_RejectsGarbage() // The middleware catches this and passes the request through as anonymous rather than 500ing. Assert.ThrowsAny(() => EasyAuthPrincipal.Decode("not-base64!!")); } + + [Fact] + public void Normalised_KeepsTheTokenClaims() + { + // The hosted app reads claims beyond the identity (e.g. azp, the app a user signed in through). + var source = WithClaims(("upn", "a@b.com"), ("azp", "client-abc"), ("scp", "user_impersonation")); + + using var result = EasyAuthPrincipal.Decode( + EasyAuthPrincipal.EncodeNormalised(source, "aad", "oid-1", "a@b.com", AdminRole)); + var root = result.RootElement; + + Assert.Equal("a@b.com", root.GetProperty("userDetails").GetString()); + Assert.Equal("admin", root.GetProperty("userRoles")[0].GetString()); + Assert.Equal(source.GetProperty("claims").GetRawText(), root.GetProperty("claims").GetRawText()); + Assert.Equal("client-abc", EasyAuthPrincipal.ExtractClaims(root).AppId); + } + + [Fact] + public void Normalised_IsNeverTransformedAgain() + { + var source = WithClaims(("upn", "a@b.com")); + + using var result = EasyAuthPrincipal.Decode( + EasyAuthPrincipal.EncodeNormalised(source, "aad", "oid-1", "a@b.com", Array.Empty())); + + Assert.False(EasyAuthPrincipal.NeedsTransform(result.RootElement)); + } + + [Fact] + public void Normalised_WithoutClaims_EmitsAnEmptyArray() + { + using var result = EasyAuthPrincipal.Decode( + EasyAuthPrincipal.EncodeNormalised(Principal("{}"), "aad", "id", "name", Array.Empty())); + + Assert.Equal(JsonValueKind.Array, result.RootElement.GetProperty("claims").ValueKind); + Assert.Equal(0, result.RootElement.GetProperty("claims").GetArrayLength()); + } } From 2a30801312f0d9b0754ae203a9628cc0e8f30529 Mon Sep 17 00:00:00 2001 From: Zacgoose <107489668+Zacgoose@users.noreply.github.com> Date: Tue, 6 Oct 2026 17:18:27 +0800 Subject: [PATCH 23/24] feat(memory): trim memory on a timer and read memory detail from the last GC of any kind - Worker.MemoryTrimIntervalMinutes (default 5, 0 disables) runs TrimMemory periodically so idle hosts return freed heap - Memory breakdown uses GCKind.Any instead of FullBlocking, which reported zeros until a blocking gen2 ran --- Services/Bridges/WorkerMetricsBridge.cs | 4 +-- Services/Configuration/WorkerSettings.cs | 6 ++++ .../Hosting/CraftHostBuilderExtensions.cs | 2 ++ Services/Hosting/MemoryTrimService.cs | 28 +++++++++++++++++++ appsettings.example.jsonc | 3 ++ docs/configuration.md | 4 +++ 6 files changed, 45 insertions(+), 2 deletions(-) create mode 100644 Services/Hosting/MemoryTrimService.cs diff --git a/Services/Bridges/WorkerMetricsBridge.cs b/Services/Bridges/WorkerMetricsBridge.cs index 464b92b..8c99d64 100644 --- a/Services/Bridges/WorkerMetricsBridge.cs +++ b/Services/Bridges/WorkerMetricsBridge.cs @@ -442,7 +442,7 @@ private static double GetContainerCpuPct() public static MemoryBreakdown GetMemoryBreakdown() { var proc = Process.GetCurrentProcess(); - var gcInfo = GC.GetGCMemoryInfo(GCKind.FullBlocking); + var gcInfo = GC.GetGCMemoryInfo(GCKind.Any); var heapBytes = GC.GetTotalMemory(false); var workingSet = proc.WorkingSet64; var containerBytes = GetContainerMemoryLimit() ?? gcInfo.TotalAvailableMemoryBytes; @@ -823,7 +823,7 @@ private static void UpdateDurationStats(WorkerStats stats, long durationMs) /// /// Force a full GC collection with LOH compaction and working-set trim. - /// Called automatically every 100 invocations and after orchestrator runs complete. + /// Called every 100 invocations and on the MemoryTrimService timer (Worker.MemoryTrimIntervalMinutes). /// Has a built-in 2-minute cooldown to avoid GC thrashing. /// Returns the MB reclaimed, or -1 if skipped due to cooldown. /// diff --git a/Services/Configuration/WorkerSettings.cs b/Services/Configuration/WorkerSettings.cs index 3317079..f7299e0 100644 --- a/Services/Configuration/WorkerSettings.cs +++ b/Services/Configuration/WorkerSettings.cs @@ -166,6 +166,12 @@ public class WorkerSettings /// public int RecycleAfterInvocations { get; set; } + /// + /// Run a memory trim (compacting full GC) every this many minutes, so freed heap is handed back + /// to the OS on idle hosts that never reach the every-100-invocations trim. 0 = disabled. Default 5. + /// + public int MemoryTrimIntervalMinutes { get; set; } = 5; + /// /// Run each worker's PowerShell pipeline on one reused thread (PSThreadOptions.ReuseThread) instead of /// spinning a new thread per invocation. Default true. This is the single biggest per-request dispatch diff --git a/Services/Hosting/CraftHostBuilderExtensions.cs b/Services/Hosting/CraftHostBuilderExtensions.cs index 834b642..810ae73 100644 --- a/Services/Hosting/CraftHostBuilderExtensions.cs +++ b/Services/Hosting/CraftHostBuilderExtensions.cs @@ -321,6 +321,8 @@ public static IServiceCollection AddCraftServices(this IServiceCollection servic services.AddSingleton(); services.AddHostedService(sp => sp.GetRequiredService()); + services.AddHostedService(); + return services; } diff --git a/Services/Hosting/MemoryTrimService.cs b/Services/Hosting/MemoryTrimService.cs new file mode 100644 index 0000000..54ca4fb --- /dev/null +++ b/Services/Hosting/MemoryTrimService.cs @@ -0,0 +1,28 @@ +using Craft.Configuration; +using Craft.Orchestration; +using Craft.Services; + +namespace Craft.Hosting; + +/// Runs on a fixed interval (Worker.MemoryTrimIntervalMinutes). +public class MemoryTrimService(ILogger logger, CraftSettings settings) : BackgroundService +{ + protected override async Task ExecuteAsync(CancellationToken stoppingToken) + { + var minutes = settings.Worker.MemoryTrimIntervalMinutes; + if (minutes <= 0) return; + + try + { + using var timer = new PeriodicTimer(TimeSpan.FromMinutes(minutes)); + while (await timer.WaitForNextTickAsync(stoppingToken)) + { + var reclaimed = WorkerMetricsBridge.TrimMemory(); + if (reclaimed >= 0) + logger.LogInformation("[System] Memory trim (timer): reclaimed ~{MB}MB {Memory}", + reclaimed, BackgroundTaskLimiter.GetMemorySnapshot()); + } + } + catch (OperationCanceledException) when (stoppingToken.IsCancellationRequested) { } + } +} diff --git a/appsettings.example.jsonc b/appsettings.example.jsonc index 76198c7..521e0e8 100644 --- a/appsettings.example.jsonc +++ b/appsettings.example.jsonc @@ -173,6 +173,9 @@ // docs/dispatch-analysis.md). Set false only to A/B or if a module misbehaves on a long-lived thread. // "ReuseRunspaceThread": true, + // Minutes between timed memory trims (compacting full GC). 0 = disabled. Default 5. + // "MemoryTrimIntervalMinutes": 5, + // Host-tier pool sizing. When set, the first matching entry overrides // HttpPoolSize/BgPoolSize above based on the runtime environment. // SkuEnv: name of the env var to read for the host tier identifier diff --git a/docs/configuration.md b/docs/configuration.md index 780ba70..853d93c 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -179,6 +179,10 @@ Controls the PowerShell runspace pools that execute all scripts. // only to A/B or if a module misbehaves on a long-lived pipeline thread. "ReuseRunspaceThread": true, + // Minutes between timed memory trims (compacting full GC that hands freed heap back to the OS). + // Runs alongside the every-100-invocations trim, sharing its 2-minute cooldown. 0 = disabled. + "MemoryTrimIntervalMinutes": 5, + // Maximum execution time (seconds) for HTTP request handlers. // When exceeded, the PowerShell pipeline is stopped and the worker is reclaimed. // 0 = no timeout (default). Recommended: 120-300 for HTTP endpoints. From 0b1db915a6e51d3f483c0440c073066882ee88a3 Mon Sep 17 00:00:00 2001 From: Zacgoose <107489668+Zacgoose@users.noreply.github.com> Date: Tue, 6 Oct 2026 17:27:07 +0800 Subject: [PATCH 24/24] style(tests): sort usings in OrchestratorBridgeLineageTests --- tests/Craft.Tests/OrchestratorBridgeLineageTests.cs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs b/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs index cfbf5fb..46064b6 100644 --- a/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs +++ b/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs @@ -1,5 +1,5 @@ -using System.Collections.ObjectModel; using System.Collections.Concurrent; +using System.Collections.ObjectModel; using System.Management.Automation; using System.Management.Automation.Runspaces; using System.Reflection;