diff --git a/Runtime/CraftRuntime/Start-CraftOrchestrator.ps1 b/Runtime/CraftRuntime/Start-CraftOrchestrator.ps1 index 7e7daac..4a1ed1d 100644 --- a/Runtime/CraftRuntime/Start-CraftOrchestrator.ps1 +++ b/Runtime/CraftRuntime/Start-CraftOrchestrator.ps1 @@ -30,6 +30,13 @@ function Start-CraftOrchestrator { worker and runs every step on it to completion, without going back to the pool between steps. A step that fails is recorded and the run carries on with the next (best-effort). + - AllowCollision (bool) — optional, default $true: runs of one name stack up side by side. + $false skips this run while another run of the same name is still + going (recurring work that must not pile up). + - MaxConcurrency (int) — optional, default 0 (no limit): at most this many of the run's tasks + run at once. Not used with Sequential. + - StopOnFailure (bool) — optional, Sequential only: the first failed step cancels the steps + after it. By default a sequential run carries on past a failure. .EXAMPLE # Fan-out (default): every task is queued up front and drained in parallel by the worker pool. @@ -68,6 +75,13 @@ function Start-CraftOrchestrator { $OrchestratorName = $InputObject.OrchestratorName ?? 'UnnamedOrchestrator' + # Collisions off: a run of this name that is still going wins, and this one is skipped up front. + $AllowCollision = $InputObject.AllowCollision -ne $false + if (-not $AllowCollision -and [Craft.Services.OrchestratorBridge]::IsRunActive($OrchestratorName)) { + Write-Warning "Craft: Skipped orchestrator '$OrchestratorName' - a run with this name is still active" + return "Craft-$OrchestratorName-Skipped" + } + # QueueFunction pattern: call the function first to generate batch items if (-not $InputObject.Batch -and $InputObject.QueueFunction) { $QueueFuncName = "Push-$($InputObject.QueueFunction.FunctionName)" @@ -140,11 +154,20 @@ function Start-CraftOrchestrator { # Lineage: pass the enclosing run explicitly. The bridge's own ambient read is null for calls # made from the pipeline thread — which is exactly where this function runs — so without this # a parent run would finalize (and dispatch its PostExecution) before its child runs complete. - $ParentRunName = $OpContext.RunName + # RunKey names the exact run when several runs share a name. + $ParentRunName = $OpContext.RunKey ?? $OpContext.RunName # Sequential mode: PowerShell marshals absent/$false to $false. When set, the orchestrator queues the # batch one task at a time in payload order rather than fanning out. $Sequential = [bool]($InputObject.Sequential) + $MaxConcurrency = [int]($InputObject.MaxConcurrency ?? 0) + $StopOnFailure = [bool]($InputObject.StopOnFailure) + if ($Sequential -and $MaxConcurrency -gt 0) { + Write-Warning "Craft: MaxConcurrency is ignored for '$OrchestratorName': a sequential run already runs one step at a time" + } + if ($StopOnFailure -and -not $Sequential) { + Write-Warning "Craft: StopOnFailure is ignored for '$OrchestratorName': it applies to sequential runs only" + } Write-Information "Craft: Queuing orchestrator '$OrchestratorName' ($TaskCount tasks, P$Priority$(if ($Sequential) { ', Sequential' })$(if ($PostExecFunctionName) { ", PostExec: $PostExecFunctionName" })$(if ($ParentRunName) { ", Parent: $ParentRunName" }))" [Craft.Services.OrchestratorBridge]::QueueOrchestrationFromFile( @@ -155,7 +178,10 @@ function Start-CraftOrchestrator { $PostExecParametersJson, $InputObject.Reference, $ParentRunName, - $Sequential + $Sequential, + $AllowCollision, + $MaxConcurrency, + $StopOnFailure ) return "Craft-$OrchestratorName" } diff --git a/Services/Auth/EasyAuthPrincipal.cs b/Services/Auth/EasyAuthPrincipal.cs index 8be18f2..9eb4f3a 100644 --- a/Services/Auth/EasyAuthPrincipal.cs +++ b/Services/Auth/EasyAuthPrincipal.cs @@ -73,6 +73,26 @@ public static JsonDocument Decode(string headerValue) => public static string Encode(T principal) => Convert.ToBase64String(Encoding.UTF8.GetBytes(JsonSerializer.Serialize(principal))); + /// + /// Encodes the normalised SWA-format principal for an EasyAuth , keeping the + /// token's original claims so the hosted app can read any of them (e.g. azp, scp). + /// + /// + /// The claims are those of the token EasyAuth validated, and userRoles marks the result as + /// already normalised (), so it is never transformed twice. + /// + public static string EncodeNormalised( + JsonElement source, string identityProvider, string userId, string userDetails, IReadOnlyList userRoles) + { + var claims = source.ValueKind == JsonValueKind.Object && + source.TryGetProperty("claims", out var sourceClaims) && + sourceClaims.ValueKind == JsonValueKind.Array + ? sourceClaims.Clone() + : JsonDocument.Parse("[]").RootElement.Clone(); + + return Encode(new { identityProvider, userId, userDetails, userRoles, claims }); + } + /// /// Pulls the identity claims out of an EasyAuth principal. /// diff --git a/Services/Bridges/JobRecord.cs b/Services/Bridges/JobRecord.cs index 29f0771..5ae776e 100644 --- a/Services/Bridges/JobRecord.cs +++ b/Services/Bridges/JobRecord.cs @@ -13,6 +13,9 @@ public class JobRecord public string Id { get; set; } = string.Empty; public string Name { get; set; } = string.Empty; public string? RunName { get; set; } + + /// The outing of the run this job belongs to, when it came from a stored run. + public string? RunKey { get; set; } public int Priority { get; set; } public string Status { get; set; } = "Queued"; public DateTime QueuedUtc { get; set; } diff --git a/Services/Bridges/OrchestratorBridge.cs b/Services/Bridges/OrchestratorBridge.cs index 17ccc6d..c32cc07 100644 --- a/Services/Bridges/OrchestratorBridge.cs +++ b/Services/Bridges/OrchestratorBridge.cs @@ -23,19 +23,27 @@ public static class OrchestratorBridge public static void Initialize(OrchestratorService service) => s_service = service; + /// True (the default) lets runs of one name stack up; false skips this run + /// while another run of the same name is unfinished. + /// At most this many of the run's tasks run at once; 0 (the default) is no + /// limit. Ignored for a sequential run. + /// Sequential runs only: the first failed step cancels the rest instead of the + /// run carrying on (the default). public static void QueueOrchestration(string name, string batchJson, int priority, string? postExecFunctionName = null, string? postExecParametersJson = null, - string? reference = null, string? parentRunName = null, bool sequential = false) + string? reference = null, string? parentRunName = null, bool sequential = false, bool allowCollision = true, + int maxConcurrency = 0, bool stopOnFailure = false) { // Sanitized here as well as at run creation so the child-run registration below // records the SAME name the service ends up creating — a raw name with a table-illegal // character would register a child link no live run ever matches. name = TableKeys.Sanitize(name); parentRunName = ResolveParentRunName(name, parentRunName); - var gated = RegisterPendingChild(parentRunName, name); + var child = RegisterPendingChild(parentRunName, name); s_pending.Enqueue(new PendingOrchestration(name, batchJson, priority, postExecFunctionName, postExecParametersJson, parentRunName, reference, - PendingChildRegistered: gated, Sequential: sequential)); + Sequential: sequential, ParentRunKey: child?.ParentRunKey, ChildKey: child?.ChildKey, + AllowCollision: allowCollision, MaxConcurrency: maxConcurrency, StopOnFailure: stopOnFailure)); } /// @@ -49,20 +57,29 @@ public static void QueueOrchestration(string name, string batchJson, int priorit /// /// The file is owned by the orchestrator from this point: it is deleted once parsed. /// + /// True (the default) lets runs of one name stack up; false skips this run + /// while another run of the same name is unfinished. + /// At most this many of the run's tasks run at once; 0 (the default) is no + /// limit. Ignored for a sequential run. + /// Sequential runs only: the first failed step cancels the rest instead of the + /// run carrying on (the default). public static void QueueOrchestrationFromFile(string name, string batchFilePath, int priority, string? postExecFunctionName = null, string? postExecParametersJson = null, - string? reference = null, string? parentRunName = null, bool sequential = false) + string? reference = null, string? parentRunName = null, bool sequential = false, bool allowCollision = true, + int maxConcurrency = 0, bool stopOnFailure = false) { name = TableKeys.Sanitize(name); parentRunName = ResolveParentRunName(name, parentRunName); - var gated = RegisterPendingChild(parentRunName, name); + var child = RegisterPendingChild(parentRunName, name); s_pending.Enqueue(new PendingOrchestration(name, string.Empty, priority, postExecFunctionName, postExecParametersJson, parentRunName, reference, batchFilePath, - PendingChildRegistered: gated, Sequential: sequential)); + Sequential: sequential, ParentRunKey: child?.ParentRunKey, ChildKey: child?.ChildKey, + AllowCollision: allowCollision, MaxConcurrency: maxConcurrency, StopOnFailure: stopOnFailure)); } /// - /// Resolve the parent run of a queued orchestration. The explicit argument wins — PowerShell + /// Resolve the parent run of a queued orchestration: a run key (exact — runs of one name can overlap) or a + /// run name (the newest outing). The explicit argument wins — PowerShell /// callers MUST pass it (read from the stamped $global:CraftOperationContext), because the /// ambient fallback cannot work for them: the pipeline runs on the runspace's reused thread, /// whose frozen ExecutionContext never sees the per-invocation AsyncLocal (see @@ -75,7 +92,7 @@ public static void QueueOrchestrationFromFile(string name, string batchFilePath, private static string? ResolveParentRunName(string name, string? parentRunName) { if (string.IsNullOrEmpty(parentRunName)) - parentRunName = OperationContext.Current?.RunName; + parentRunName = OperationContext.Current?.RunKey ?? OperationContext.Current?.RunName; if (string.IsNullOrEmpty(parentRunName)) return null; // Sanitized like the child name: the parent was created under its sanitized name, and the @@ -88,76 +105,87 @@ public static void QueueOrchestrationFromFile(string name, string batchFilePath, } /// - /// Register the child link at ENQUEUE time — while the parent task's script is still executing, - /// so the parent cannot pass its completion check before the gate exists. Registering after - /// StartFromBatchAsync (the old shape) loses that race for the parent's LAST task: the drain - /// runs in a background Task.Run while the enqueuing task is marked terminal immediately, so - /// the parent would finalize — and dispatch PostExecution — before its child was visible. - /// Returns whether a gate was taken, so the drain releases exactly what was registered. + /// Make the parent wait for this child, at ENQUEUE time — while the parent's task is still executing, so + /// the parent cannot reach its barrier first. The returned keys travel with the queued run: the child + /// fills the placeholder when it finishes, and a child that is never created releases it on the drain. /// - private static bool RegisterPendingChild(string? parentRunName, string childName) => - !string.IsNullOrEmpty(parentRunName) && - s_service?.TryRegisterPendingChildRun(parentRunName, childName) == true; + private static (string ParentRunKey, string ChildKey)? RegisterPendingChild(string? parentRunName, string childName) => + string.IsNullOrEmpty(parentRunName) ? null : s_service?.RegisterPendingChild(parentRunName, childName); /// - /// Synchronous drain — blocks until all pending orchestrations are started. - /// Safe to call from any context (no SynchronizationContext on background workers). + /// Whether a run of this name is unfinished or already queued here to start, counting outings that carry a + /// queue id suffix (Name-{guid}) as the same name. Lets a caller that does not want overlapping runs + /// (allowCollision: false) skip, and say so, before building the batch. The start itself checks + /// again, so a run that appears in between is still skipped. + /// PS usage: [Craft.Services.OrchestratorBridge]::IsRunActive($name). /// + public static bool IsRunActive(string name) + { + var family = WorkStore.CollisionFamily(TableKeys.Sanitize(name)); + if (s_pending.Any(p => WorkStore.CollisionFamily(TableKeys.Sanitize(p.Name)) == family)) return true; + return s_service != null && Task.Run(() => s_service.IsRunActiveAsync(name)).GetAwaiter().GetResult(); + } + + private static readonly System.Text.Json.JsonSerializerOptions s_inspectJson = new() { WriteIndented = true }; + + /// + /// Why a run is or is not moving, as JSON: counts and mode, whether the scheduler can see it, its claims + /// and who holds them, child runs it waits for, its aggregation, the instance lock, and a diagnosis. + /// Takes a run key, or a run name (every unfinished run of it, else the latest). + /// PS usage: [Craft.Services.OrchestratorBridge]::InspectRun('MailboxRules_contoso.com'). + /// + public static string InspectRun(string nameOrKey) => s_service == null + ? "{\"error\":\"orchestrator not initialised\"}" + : System.Text.Json.JsonSerializer.Serialize(Task.Run(() => s_service.InspectRunAsync(nameOrKey)).GetAwaiter().GetResult(), s_inspectJson); + + /// + /// Rebuild the Ready and Finished indexes from the active-run list now, as the pump does at startup: relists + /// unfinished runs, retires finished ones, removes runs whose creation never finished. Returns a JSON summary. + /// PS usage: [Craft.Services.OrchestratorBridge]::RepairIndexes(). + /// + public static string RepairIndexes() => s_service == null + ? "{\"error\":\"orchestrator not initialised\"}" + : System.Text.Json.JsonSerializer.Serialize(Task.Run(() => s_service.RepairIndexesAsync()).GetAwaiter().GetResult(), s_inspectJson); + + /// Synchronous drain — blocks until all pending orchestrations are started. public static void DrainPending() { while (s_pending.TryDequeue(out var p)) - { - try - { - if (s_service == null) { DiscardUndispatchable(p); continue; } - s_service.StartFromBatchAsync(p.Name, p.BatchJson, p.Priority, - p.PostExecFunctionName, p.PostExecParametersJson, CancellationToken.None, - p.ParentRunName, p.Reference, p.BatchFilePath, p.Sequential) - .GetAwaiter().GetResult(); - } - catch (Exception ex) - { - s_service?._logger.LogError(ex, "[Orchestrator] DrainPending failed for {Name}", p.Name); - } - finally - { - // The enqueue-time gate lifts on EVERY path once the start attempt is over: a - // started child is in _activeRuns by now (which takes over blocking the parent), - // and one that failed to start must stop blocking — a leaked gate would defer the - // parent's finalize forever, re-checked every 60s for the process lifetime. - if (p.PendingChildRegistered) - s_service?.ReleasePendingChildRun(p.Name); - } - } + Task.Run(() => StartAsync(p)).GetAwaiter().GetResult(); DrainPendingPlanners(); } - /// - /// Async drain — preferred from async call sites (PostExec lambdas, ExecuteScript). - /// + /// Async drain — preferred from async call sites (PostExec, ExecuteScript). public static async Task DrainPendingAsync() { while (s_pending.TryDequeue(out var p)) + await StartAsync(p); + await DrainPendingPlannersAsync(); + } + + private static async Task StartAsync(PendingOrchestration p) + { + var created = false; + try { - try - { - if (s_service == null) { DiscardUndispatchable(p); continue; } - await s_service.StartFromBatchAsync(p.Name, p.BatchJson, p.Priority, - p.PostExecFunctionName, p.PostExecParametersJson, CancellationToken.None, - p.ParentRunName, p.Reference, p.BatchFilePath, p.Sequential); - } - catch (Exception ex) - { - s_service?._logger.LogError(ex, "[Orchestrator] DrainPending failed for {Name}", p.Name); - } - finally + if (s_service == null) { DiscardUndispatchable(p); return; } + created = await s_service.StartFromBatchAsync(p.Name, p.BatchJson, p.Priority, + p.PostExecFunctionName, p.PostExecParametersJson, CancellationToken.None, + p.ParentRunName, p.Reference, p.BatchFilePath, p.Sequential, p.ParentRunKey, p.ChildKey, p.AllowCollision, + p.MaxConcurrency, p.StopOnFailure); + } + catch (Exception ex) + { + s_service?._logger.LogError(ex, "[Orchestrator] DrainPending failed for {Name}", p.Name); + } + finally + { + if (!created && p.ParentRunKey != null && p.ChildKey != null && s_service != null) { - // See DrainPending: the gate lifts whatever the outcome of the start attempt. - if (p.PendingChildRegistered) - s_service?.ReleasePendingChildRun(p.Name); + try { await s_service.AbandonPendingChildAsync(p.ParentRunKey, p.ChildKey); } + catch (Exception ex) { s_service._logger.LogWarning(ex, "[Orchestrator] Could not release {Name} from its parent", p.Name); } } } - await DrainPendingPlannersAsync(); } /// @@ -179,15 +207,17 @@ private static void DiscardUndispatchable(PendingOrchestration p) /// /// A queued run. Exactly one of and - /// carries the batch; the file path wins when both are set. - /// records whether enqueue took a pending-child gate on - /// the orchestrator, so the drain releases exactly the gates that were taken — releasing on a - /// refused registration could lift a gate held by ANOTHER queued entry of the same child name. + /// carries the batch; the file path wins when both are set. and + /// are set when the parent was made to wait for this run. /// public record PendingOrchestration(string Name, string BatchJson, int Priority, string? PostExecFunctionName, string? PostExecParametersJson, string? ParentRunName, - string? Reference = null, string? BatchFilePath = null, - bool PendingChildRegistered = false, bool Sequential = false); + string? Reference = null, string? BatchFilePath = null, bool Sequential = false, + string? ParentRunKey = null, string? ChildKey = null, bool AllowCollision = true, int MaxConcurrency = 0, + bool StopOnFailure = false) + { + public bool PendingChildRegistered => ChildKey != null; + } private static readonly ConcurrentQueue s_pendingPlanners = new(); diff --git a/Services/Bridges/QueueStatusBridge.cs b/Services/Bridges/QueueStatusBridge.cs index f3bd14b..47db101 100644 --- a/Services/Bridges/QueueStatusBridge.cs +++ b/Services/Bridges/QueueStatusBridge.cs @@ -190,7 +190,7 @@ private static List GetTaskDetails(string runName) { if (s_jobManager == null) return []; - var jobs = s_jobManager.GetJobs(runName, limit: 100); + var jobs = s_jobManager.GetRunJobs(runName, limit: 100); return jobs.Select(j => new TaskDetail { Timestamp = (j.CompletedUtc ?? j.StartedUtc ?? j.QueuedUtc).ToString("O"), diff --git a/Services/Bridges/WorkerMetricsBridge.cs b/Services/Bridges/WorkerMetricsBridge.cs index c095126..8c99d64 100644 --- a/Services/Bridges/WorkerMetricsBridge.cs +++ b/Services/Bridges/WorkerMetricsBridge.cs @@ -223,12 +223,13 @@ public static WorkerMetricsSnapshot GetSnapshot() // the durable queue table. GetCached never blocks: a stale/missing snapshot kicks off a // background refresh and this poll reports what is known now. var durable = s_queueReader?.GetCached(); + var waiting = durable == null ? 0 : s_queueReader!.WaitingInStorage(durable); snapshot.Jobs = new JobMetrics { - Queued = summary.Queued + (durable?.Unclaimed ?? 0), + Queued = summary.Queued + waiting, QueuedLocal = summary.Queued, - QueuedDurable = durable?.Unclaimed ?? 0, + QueuedDurable = waiting, Running = summary.Running, Completed = summary.Completed, Failed = summary.Failed, @@ -441,7 +442,7 @@ private static double GetContainerCpuPct() public static MemoryBreakdown GetMemoryBreakdown() { var proc = Process.GetCurrentProcess(); - var gcInfo = GC.GetGCMemoryInfo(GCKind.FullBlocking); + var gcInfo = GC.GetGCMemoryInfo(GCKind.Any); var heapBytes = GC.GetTotalMemory(false); var workingSet = proc.WorkingSet64; var containerBytes = GetContainerMemoryLimit() ?? gcInfo.TotalAvailableMemoryBytes; @@ -590,7 +591,8 @@ _ when name.Contains("PowerShell", StringComparison.OrdinalIgnoreCase) || /// Get a summary of just the busy/available counts. public static WorkerSummary GetSummary() { - var durable = s_queueReader?.GetCached(); + var snapshot = s_queueReader?.GetCached(); + var waiting = snapshot == null ? 0 : s_queueReader!.WaitingInStorage(snapshot); var localQueued = s_jobManager?.QueuedCount ?? 0; return new WorkerSummary @@ -605,9 +607,9 @@ public static WorkerSummary GetSummary() LimiterWaiting = s_limiter?.Waiting ?? 0, LimiterMax = s_limiter?.CurrentMax ?? 0, IsHttpThrottled = s_limiter?.IsHttpThrottled ?? false, - JobsQueued = localQueued + (durable?.Unclaimed ?? 0), + JobsQueued = localQueued + waiting, JobsQueuedLocal = localQueued, - JobsQueuedDurable = durable?.Unclaimed ?? 0, + JobsQueuedDurable = waiting, JobsActive = s_jobManager?.ActiveCount ?? 0, }; } @@ -660,7 +662,7 @@ public static bool CancelJob(string jobId) var row = snap?.Rows.FirstOrDefault(r => !r.Claimed && $"{r.RunName}-{r.TaskId}" == jobId); if (row == null) return false; - return await s_orchestrator.TryCancelQueuedTaskAsync(row.RunName, row.TaskId); + return await s_orchestrator.TryCancelQueuedTaskAsync(row.RunKey, row.Seq); }); } @@ -693,10 +695,8 @@ public static bool DeleteJob(string jobId) => s_jobManager?.DeleteJob(jobId) ?? false; /// - /// Empty the durable job queue — a maintenance/reset primitive. Returns the number of queue rows - /// removed, or -1 if the orchestrator is unavailable or the clear failed. In-flight work is - /// unaffected and Pending tasks may be re-driven, so pair with when the - /// intent is to STOP work rather than clear a wedged or corrupted queue. + /// Empty the durable job queue — cancel every task still waiting in storage. Returns how many were + /// cancelled, or -1 if the orchestrator is unavailable or the clear failed. Running tasks finish. /// PS usage: [Craft.Services.WorkerMetricsBridge]::ClearQueue(). /// public static int ClearQueue() @@ -714,28 +714,20 @@ public static int ClearQueue() } /// - /// Change a queued job's priority. In the local buffer this re-enqueues at the new priority; for an - /// unclaimed durable row it moves the row to the new priority bucket (keeping its age) and records - /// the override on the task so a restart re-queues it at the operator's priority. + /// Change a queued job's priority. In the local buffer this re-enqueues at the new priority; for a task + /// still in storage it moves the task's whole run to the new priority band, since a run's tasks share + /// one queue position. /// public static bool ChangePriority(string jobId, int newPriority) { if (s_jobManager?.ChangePriority(jobId, newPriority) == true) return true; - if (s_queueReader == null) return false; + if (s_queueReader == null || s_orchestrator == null) return false; return RunBridged(async ct => { var snap = await s_queueReader.GetAsync(TimeSpan.FromSeconds(2), ct); var row = snap?.Rows.FirstOrDefault(r => !r.Claimed && $"{r.RunName}-{r.TaskId}" == jobId); - if (row == null) return false; - - var moved = await s_queueReader.Queue.ReprioritizeTaskAsync(row.RunName, row.TaskId, newPriority, ct); - if (moved == 0) return false; - - // Best-effort durability of the override itself: only effective where the run is live, - // which on the dispatching node it is. The row move above is what changes dispatch order. - s_orchestrator?.PriorityChanged(new JobDescriptor(row.RunName, row.TaskId, row.Priority), newPriority); - return true; + return row != null && await s_orchestrator.ReprioritizeRunAsync(row.RunName, newPriority); }); } @@ -831,7 +823,7 @@ private static void UpdateDurationStats(WorkerStats stats, long durationMs) /// /// Force a full GC collection with LOH compaction and working-set trim. - /// Called automatically every 100 invocations and after orchestrator runs complete. + /// Called every 100 invocations and on the MemoryTrimService timer (Worker.MemoryTrimIntervalMinutes). /// Has a built-in 2-minute cooldown to avoid GC thrashing. /// Returns the MB reclaimed, or -1 if skipped due to cooldown. /// diff --git a/Services/Configuration/OrchestratorSettings.cs b/Services/Configuration/OrchestratorSettings.cs index 8c97e36..acb79cb 100644 --- a/Services/Configuration/OrchestratorSettings.cs +++ b/Services/Configuration/OrchestratorSettings.cs @@ -1,74 +1,16 @@ namespace Craft.Configuration; /// -/// Orchestrator settings — fan-out/fan-in task execution with crash recovery. +/// Orchestrator settings — fan-out/fan-in task execution, with all state in storage. /// public class OrchestratorSettings { /// - /// Prefix for the three Azure Tables used by the orchestrator. - /// Tables created: {Prefix}Runs, {Prefix}Tasks, {Prefix}Results. + /// Prefix for the orchestrator's Azure Tables. + /// Tables created: {Prefix}Work, {Prefix}Ready, {Prefix}Names, {Prefix}Finished, {Prefix}TaskResults. /// public string TablePrefix { get; set; } = "Orchestrator"; - /// - /// Batch and coalesce per-task/run status writes through OrchestratorStatusWriter instead of writing each - /// individually. Removes the per-task Azure Table write from the fan-out critical path (the throughput - /// ceiling — see docs/orch-analysis.md). Default true. Results are never batched (their chunking path is - /// untouched). Set false to fall back to the original per-task writes (for A/B). - /// - public bool BatchStatusWrites { get; set; } = true; - - /// - /// Coalesce SMALL task results (those that fit one Azure Table property) through the batched status - /// writer instead of a per-task upsert on the fan-out critical path. Each result is written BEFORE - /// its task's terminal marker in the same flush, so a result is always durable before the task is - /// counted done (and therefore before finalize/post-execution reads it). Large results keep the - /// directly-awaited chunked path. Default true; only applies when is - /// also true. Set false to fall back to the original per-task awaited result write. - /// - public bool BatchResultWrites { get; set; } = true; - - /// - /// When batching status writes, write the pre-invoke "Running" marker under a synchronous barrier so it - /// is durable BEFORE the task invokes (batched with other concurrently-starting tasks). Preserves the - /// AttemptCount/MaxRetries poison-task guarantee. Default true. False = eventual (faster, weaker: the - /// marker rides the periodic flush, so a host crash within the flush window may not advance AttemptCount). - /// - public bool DurableRunningBarrier { get; set; } = true; - - /// How often (ms) the status writer flushes coalesced writes. Also the barrier latency ceiling. - /// Default 25. - public int StatusFlushIntervalMs { get; set; } = 25; - - /// - /// How long a task will wait for its durable "Running" marker before giving up (seconds, default 90). - /// - /// This wait sits between the JobManager dispatching a task and that task checking out a worker, so - /// an unbounded one is a whole-host outage: production wedged for 101 and 75 minutes with all 8 - /// limiter slots held by tasks blocked here, every BG worker idle, and the heap at 24% of its cap. - /// On timeout the task is deferred, NOT failed — the marker never landed, so storage still has it - /// Pending and it is retried. Must exceed , since a waiter - /// may need the in-flight flush to finish plus one more. - /// - public int RunningBarrierTimeoutSeconds { get; set; } = 90; - - /// - /// Ceiling on one flush of coalesced status writes (seconds, default 30). The drain loop is a single - /// loop, so an unbounded storage call inside a flush stops every status write and every barrier - /// waiter in the process. Writes that do not complete are put back and retried on the next flush. - /// - public int StatusFlushTimeoutSeconds { get; set; } = 30; - - /// - /// How many per-run status writes may be in flight within one flush (default 8). - /// - /// Writes are grouped by run because a batch must share a partition key. The workload that broke - /// this was ~600 runs of ONE task each (one per tenant), which turned a "batch" into hundreds of - /// sequential round-trips inside a single flush. 1 restores the old sequential behaviour. - /// - public int StatusFlushConcurrency { get; set; } = 8; - /// /// PowerShell function used to execute individual orchestrator tasks. /// Receives a hashtable with task parameters. @@ -87,44 +29,25 @@ public class OrchestratorSettings /// public string PostExecFunction { get; set; } = "Invoke-CraftPostExecution"; - /// Maximum number of times a task can be interrupted before being marked Failed. + /// Maximum number of times a task can be interrupted before being marked Failed, and how many + /// times a run's PostExecution is attempted. public int MaxRetries { get; set; } = 3; /// - /// How long a run's rows outlive it (hours, default 48). A run that finished — or that nothing is - /// driving and that last wrote to storage — longer ago than this is removed from all three tables, - /// together with any Tasks/Results partition whose Run row is already gone. Craft itself needs the - /// rows only while a run is live; they stay this long for operators reading recent history. + /// How long a finished run's rows outlive it (hours, default 48). Craft needs them only while the run + /// is live; they stay this long for operators reading recent history. /// public int RetentionHours { get; set; } = 48; /// /// How often the retention sweep runs after the one at startup (hours, default 4). 0 disables the - /// periodic sweep; the startup pass, which follows crash recovery, still runs. + /// periodic sweep; the startup pass still runs. /// public int CleanupIntervalHours { get; set; } = 4; /// - /// The per-run status/re-drive tick cadence (seconds, default 60). Each live run has a timer firing at - /// this interval to log status, re-drive orphaned tasks, and re-check completion. It is also the floor - /// of the re-drive backoff. Raising it cuts per-run overhead at high live-run counts; the perf harness - /// lowers it to exercise the backoff in compressed time. Minimum 1s. + /// How often each active run's status line is considered for logging (seconds, default 60). A line is + /// written when the run's counts change, or every ten minutes when they do not. Minimum 1s. /// public int StatusTimerIntervalSeconds { get; set; } = 60; - - /// - /// Whether the per-run re-drive backs off geometrically once it has verified a run has no orphaned - /// tasks (default true). When false the re-drive verifies against storage on every tick — the old - /// behaviour, retained as a safety switch and for A/B measurement of the backoff's effect. - /// - public bool RedriveBackoff { get; set; } = true; - - /// - /// Whether a task sheds its Parameters payload from the in-memory run graph once it is durably - /// persisted and enqueued, rehydrating it from the Tasks table at dispatch (default true). This bounds - /// the retained memory of a large pending backlog — thousands of runs each holding every task's payload - /// is what drives the live-set toward the GC heap ceiling — at the cost of one point read per task at - /// dispatch. False keeps the payload resident the whole time (the old behaviour), for A/B or safety. - /// - public bool ShedPendingParameters { get; set; } = true; } diff --git a/Services/Configuration/WorkerSettings.cs b/Services/Configuration/WorkerSettings.cs index 3317079..f7299e0 100644 --- a/Services/Configuration/WorkerSettings.cs +++ b/Services/Configuration/WorkerSettings.cs @@ -166,6 +166,12 @@ public class WorkerSettings /// public int RecycleAfterInvocations { get; set; } + /// + /// Run a memory trim (compacting full GC) every this many minutes, so freed heap is handed back + /// to the OS on idle hosts that never reach the every-100-invocations trim. 0 = disabled. Default 5. + /// + public int MemoryTrimIntervalMinutes { get; set; } = 5; + /// /// Run each worker's PowerShell pipeline on one reused thread (PSThreadOptions.ReuseThread) instead of /// spinning a new thread per invocation. Default true. This is the single biggest per-request dispatch diff --git a/Services/Hosting/CraftAuthMiddleware.cs b/Services/Hosting/CraftAuthMiddleware.cs index 8a3cb6c..c24152c 100644 --- a/Services/Hosting/CraftAuthMiddleware.cs +++ b/Services/Hosting/CraftAuthMiddleware.cs @@ -51,7 +51,7 @@ public static WebApplication UseCraftAuth( if (hasPrincipal) { // Returns false when the caller was rejected and a response has already been written. - if (!await TryNormalisePrincipalAsync(context, authService, logger, existingHeader.ToString())) + if (!await TryNormalisePrincipalAsync(context, ids => authService.GetUserRoles(ids), logger, existingHeader.ToString())) return; } else if (isDevelopment) @@ -68,9 +68,10 @@ public static WebApplication UseCraftAuth( return app; } + /// Looks a signed-in user up in the allowedUsers table by its identifiers. /// if the request was rejected and the response is already written. - private static async Task TryNormalisePrincipalAsync( - HttpContext context, AuthService authService, ILogger logger, string headerValue) + internal static async Task TryNormalisePrincipalAsync( + HttpContext context, Func, Task> resolveUserRoles, ILogger logger, string headerValue) { try { @@ -91,13 +92,8 @@ private static async Task TryNormalisePrincipalAsync( // Service principal. The idp header MUST stay "aad": the hosted app keys off it to treat // the caller as an API client and resolve its name from the ApiClients table. The real // provider goes in identityProvider for audit only. - context.Request.Headers["x-ms-client-principal"] = EasyAuthPrincipal.Encode(new - { - identityProvider = realIdp, - userId = claims.ObjectId ?? claims.AppId, - userDetails = claims.AppId, - userRoles = Array.Empty(), - }); + context.Request.Headers["x-ms-client-principal"] = EasyAuthPrincipal.EncodeNormalised( + root, realIdp, claims.ObjectId ?? claims.AppId, claims.AppId, Array.Empty()); context.Request.Headers["x-ms-client-principal-idp"] = "aad"; context.Request.Headers["x-ms-client-principal-name"] = claims.AppId; return true; @@ -114,7 +110,7 @@ private static async Task TryNormalisePrincipalAsync( // Resolve roles by the display name AND the stable object id (a GitHub user's numeric id, // an Entra user's oid), so an allowedUsers row keyed on either one grants the user its roles. - var roles = await authService.GetUserRoles(new[] { userName, claims.ObjectId }); + var roles = await resolveUserRoles(new[] { userName, claims.ObjectId }); if (roles is null) { // Authenticated by the platform but not authorised here. Strip the header so nothing @@ -125,13 +121,8 @@ private static async Task TryNormalisePrincipalAsync( return false; } - context.Request.Headers["x-ms-client-principal"] = EasyAuthPrincipal.Encode(new - { - identityProvider = realIdp, - userId = claims.ObjectId ?? userName, - userDetails = userName, - userRoles = roles, - }); + context.Request.Headers["x-ms-client-principal"] = EasyAuthPrincipal.EncodeNormalised( + root, realIdp, claims.ObjectId ?? userName, userName, roles); context.Request.Headers["x-ms-client-principal-idp"] = "azureStaticWebApps"; context.Request.Headers["x-ms-client-principal-name"] = userName; diff --git a/Services/Hosting/CraftHostBuilderExtensions.cs b/Services/Hosting/CraftHostBuilderExtensions.cs index ab3f637..810ae73 100644 --- a/Services/Hosting/CraftHostBuilderExtensions.cs +++ b/Services/Hosting/CraftHostBuilderExtensions.cs @@ -275,15 +275,15 @@ public static IServiceCollection AddCraftServices(this IServiceCollection servic services.AddSingleton(); services.AddSingleton(); - services.AddSingleton(); - // The durable job queue. Registered alongside the orchestrator store because it shares the - // same ICraftTableStore and therefore the same bounded connection pool. - services.AddSingleton(); + // Durable orchestration state and task results, over the shared ICraftTableStore (and its + // bounded connection pool). + services.AddSingleton(_ => new PartitionRateLimiter()); + services.AddSingleton(); + services.AddSingleton(); // Table-backed queue view for the status APIs — the JobManager only buffers a worker-pool-sized // slice of the backlog, so status must read the tables. All roles: an HTTP-only node serves the // worker-health endpoint for work that runs elsewhere. services.AddSingleton(); - services.AddSingleton(); services.AddSingleton(); services.AddSingleton(); services.AddSingleton(); @@ -297,10 +297,10 @@ public static IServiceCollection AddCraftServices(this IServiceCollection servic if (roles.Background) { services.AddHostedService(sp => sp.GetRequiredService()); - // Feeds the JobManager from the durable queue a batch at a time, so the backlog lives - // in storage rather than in this process. - services.AddSingleton(); - services.AddHostedService(sp => sp.GetRequiredService()); + // Feeds the JobManager from storage a batch at a time, so the backlog lives in storage + // rather than in this process. + services.AddSingleton(); + services.AddHostedService(sp => sp.GetRequiredService()); services.AddHostedService(sp => sp.GetRequiredService()); services.AddHostedService(sp => sp.GetRequiredService()); } @@ -321,6 +321,8 @@ public static IServiceCollection AddCraftServices(this IServiceCollection servic services.AddSingleton(); services.AddHostedService(sp => sp.GetRequiredService()); + services.AddHostedService(); + return services; } diff --git a/Services/Hosting/MemoryTrimService.cs b/Services/Hosting/MemoryTrimService.cs new file mode 100644 index 0000000..54ca4fb --- /dev/null +++ b/Services/Hosting/MemoryTrimService.cs @@ -0,0 +1,28 @@ +using Craft.Configuration; +using Craft.Orchestration; +using Craft.Services; + +namespace Craft.Hosting; + +/// Runs on a fixed interval (Worker.MemoryTrimIntervalMinutes). +public class MemoryTrimService(ILogger logger, CraftSettings settings) : BackgroundService +{ + protected override async Task ExecuteAsync(CancellationToken stoppingToken) + { + var minutes = settings.Worker.MemoryTrimIntervalMinutes; + if (minutes <= 0) return; + + try + { + using var timer = new PeriodicTimer(TimeSpan.FromMinutes(minutes)); + while (await timer.WaitForNextTickAsync(stoppingToken)) + { + var reclaimed = WorkerMetricsBridge.TrimMemory(); + if (reclaimed >= 0) + logger.LogInformation("[System] Memory trim (timer): reclaimed ~{MB}MB {Memory}", + reclaimed, BackgroundTaskLimiter.GetMemorySnapshot()); + } + } + catch (OperationCanceledException) when (stoppingToken.IsCancellationRequested) { } + } +} diff --git a/Services/Hosting/OperationContext.cs b/Services/Hosting/OperationContext.cs index cfe22b6..05ba74e 100644 --- a/Services/Hosting/OperationContext.cs +++ b/Services/Hosting/OperationContext.cs @@ -58,6 +58,12 @@ public sealed class Invocation /// public string? RunName { get; init; } + /// + /// Storage key of the enclosing run. Runs of one name can overlap, so this — not + /// — is what identifies the parent exactly when a task queues a child. + /// + public string? RunKey { get; init; } + /// /// Queue priority of the enclosing run, exposed so nested enqueues can inherit it. /// PowerShell cannot read this statically — the pipeline thread never sees the AsyncLocal diff --git a/Services/Orchestration/FinishBatcher.cs b/Services/Orchestration/FinishBatcher.cs new file mode 100644 index 0000000..2090070 --- /dev/null +++ b/Services/Orchestration/FinishBatcher.cs @@ -0,0 +1,104 @@ +using Craft.Storage; + +namespace Craft.Orchestration; + +/// +/// Coalesces task finishes per run, so tasks of one run finishing together share one transaction instead of +/// each paying for its own (and racing each other on the run header). Callers await their own outcome. +/// +/// A finish that cannot be written (storage unavailable, or losing races for longer than the store retries) +/// is not dropped: it is retried in the background, backing off to a minute between attempts, for +/// . Each attempt is guarded by the claim's owner, so a claim taken over meanwhile is +/// never overwritten. Only if every attempt fails does the task fall back to its claim lapsing and running +/// again, and that is logged as an error naming the run. +/// +public sealed class FinishBatcher(WorkStore store, ILogger logger, TimeSpan? window = null) +{ + private readonly TimeSpan _window = window ?? TimeSpan.FromMilliseconds(15); + private readonly object _lock = new(); + private readonly Dictionary Done)>> _pending = new(StringComparer.Ordinal); + + /// How long a finish that cannot be written keeps being retried. + internal TimeSpan GiveUpAfter { get; set; } = TimeSpan.FromMinutes(30); + + /// The first background retry delay, doubling to a minute; tests shorten it. + internal TimeSpan FirstRetry { get; set; } = TimeSpan.FromSeconds(1); + + /// Finishes waiting on a background retry, for status and tests. + public int Retrying => Volatile.Read(ref _retrying); + private int _retrying; + + public Task FinishAsync(string runKey, WorkStore.Finish finish) + { + var done = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + bool first; + lock (_lock) + { + first = !_pending.TryGetValue(runKey, out var list); + if (first) _pending[runKey] = list = []; + list!.Add((finish, done)); + } + if (first) _ = FlushAfterWindowAsync(runKey); + return done.Task; + } + + private async Task FlushAfterWindowAsync(string runKey) + { + await Task.Delay(_window); + List<(WorkStore.Finish Finish, TaskCompletionSource Done)> batch; + lock (_lock) + { + batch = _pending[runKey]; + _pending.Remove(runKey); + } + + var finishes = batch.Select(b => b.Finish).ToList(); + try + { + var outcome = await store.FinishAsync(runKey, finishes); + foreach (var (_, done) in batch) done.TrySetResult(outcome); + } + catch (Exception ex) + { + logger.LogWarning(ex, "[Orchestrator] Could not record {Count} finished task(s) of {Run}; retrying in the background", + batch.Count, runKey); + foreach (var (_, done) in batch) done.TrySetResult(null); + _ = RetryAsync(runKey, finishes); + } + } + + private async Task RetryAsync(string runKey, List finishes) + { + Interlocked.Add(ref _retrying, finishes.Count); + try + { + var delay = FirstRetry; + var giveUpAt = DateTime.UtcNow + GiveUpAfter; + for (var attempt = 1; ; attempt++) + { + await Task.Delay(delay); + try + { + await store.FinishAsync(runKey, finishes); + logger.LogInformation("[Orchestrator] Recorded {Count} finished task(s) of {Run} after {Attempts} retr{Ies}", + finishes.Count, runKey, attempt, attempt == 1 ? "y" : "ies"); + return; + } + catch (Exception ex) + { + if (DateTime.UtcNow >= giveUpAt) + { + logger.LogError(ex, "[Orchestrator] Gave up recording {Count} finished task(s) of {Run} after {Attempts} attempts; they run again when their claims lapse", + finishes.Count, runKey, attempt + 1); + return; + } + delay = delay * 2 > TimeSpan.FromMinutes(1) ? TimeSpan.FromMinutes(1) : delay * 2; + } + } + } + finally + { + Interlocked.Add(ref _retrying, -finishes.Count); + } + } +} diff --git a/Services/Orchestration/JobDescriptor.cs b/Services/Orchestration/JobDescriptor.cs index e5d5d5d..6af62bf 100644 --- a/Services/Orchestration/JobDescriptor.cs +++ b/Services/Orchestration/JobDescriptor.cs @@ -4,14 +4,24 @@ namespace Craft.Orchestration; /// The identity of a queued orchestrator task — everything the queue needs to hold, and nothing more. /// /// Orchestrator fan-out is where queue depth comes from (production peaked at 783 queued with 3.7-hour -/// waits), and it used to enqueue a closure capturing the whole graph, the +/// waits), and it used to enqueue a closure capturing the whole run graph, the /// task, the script path and the service. A descriptor replaces all of that with two string references /// and an int; turns it back into runnable work at dispatch time. /// /// Run this task belongs to — the storage partition key. /// Task id within the run — the storage row key. /// Dispatch priority (lower = higher). -public readonly record struct JobDescriptor(string RunName, string TaskId, int Priority); +public readonly record struct JobDescriptor(string RunName, string TaskId, int Priority) +{ + /// The run's storage key (its partition); null for a descriptor that names a run only. + public string? RunKey { get; init; } + + /// The task's position in its run, which addresses its row. + public int Seq { get; init; } + + /// Which execution of the task this is (1 for the first). + public int Attempt { get; init; } +} /// /// Rehydrates a into runnable work. Returns null when the descriptor is diff --git a/Services/Orchestration/JobManager.cs b/Services/Orchestration/JobManager.cs index 9ab490e..55762e6 100644 --- a/Services/Orchestration/JobManager.cs +++ b/Services/Orchestration/JobManager.cs @@ -57,6 +57,12 @@ public class JobManager : BackgroundService // ── Tracking ── private readonly ConcurrentDictionary _jobs = new(); + /// + /// Jobs cancelled while queued, with whether the state writer was told. A cancel that finds the job still + /// in the queue tells it straight away; one that lands after the dispatcher has dequeued the job cannot see + /// its descriptor, so the dispatcher (or the job's start) tells it instead. Without that, the claim behind + /// the job was never finished, lapsed half an hour later, and ran after all. + /// private readonly ConcurrentDictionary _cancelledJobIds = new(); private readonly ConcurrentDictionary> _pendingWork = new(); @@ -79,6 +85,10 @@ public class JobManager : BackgroundService public int ActiveCount => _activeCount; public int QueuedCount { get { lock (_queueLock) return _pendingQueue.Count; } } + /// Raised each time a job leaves the queue for a worker, so a feeder can top the queue up at once + /// instead of waiting for its next poll. + public event Action? Dispatched; + /// /// Is this job still in flight — queued or running? /// @@ -189,6 +199,7 @@ public string Enqueue(JobDescriptor descriptor, string name, string? id = null) Id = jobId, Name = name, RunName = descriptor.RunName, + RunKey = descriptor.RunKey, Priority = descriptor.Priority, Status = "Queued", QueuedUtc = DateTime.UtcNow @@ -200,11 +211,9 @@ public string Enqueue(JobDescriptor descriptor, string name, string? id = null) // writes status onto the queue item's own record), so the TRACKED record sat frozen at the // previous outing's "Completed" while a live copy of the job was queued or running. // - // IsQueuedOrRunning reads this dictionary, and JobQueuePump.ReleaseFinishedAsync treats a "no" - // as permission to DELETE that task's durable queue row. A stale record therefore had the pump - // dropping rows out from under running work — observed live releasing 7-9 "finished" jobs per - // second against 8 slots. RedrivePendingTasks consults the same predicate, so it was misreading - // task state for the same reason. + // IsQueuedOrRunning reads this dictionary, and WorkPump treats a "no" as the job being done and stops + // renewing its claim. A stale record once had the pump dropping work out from under running jobs — + // observed live releasing 7-9 "finished" jobs per second against 8 slots. _jobs[jobId] = record; lock (_queueLock) @@ -257,6 +266,11 @@ protected override async Task ExecuteAsync(CancellationToken stoppingToken) { _pendingQueue.TryDequeue(out job, out _); } + if (job != null) + { + try { Dispatched?.Invoke(); } + catch (Exception ex) { _logger.LogDebug(ex, "[JobManager] A dispatch listener failed"); } + } if (job == null) { @@ -277,9 +291,11 @@ protected override async Task ExecuteAsync(CancellationToken stoppingToken) // The work ref must go too: CancelJob only marks the id, so leaving the entry here // stranded the captured closure (and everything it captured) in _pendingWork forever — // nothing else ever removes it, not even CleanupOldJobs. - if (_cancelledJobIds.TryRemove(job.Record.Id, out _)) + if (_cancelledJobIds.TryRemove(job.Record.Id, out var notified)) { _pendingWork.TryRemove(job.Record.Id, out _); + if (!notified && job.Descriptor is { } cancelled) + NotifyStateWriter(w => w.Cancelled(cancelled), "cancellation", job.Record.Name); _limiter.ReleaseSlot(); slotHeld = false; continue; @@ -357,15 +373,32 @@ private async Task RunJobAsync(QueuedJob job, CancellationToken ct) var parentInvocation = new OperationContext.Invocation(job.Record.Name) { RunName = job.Record.RunName, + RunKey = job.Descriptor?.RunKey, Priority = job.Descriptor != null ? job.Record.Priority : job.InheritPriority, Category = "Job" }; opScope = OperationContext.Set(parentInvocation); - job.Record.Status = "Running"; - job.Record.StartedUtc = DateTime.UtcNow; + // Cancelled or withdrawn after the dispatcher's own check but before the job got going. Under the + // record's lock so WithdrawJob either stops it here or sees it Running, never both. + bool skip, notified; + lock (job.Record) + { + skip = _cancelledJobIds.TryRemove(job.Record.Id, out notified); + if (!skip) + { + job.Record.Status = "Running"; + job.Record.StartedUtc = DateTime.UtcNow; + } + } + if (skip) + { + if (!notified && job.Descriptor is { } cancelled) + NotifyStateWriter(w => w.Cancelled(cancelled), "cancellation", job.Record.Name); + return; + } - var queueTime = job.Record.StartedUtc.Value - job.Record.QueuedUtc; + var queueTime = job.Record.StartedUtc!.Value - job.Record.QueuedUtc; if (queueTime.TotalSeconds > 1) { _logger.LogInformation( @@ -388,7 +421,9 @@ private async Task RunJobAsync(QueuedJob job, CancellationToken ct) await work(ct); - job.Record.Status = "Completed"; + var (status, error) = _lastStep.TryRemove(job.Record.Id, out var last) ? last : ("Completed", null); + job.Record.Status = status; + job.Record.LastError = error; job.Record.CompletedUtc = DateTime.UtcNow; } catch (OperationCanceledException) when (ct.IsCancellationRequested) @@ -406,6 +441,7 @@ private async Task RunJobAsync(QueuedJob job, CancellationToken ct) } finally { + _lastStep.TryRemove(job.Record.Id, out _); // Accounting first, context teardown second: ReleaseSlot cannot throw, Dispose can. Interlocked.Decrement(ref _activeCount); Interlocked.Increment(ref _totalProcessed); @@ -442,7 +478,7 @@ public List GetRunSummaries() .GroupBy(j => j.RunName!) .Select(g => { - var jobs = g.ToList(); + var jobs = LatestOuting(g); return new JobRunSummary { Name = g.Key, @@ -463,6 +499,18 @@ public List GetRunSummaries() .ToList(); } + /// A run's jobs, newest first, from its latest outing only: earlier runs of the same name are left out. + public List GetRunJobs(string runName, int limit) => + LatestOuting(_jobs.Values.Where(j => string.Equals(j.RunName, runName, StringComparison.OrdinalIgnoreCase))) + .OrderByDescending(j => j.QueuedUtc).Take(limit).ToList(); + + private static List LatestOuting(IEnumerable jobs) + { + var list = jobs.ToList(); + var latest = list.Where(j => j.RunKey != null).MaxBy(j => j.QueuedUtc)?.RunKey; + return latest == null ? list : list.Where(j => j.RunKey == latest).ToList(); + } + public List GetJobs(string? runName = null, string? status = null, int? limit = null) { var query = _jobs.Values.AsEnumerable(); @@ -532,15 +580,23 @@ public override void Dispose() public bool CancelJob(string jobId) { if (!_jobs.TryGetValue(jobId, out var record)) return false; - if (record.Status != "Queued") return false; + JobDescriptor? descriptor; + lock (record) + { + if (record.Status != "Queued") return false; - record.Status = "Cancelled"; - record.CompletedUtc = DateTime.UtcNow; - record.LastError = "Cancelled by user"; - _cancelledJobIds.TryAdd(jobId, true); + record.Status = "Cancelled"; + record.CompletedUtc = DateTime.UtcNow; + record.LastError = "Cancelled by user"; - JobDescriptor? descriptor; - lock (_queueLock) descriptor = FindLiveEntry(record)?.Descriptor; + // Under the queue lock, so the dispatcher either still finds the entry (and we tell the state writer + // here) or has already dequeued it (and tells it when it skips the job). + lock (_queueLock) + { + descriptor = FindLiveEntry(record)?.Descriptor; + _cancelledJobIds[jobId] = descriptor != null; + } + } if (descriptor is { } d) NotifyStateWriter(w => w.Cancelled(d), "cancellation", record.Name); @@ -548,6 +604,61 @@ public bool CancelJob(string jobId) return true; } + /// + /// A job that runs a sequential run's steps one after another has finished a step. The step is kept as its + /// own finished record and the job carries on under ; when there is no next step, + /// this step's outcome becomes the job's. Job lists, run summaries and queue pages then show every step, + /// not just the one the job was dispatched for. + /// + public void AdvanceStep(string jobId, string status, string? error, string? nextName) + { + if (!_jobs.TryGetValue(jobId, out var record)) return; + if (nextName == null) + { + _lastStep[jobId] = (status, error); + return; + } + var now = DateTime.UtcNow; + var stepId = $"{jobId}#{Interlocked.Increment(ref _stepRecords)}"; + _jobs[stepId] = new JobRecord + { + Id = stepId, + Name = record.Name, + RunName = record.RunName, + RunKey = record.RunKey, + Priority = record.Priority, + Status = status, + QueuedUtc = record.QueuedUtc, + StartedUtc = record.StartedUtc, + CompletedUtc = now, + LastError = error, + }; + record.Name = nextName; + record.QueuedUtc = now; + record.StartedUtc = now; + } + + private readonly ConcurrentDictionary _lastStep = new(); + private long _stepRecords; + + /// + /// Take a queued job back without running it and without recording an outcome, so its owner can hand the + /// work to another process. False once it has started. + /// + public bool WithdrawJob(string jobId) + { + if (!_jobs.TryGetValue(jobId, out var record)) return false; + lock (record) + { + if (record.Status != "Queued") return false; + record.Status = "Cancelled"; + record.CompletedUtc = DateTime.UtcNow; + record.LastError = "Withdrawn at shutdown"; + _cancelledJobIds[jobId] = true; + } + return true; + } + /// Cancel all queued jobs in a run group. public int CancelRun(string runName) { @@ -559,6 +670,8 @@ public int CancelRun(string runName) var descriptors = new List(toCancel.Count); lock (_queueLock) { + // Marked under the lock for the same reason as CancelJob: found here means told here. + foreach (var record in toCancel) _cancelledJobIds[record.Id] = false; var wanted = toCancel.ToDictionary(r => r.Id, r => r); foreach (var (entry, _) in _pendingQueue.UnorderedItems) { @@ -567,6 +680,7 @@ public int CancelRun(string runName) if (!ReferenceEquals(entry.Record, rec)) continue; if (_reprioritized.TryGetValue(rec.Id, out var live) && entry.Epoch != live) continue; descriptors.Add(d); + _cancelledJobIds[rec.Id] = true; } } @@ -575,7 +689,6 @@ public int CancelRun(string runName) record.Status = "Cancelled"; record.CompletedUtc = DateTime.UtcNow; record.LastError = "Run cancelled by user"; - _cancelledJobIds.TryAdd(record.Id, true); } foreach (var d in descriptors) diff --git a/Services/Orchestration/JobQueuePump.cs b/Services/Orchestration/JobQueuePump.cs deleted file mode 100644 index a5e030c..0000000 --- a/Services/Orchestration/JobQueuePump.cs +++ /dev/null @@ -1,268 +0,0 @@ -using Craft.Configuration; -using Craft.Storage; - -namespace Craft.Orchestration; - -/// -/// Keeps the in-memory job queue small by feeding it from storage a batch at a time. -/// -/// The point is that the backlog lives in the table, not in this process: the JobManager holds at most -/// a worker-pool-sized buffer plus whatever is in flight, and tops up only when it runs low. A run of -/// 7,336 tasks is 7,336 rows in storage and a handful of objects here. -/// -/// Deliberately a separate pump rather than a change to the dispatch loop. That loop owns the limiter -/// slot lifecycle, and its invariants were written to close a leak that wedged a production instance for -/// 28 hours (see BackgroundTaskLimiter's accounting notes). Feeding it through the enqueue path it -/// already has means none of that is disturbed — this component can only ever add work. -/// -/// Lease handling is what makes the claim safe to hold in memory: a claimed row stays owned by this -/// instance for LeaseSeconds, renewed while the job is still in flight, and reclaimable by anyone once -/// it lapses. An instance that dies mid-batch gives its work back without anything having to notice. -/// -public class JobQueuePump : BackgroundService -{ - private readonly ILogger _logger; - private readonly JobQueueStore _queue; - private readonly JobManager _jobs; - private readonly string _owner; - private readonly int _batchSize; - private readonly int _lowWater; - private readonly TimeSpan _lease; - private readonly TimeSpan _pollInterval; - private readonly TimeSpan _idlePollInterval; - - /// Renew a claim only once its lease has less than this left. The lease is set comfortably - /// longer than a task can run, so most claims finish without ever needing a renewal; this is the tail - /// (a deep buffer, or a genuinely long task) that has been held long enough to approach expiry. - private readonly TimeSpan _renewWhenWithin; - - /// Rows claimed by this instance, by the job id they were handed to the JobManager under. - private readonly Dictionary _inFlight = new(StringComparer.Ordinal); - - /// UTC lease expiry per in-flight job id, so renewal can skip claims with plenty of lease - /// left instead of re-reading every row every tick. - private readonly Dictionary _leaseExpiry = new(StringComparer.Ordinal); - - /// Completes when startup recovery is done; nothing is claimed before it. Null (tests, a host - /// without an orchestrator) claims immediately. - private readonly Task? _claimGate; - - public JobQueuePump(ILogger logger, JobQueueStore queue, JobManager jobs, - IConfiguration configuration, CraftSettings settings, OrchestratorService? orchestrator = null) - { - _logger = logger; - _queue = queue; - _jobs = jobs; - _claimGate = orchestrator?.RecoveryDone; - - // Identifies this instance's claims. The container id is stable for the life of the process and - // distinct per instance, which is exactly the scope a lease needs. - _owner = Environment.GetEnvironmentVariable("HOSTNAME") - ?? $"instance-{Environment.ProcessId}"; - - // One batch per worker, so a refill hands the pool exactly enough to stay busy. - _batchSize = Math.Max(1, configuration.GetValue("JobQueueBatchSize", Math.Max(1, settings.Worker.BgPoolSize))); - - // Top up before the buffer empties, so workers never wait on a storage round-trip. - _lowWater = Math.Max(0, configuration.GetValue("JobQueueLowWaterMark", 2)); - - // Comfortably longer than the 1200s task timeout: a lease that lapses under a task still running - // would hand its work to a second worker. - _lease = TimeSpan.FromSeconds(Math.Max(60, configuration.GetValue("JobQueueLeaseSeconds", 1800))); - - // Renew in the last third of the lease. Below the lease length by construction, so a claim is - // never renewed on the same tick it was taken. - _renewWhenWithin = TimeSpan.FromTicks(_lease.Ticks / 3); - - _pollInterval = TimeSpan.FromMilliseconds( - Math.Max(100, configuration.GetValue("JobQueuePollIntervalMs", 1000))); - - // Ceiling for the idle backoff. Clamped to at least the base interval, or "backing off" would - // speed the loop up. - _idlePollInterval = TimeSpan.FromMilliseconds(Math.Max( - _pollInterval.TotalMilliseconds, - configuration.GetValue("JobQueueIdlePollIntervalMs", 10_000))); - } - - protected override async Task ExecuteAsync(CancellationToken stoppingToken) - { - _logger.LogInformation( - "[JobQueuePump] Started: owner={Owner} batch={Batch} lowWater={Low} lease={Lease}s poll={Poll}ms idlePoll={IdlePoll}ms", - _owner, _batchSize, _lowWater, _lease.TotalSeconds, - _pollInterval.TotalMilliseconds, _idlePollInterval.TotalMilliseconds); - - // No claim before startup recovery has run. A row claimed earlier rehydrates its run from storage — - // stale Running markers from the previous process included — into the live graph ahead of recovery, - // whose reset then lands on a copy that loses the _activeRuns race. Rows enqueued meanwhile (timer or - // HTTP-started runs) just wait in the table and are claimed on the first cycle after. - if (_claimGate is { IsCompleted: false }) - { - _logger.LogInformation("[JobQueuePump] Waiting for startup recovery before claiming"); - try { await _claimGate.WaitAsync(stoppingToken); } - catch (OperationCanceledException) { return; } - _logger.LogInformation("[JobQueuePump] Startup recovery done — claiming"); - } - - var idleTicks = 0; - - while (!stoppingToken.IsCancellationRequested) - { - var claimedAny = false; - try - { - // Ensure the queue schema is migrated before this pump ever claims. It is a cheap bool - // check after the first success; before it, claiming could hand out a row the one-time - // key migration is still rewriting. Idempotent and shared with the enqueue paths. - await _queue.InitializeAsync(stoppingToken); - - await ReleaseFinishedAsync(stoppingToken); - claimedAny = await RefillAsync(stoppingToken); - await RenewAsync(stoppingToken); - } - catch (OperationCanceledException) when (stoppingToken.IsCancellationRequested) - { - break; - } - catch (Exception ex) - { - // One bad cycle must not end the pump; the next tick tries again. A pump that dies - // silently would look exactly like an empty queue. - // - // The log itself is guarded: under a pegged GC hard limit even LogError allocates and can - // throw OOM, and this catch is the last frame before ExecuteAsync — an escape here faults - // the service and, with the host's default StopHost behaviour, restarts the container - // mid-run. A failed log is never worth the pump. (BackgroundTaskLimiter guards its logs - // past the point of commitment for the same reason.) - try { _logger.LogError(ex, "[JobQueuePump] Cycle failed; continuing"); } - catch { /* logging is never worth the loop */ } - } - - // Idle means this pump has nothing: it claimed nothing AND holds nothing. Note what is - // deliberately NOT idle — a full buffer. RefillAsync returns early without claiming while - // QueuedCount is above the low-water mark, and treating that as idle would back the loop off - // exactly when a busy run is about to need its next batch. A refill delivers at most - // batchSize jobs per tick, so the poll interval is a hard throughput ceiling of - // batchSize/interval: at a flat 10s that is 0.8 tasks/sec, which on a 7,336-task fan-out is - // hours of pure waiting however fast the tasks are. Backing off only when genuinely empty - // keeps that ceiling at the base interval whenever it could bind. - if (claimedAny || _inFlight.Count > 0) idleTicks = 0; - else idleTicks++; - - // Wait for the poll interval OR an enqueue signal, whichever comes first. The signal is what - // makes a freshly-queued run start now instead of on the next tick — a cold system had backed - // the interval off toward its idle ceiling, so without this the first task of a quiet-time - // orchestration waited up to that ceiling just to be claimed. The interval stays as the - // backstop (a missed signal, cross-instance work, freed leases), so this only removes the wait. - try { await _queue.WaitForWorkAsync(NextDelay(idleTicks), stoppingToken); } - catch (OperationCanceledException) { break; } - } - - _logger.LogInformation("[JobQueuePump] Stopped ({InFlight} claims still held)", _inFlight.Count); - } - - /// - /// Drop the queue rows for jobs the JobManager has finished with. - /// - /// Removal is deliberately AFTER the work is done, not at claim time: a row deleted on claim would - /// take the task with it if this instance died holding it. Until then the lease is what stops anyone - /// else running it. - /// - private async Task ReleaseFinishedAsync(CancellationToken ct) - { - if (_inFlight.Count == 0) return; - - var finished = _inFlight.Where(kv => !_jobs.IsQueuedOrRunning(kv.Key)).ToList(); - if (finished.Count == 0) return; - - // One transaction per partition rather than two point deletes per task. Tracking is cleared only - // after the storage delete lands, so a failure here throws, the cycle guard logs it, and the same - // claims are retried next tick (the delete is idempotent — a row already gone is tolerated). - await _queue.RemoveBatchAsync(finished.Select(kv => kv.Value).ToList(), ct); - - foreach (var (jobId, _) in finished) - { - _inFlight.Remove(jobId); - _leaseExpiry.Remove(jobId); - } - - _logger.LogDebug("[JobQueuePump] Released {Count} finished job(s)", finished.Count); - } - - /// - /// How long to wait before the next cycle: the base interval while there is anything to do, doubling - /// toward once the pump has gone quiet. - /// - /// The doubling matters more than the ceiling — it means a queue that goes briefly empty between - /// batches barely slows down, while one that is empty for minutes stops scanning storage every - /// second. Any tick that claims or holds work resets it, so the pump is back at full speed on the - /// cycle after work appears. - /// - private TimeSpan NextDelay(int idleTicks) - { - if (idleTicks <= 0) return _pollInterval; - - // Clamped before the shift so the multiplier cannot overflow on a long idle stretch. - var factor = 1L << Math.Min(idleTicks, 20); - var ms = Math.Min(_idlePollInterval.TotalMilliseconds, _pollInterval.TotalMilliseconds * factor); - return TimeSpan.FromMilliseconds(ms); - } - - /// - /// Claim another batch once the buffer has drawn down to the low-water mark. Returns whether - /// anything was claimed, which is what tells the loop this tick was not idle. - /// - private async Task RefillAsync(CancellationToken ct) - { - if (_jobs.QueuedCount > _lowWater) return false; - - var claimed = await _queue.ClaimBatchAsync(_owner, _batchSize, _lease, ct); - if (claimed.Count == 0) return false; - - var leaseExpiry = DateTime.UtcNow + _lease; - foreach (var job in claimed) - { - // Enqueued by identity, the way the orchestrator already does it, so the work is rebuilt at - // dispatch time and nothing but the descriptor is retained here. - var name = $"{job.RunName}-{job.TaskId}"; - var jobId = _jobs.Enqueue(new JobDescriptor(job.RunName, job.TaskId, job.Priority), name); - _inFlight[jobId] = job; - // Slightly earlier than the lease storage actually recorded (claimed a moment before this), - // so renewal errs toward being early rather than late. - _leaseExpiry[jobId] = leaseExpiry; - } - - _logger.LogDebug("[JobQueuePump] Claimed {Count} job(s) ({Queued} queued after refill)", - claimed.Count, _jobs.QueuedCount); - - return true; - } - - /// - /// Extend the lease on everything still in flight. A failure here means a claim lapsed and the work - /// may already have been taken by someone else, so it is logged loudly — but not acted on, because - /// the task itself is the JobManager's to finish or fail. - /// - private async Task RenewAsync(CancellationToken ct) - { - if (_inFlight.Count == 0) return; - - // Only the claims whose lease is actually running low. Everything else has ample lease left and - // does not need a storage round-trip this tick — the old code re-read every in-flight row every - // tick regardless of how much lease remained. - var now = DateTime.UtcNow; - var dueIds = _inFlight.Keys - .Where(id => !_leaseExpiry.TryGetValue(id, out var expiry) || expiry - now < _renewWhenWithin) - .ToList(); - if (dueIds.Count == 0) return; - - var due = dueIds.Select(id => _inFlight[id]).ToList(); - if (!await _queue.RenewAsync(due, _owner, _lease, ct)) - { - _logger.LogWarning("[JobQueuePump] One or more leases could not be renewed — work may have been reclaimed"); - return; - } - - var renewedExpiry = now + _lease; - foreach (var id in dueIds) _leaseExpiry[id] = renewedExpiry; - } -} diff --git a/Services/Orchestration/JobQueueStatusReader.cs b/Services/Orchestration/JobQueueStatusReader.cs index aabd61d..6ec3c1b 100644 --- a/Services/Orchestration/JobQueueStatusReader.cs +++ b/Services/Orchestration/JobQueueStatusReader.cs @@ -4,148 +4,86 @@ namespace Craft.Orchestration; /// -/// The table-backed view of the job queue, for the status APIs. -/// -/// Since ownership of queued tasks moved into the {prefix}Queue table, the in-memory JobManager holds -/// only a worker-pool-sized buffer of claims plus whatever closure jobs were enqueued directly. Every -/// consumer that used to read it as "the queue" — the worker-health page, /API/jobs/*, the stats -/// history — was therefore reporting the buffer as if it were the backlog: a 7,000-task fan-out showed -/// eight queued jobs. This type merges the two truths: the JobManager for what THIS instance is doing, -/// the tables for what exists. -/// -/// Reads are cached with a short TTL and refreshed single-flight, because the queue scan is -/// proportional to the backlog and the snapshot consumers poll — the stats sampler on its timer, the -/// dashboard at a few hertz, the perf harness at 4 Hz. One scan per TTL window serves all of them. -/// A refresh failure keeps the previous snapshot: stale numbers with an honest timestamp beat an -/// exception on a health endpoint. -/// -/// THE TTL ADAPTS TO WHAT THE SCAN ACTUALLY COSTS, and that is not a refinement. The snapshot used to -/// be stamped with the time the refresh STARTED, so on a large backlog an 80-second scan produced a -/// snapshot already 80 seconds past a 5-second TTL the moment it was stored. The next poll — the -/// worker-health page sits at a few hertz — saw it stale and started another. The single-flight gate -/// kept it to one at a time, but they ran back to back, permanently, against the largest table in the -/// account, on the instance least able to afford it. Measured on the instance that motivated this: -/// full scans of a 743,000-row queue looping continuously, alongside 12,028 per-run counter reads -/// each (see BuildSnapshotAsync). A customer reporting sluggishness had "kept the Worker Health page -/// open" — which is what was driving it. -/// -/// So: stamp on completion, and never re-scan more often than the last scan took. +/// The status APIs' view of queued work: the JobManager for what this process holds, the Ready list for what +/// exists durably. One Ready scan (one small row per active run) feeds every count, and only the head of the +/// queue is read row by row, so a snapshot costs the same whatever the backlog. Snapshots are cached and +/// re-scanned no more often than four times the last scan took. /// public class JobQueueStatusReader : IDisposable { private readonly ILogger _logger; private readonly JobManager _jobs; - private readonly JobQueueStore _queue; - private readonly OrchestratorTableStore _store; + private readonly WorkStore _store; private static readonly TimeSpan DefaultTtl = TimeSpan.FromSeconds(5); - /// - /// How many runs will resolve counter rows for. Each is a point - /// read, cheap alone and not in the thousands. A run without one falls back to local numbers, which - /// is the same graceful path a pre-counter run already takes — so this degrades detail, not - /// correctness. Truncation is logged rather than silent. - /// - private const int MaxCounterLookups = 200; + /// Rows listed from the head of the queue; runs past it are counted, not listed. + internal const int HeadRows = 2_000; + private const int HeadRuns = 50; private volatile QueueSnapshot? _cached; private readonly SemaphoreSlim _refreshGate = new(1, 1); - - /// - /// How long the last successful snapshot took to build. The floor under the effective TTL, so a - /// scan that costs 80s is not re-run 5s later. Volatile read/write of a long via Interlocked — - /// ticks rather than TimeSpan so it stays atomic. - /// private long _lastBuildTicks; - public JobQueueStatusReader(ILogger logger, JobManager jobs, - JobQueueStore queue, OrchestratorTableStore store) + public JobQueueStatusReader(ILogger logger, JobManager jobs, WorkStore store) { _logger = logger; _jobs = jobs; - _queue = queue; _store = store; } - /// The underlying durable queue, for callers that need its maintenance operations. - public JobQueueStore Queue => _queue; + /// A task waiting in storage, as the job listings show it. + /// A task waiting in storage, as the job listings show it; and + /// address its row, so acting on it is a point read. + public sealed record QueuedRow(string RunName, string TaskId, int Priority, DateTime QueuedUtc, bool Claimed, + string RunKey = "", int Seq = 0); - /// - /// Per-run slice of the durable queue, derived entirely from the rows the scan already returned. - /// - /// Deliberately carries no counter data. The counter lives in a different table, partitioned per - /// run, so folding it in here meant one point read per distinct run on EVERY refresh — 12,028 of - /// them on the instance that motivated this — even though only - /// ever read those fields. The two paths that poll hardest, and - /// , never used them at all. - /// - public sealed record RunQueueInfo(int Unclaimed, int Claimed, int MinPriority, - DateTime? OldestQueuedUtc); + /// A run name's durable counts (summed over runs sharing the name). Merged views subtract what this + /// process holds right now, never the counts held when the snapshot was taken. + public sealed record RunQueueInfo(int Unclaimed, int Claimed, int MinPriority, DateTime? OldestQueuedUtc, + int Total, int Done, string? Reference, int Failed = 0) + { + public int Outstanding => Math.Max(0, Total - Done); + } - /// One scan of the queue table, aggregated the way the status APIs consume it. - public sealed record QueueSnapshot(DateTime TakenUtc, IReadOnlyList Rows, - int Unclaimed, int Claimed, DateTime? OldestUnclaimedUtc, - IReadOnlyDictionary ByRun) + /// One Ready scan. is the head of the queue only; the counts cover every run. + public sealed record QueueSnapshot(DateTime TakenUtc, IReadOnlyList Rows, int Total, int Unclaimed, + int Claimed, DateTime? OldestUnclaimedUtc, IReadOnlyDictionary ByRun) { - public int Total => Rows.Count; public double AgeSeconds => (DateTime.UtcNow - TakenUtc).TotalSeconds; } - /// - /// The most recent snapshot without ever blocking on storage — for the paths that must stay cheap - /// and non-blocking (GetSnapshot on a PS worker, the stats sampler, the 4 Hz allocation poll). - /// A stale or missing snapshot kicks off a background refresh and returns what exists NOW; the - /// refreshed data is simply what the next call sees. - /// - /// - /// The shortest interval we will re-scan at: never less than the requested age, and never less than - /// the last scan took. On a healthy instance the scan is milliseconds and this is just the 5s TTL; - /// on a degraded one it is what stops the refresh loop from consuming the storage account. - /// private TimeSpan EffectiveTtl(TimeSpan? maxAge) { var requested = maxAge ?? DefaultTtl; - var lastBuild = TimeSpan.FromTicks(Interlocked.Read(ref _lastBuildTicks)); - return lastBuild > requested ? lastBuild : requested; + var floor = TimeSpan.FromTicks(Interlocked.Read(ref _lastBuildTicks) * 4); + return floor > requested ? floor : requested; } + /// The latest snapshot without blocking; a stale one starts a refresh in the background. public QueueSnapshot? GetCached(TimeSpan? maxAge = null) { var cached = _cached; if (cached == null || DateTime.UtcNow - cached.TakenUtc > EffectiveTtl(maxAge)) - { _ = Task.Run(() => GetAsync(maxAge, CancellationToken.None)); - } return cached; } - /// - /// A snapshot no older than , refreshing if needed. Returns the previous - /// snapshot when the refresh fails, and null only when storage has never answered at all. - /// public async Task GetAsync(TimeSpan? maxAge = null, CancellationToken ct = default) { - var ttl = EffectiveTtl(maxAge); var cached = _cached; - if (cached != null && DateTime.UtcNow - cached.TakenUtc <= ttl) return cached; + if (cached != null && DateTime.UtcNow - cached.TakenUtc <= EffectiveTtl(maxAge)) return cached; await _refreshGate.WaitAsync(ct); try { - // Re-check under the gate, against a TTL recomputed AFTER the wait: whoever held the gate - // may also have just told us how expensive this scan is now. cached = _cached; if (cached != null && DateTime.UtcNow - cached.TakenUtc <= EffectiveTtl(maxAge)) return cached; - - var startedUtc = DateTime.UtcNow; - var snapshot = await BuildSnapshotAsync(ct); - Interlocked.Exchange(ref _lastBuildTicks, (DateTime.UtcNow - startedUtc).Ticks); - _cached = snapshot; - } - catch (OperationCanceledException) when (ct.IsCancellationRequested) - { - throw; + var started = DateTime.UtcNow; + _cached = await BuildSnapshotAsync(ct); + Interlocked.Exchange(ref _lastBuildTicks, (DateTime.UtcNow - started).Ticks); } + catch (OperationCanceledException) when (ct.IsCancellationRequested) { throw; } catch (Exception ex) { _logger.LogWarning(ex, "[JobQueueStatus] Queue snapshot refresh failed — serving previous data"); @@ -154,114 +92,88 @@ private TimeSpan EffectiveTtl(TimeSpan? maxAge) { _refreshGate.Release(); } - return _cached; } - /// - /// One scan of the queue table, aggregated. No per-run storage reads: everything here comes from - /// the rows the scan already returned, so the cost is one scan regardless of how many runs the - /// backlog spans. - /// private async Task BuildSnapshotAsync(CancellationToken ct) { - var rows = await _queue.ListQueuedAsync(ct); + // Local jobs are counted per run name; runs sharing a name take them oldest first. + var local = _jobs.GetJobs().Where(j => j.RunName != null && j.Status is "Queued" or "Running") + .GroupBy(j => j.RunName!).ToDictionary(g => g.Key, g => g.Count(), StringComparer.Ordinal); - var unclaimed = 0; - DateTime? oldestUnclaimed = null; var byRun = new Dictionary(StringComparer.Ordinal); + var head = new List(); + int total = 0, unclaimed = 0, runsListed = 0; + DateTime? oldest = null; - foreach (var group in rows.GroupBy(r => r.RunName)) + await foreach (var e in _store.ReadReadyAsync(1000, ct)) { - var runUnclaimed = 0; - var runClaimed = 0; - var minPriority = int.MaxValue; - DateTime? oldest = null; - - foreach (var row in group) + var outstanding = Math.Max(0, e.Total - e.Done); + var claimed = Math.Min(outstanding, local.GetValueOrDefault(e.Name)); + local[e.Name] = local.GetValueOrDefault(e.Name) - claimed; + total += outstanding; + unclaimed += outstanding - claimed; + if (outstanding > claimed && (oldest == null || e.StartedUtc < oldest)) oldest = e.StartedUtc; + byRun[e.Name] = byRun.TryGetValue(e.Name, out var same) + ? new RunQueueInfo(same.Unclaimed + outstanding - claimed, same.Claimed + claimed, Math.Min(same.MinPriority, e.Band), + same.OldestQueuedUtc, same.Total + e.Total, same.Done + e.Done, same.Reference ?? e.Reference, same.Failed + e.Failed) + : new RunQueueInfo(outstanding - claimed, claimed, e.Band, e.StartedUtc, e.Total, e.Done, e.Reference, e.Failed); + + if (head.Count < HeadRows && runsListed < HeadRuns && outstanding > claimed) { - if (row.Claimed) runClaimed++; - else - { - runUnclaimed++; - if (oldest == null || row.QueuedUtc < oldest) oldest = row.QueuedUtc; - } - if (row.Priority < minPriority) minPriority = row.Priority; + runsListed++; + foreach (var t in await _store.GetTasksAsync(e.RunKey, 'P', HeadRows - head.Count, ct)) + head.Add(new QueuedRow(e.Name, t.TaskId, e.Band, e.StartedUtc, false, e.RunKey, t.Seq)); } - - unclaimed += runUnclaimed; - if (oldest != null && (oldestUnclaimed == null || oldest < oldestUnclaimed)) - oldestUnclaimed = oldest; - - byRun[group.Key] = new RunQueueInfo(runUnclaimed, runClaimed, - minPriority == int.MaxValue ? 0 : minPriority, oldest); } - // Stamped on COMPLETION, not on entry. A scan that took longer than the TTL would otherwise - // return a snapshot that is already expired, and the next poll would start another immediately. - return new QueueSnapshot(DateTime.UtcNow, rows, unclaimed, rows.Count - unclaimed, - oldestUnclaimed, byRun); + return new QueueSnapshot(DateTime.UtcNow, head, total, unclaimed, total - unclaimed, oldest, byRun); } // ─── Merged views ─── - /// - /// The JobManager summary with the durable backlog folded in: Queued counts local jobs PLUS - /// unclaimed queue rows (claimed rows are already represented by local records on the instance - /// that holds them), and OldestQueuedUtc considers both sources. - /// + /// The JobManager summary with the durable backlog folded in. public async Task GetSummaryAsync(CancellationToken ct = default) { var summary = _jobs.GetSummary(); summary.QueuedLocal = summary.Queued; - var snap = await GetAsync(ct: ct); if (snap == null) return summary; - summary.QueuedDurable = snap.Unclaimed; - summary.Queued += snap.Unclaimed; - - if (snap.OldestUnclaimedUtc is { } oldest - && (summary.OldestQueuedUtc == null || oldest < summary.OldestQueuedUtc)) - { + summary.QueuedDurable = WaitingInStorage(snap); + summary.Queued += summary.QueuedDurable; + if (snap.OldestUnclaimedUtc is { } oldest && (summary.OldestQueuedUtc == null || oldest < summary.OldestQueuedUtc)) summary.OldestQueuedUtc = oldest; - } - return summary; } /// - /// The job listing the worker-health page shows: local records merged with the unclaimed durable - /// backlog. Durable rows only participate when the status filter admits "Queued" — every other - /// status describes work an instance has already claimed, which local records cover. + /// Tasks waiting in storage, unclaimed: everything the snapshot found outstanding, less what this process holds + /// now (its queued and running orchestrator jobs). The snapshot's own + /// subtracted what was held when it was taken, so adding it to a live local count double-counts every claim + /// made since (seen live: 202 queued on a 200-task run). Every merged count goes through here. /// + public int WaitingInStorage(QueueSnapshot snap) => + Math.Max(0, snap.Total - _jobs.GetJobs().Count(j => j.RunName != null && j.Status is "Queued" or "Running")); + + /// Local job records plus the head of the durable queue, for the job listing. public async Task> GetJobDetailsAsync(string? runName = null, string? status = null, int limit = 100, CancellationToken ct = default) { var local = _jobs.GetJobDetails(runName, status, limit); - - var includeDurable = string.IsNullOrEmpty(status) - || status.Equals("Queued", StringComparison.OrdinalIgnoreCase); - if (!includeDurable) return local; + if (!string.IsNullOrEmpty(status) && !status.Equals("Queued", StringComparison.OrdinalIgnoreCase)) return local; var snap = await GetAsync(ct: ct); if (snap == null || snap.Rows.Count == 0) return local; var now = DateTime.UtcNow; var merged = new List(local); - + var held = _jobs.GetJobs().Where(j => j.Status is "Queued" or "Running").Select(j => j.Name).ToHashSet(StringComparer.Ordinal); foreach (var row in snap.Rows) { - // Claimed rows are queued or running inside some instance's JobManager; on this instance - // they are the local records already in the list. The id guard closes the enqueue/claim - // race window on top of that. - if (row.Claimed) continue; - if (!string.IsNullOrEmpty(runName) - && !string.Equals(row.RunName, runName, StringComparison.OrdinalIgnoreCase)) continue; - + if (!string.IsNullOrEmpty(runName) && !string.Equals(row.RunName, runName, StringComparison.OrdinalIgnoreCase)) continue; var id = $"{row.RunName}-{row.TaskId}"; - if (_jobs.IsQueuedOrRunning(id)) continue; - + if (held.Contains(id)) continue; merged.Add(new JobDetail { Id = id, @@ -273,21 +185,10 @@ public async Task> GetJobDetailsAsync(string? runName = null, st WaitSeconds = Math.Max(0, (now - row.QueuedUtc).TotalSeconds), }); } - - // Same ordering contract as JobManager.GetJobDetails, re-applied across the merged set. - return merged - .OrderBy(j => j.Priority) - .ThenBy(j => j.QueuedUtc) - .Take(limit) - .ToList(); + return merged.OrderBy(j => j.Priority).ThenBy(j => j.QueuedUtc).Take(limit).ToList(); } - /// - /// Run summaries with durable truth folded in: Total from the run's counter row, Queued including - /// the unclaimed backlog, Completed at least the durably-terminal count. Runs that exist only in - /// the table — a backlog nothing here has claimed yet — get a synthesized entry, because a run the - /// page cannot see is a run nobody can cancel. - /// + /// Run summaries: local job records with each active run's durable size and progress folded in. public async Task> GetRunSummariesAsync(CancellationToken ct = default) { var summaries = _jobs.GetRunSummaries(); @@ -295,68 +196,26 @@ public async Task> GetRunSummariesAsync(CancellationToken ct if (snap == null) return summaries; var byName = summaries.ToDictionary(s => s.Name, StringComparer.Ordinal); - foreach (var (run, info) in snap.ByRun) { - if (byName.TryGetValue(run, out var summary)) - { - summary.Queued += info.Unclaimed; - } - else + if (!byName.TryGetValue(run, out var summary)) { - // Claimed rows here belong to another instance's buffer — not distinguishable from - // queued at this distance, and not terminal, so Queued is the honest bucket. - var synthesized = new JobRunSummary - { - Name = run, - Priority = info.MinPriority, - Queued = info.Unclaimed + info.Claimed, - Total = info.Unclaimed + info.Claimed, - }; - summaries.Add(synthesized); - byName[run] = synthesized; + summary = new JobRunSummary { Name = run, Priority = info.MinPriority, StartedUtc = info.OldestQueuedUtc }; + summaries.Add(summary); + byName[run] = summary; } + // Storage is the truth for a run still going: outstanding work is running here or waiting, and done + // splits into completed and failed. Local job history can include earlier outings of the name. + summary.Reference ??= info.Reference; + summary.Total = info.Total; + summary.Running = Math.Min(summary.Running, info.Outstanding); + summary.Queued = info.Outstanding - summary.Running; + summary.Failed = info.Failed; + summary.Completed = Math.Max(0, info.Done - info.Failed); + summary.CompletedUtc = null; } - // Counter rows, resolved HERE rather than in the snapshot, because this is the only caller that - // reads them. Each is a point read on the run's own task partition. - // - // The counter is a run's true size and durable progress — a restart or a multi-instance claim - // pattern otherwise shrinks Total to whatever this node happened to process. It covers runs with - // queue rows and runs whose backlog is fully claimed alike, which is why this is one pass over - // the active summaries rather than two loops keyed off the snapshot. - var ordered = summaries - .OrderBy(r => r.Priority) - .ThenByDescending(r => r.StartedUtc) - .ToList(); - - var active = ordered.Where(s => s.Queued + s.Running > 0).ToList(); - var lookups = Math.Min(active.Count, MaxCounterLookups); - - for (var i = 0; i < lookups; i++) - { - try - { - if (await _store.GetCounterAsync(active[i].Name, ct) is { } counter) - Overlay(active[i], counter.Remaining, counter.Total); - } - catch (Exception ex) - { - _logger.LogDebug(ex, "[JobQueueStatus] Counter read failed for {Run}", active[i].Name); - } - } - - if (active.Count > lookups) - { - // Said out loud: a silent cap here would read as "these runs have no durable progress" - // rather than "we did not look", and the two are indistinguishable in the UI. - _logger.LogDebug( - "[JobQueueStatus] Resolved counters for {Looked} of {Active} active runs (cap {Cap}) — " + - "the remainder report local numbers only", - lookups, active.Count, MaxCounterLookups); - } - - return ordered; + return summaries.OrderBy(r => r.Priority).ThenByDescending(r => r.StartedUtc).ToList(); } public void Dispose() @@ -364,19 +223,4 @@ public void Dispose() GC.SuppressFinalize(this); _refreshGate.Dispose(); } - - /// - /// Fold a run's counter into its summary. Total is authoritative when present. The durably-terminal - /// count (Total − Remaining) spans Completed, Failed and Cancelled across every instance and every - /// restart; local Failed is kept (it is a lower bound) and the rest raises Completed. - /// - private static void Overlay(JobRunSummary summary, int? remaining, int? total) - { - if (total is not { } t || remaining is not { } r) return; - - if (t > summary.Total) summary.Total = t; - - var durablyDone = Math.Max(0, t - r); - summary.Completed = Math.Max(summary.Completed, durablyDone - summary.Failed); - } } diff --git a/Services/Orchestration/MarkerNotPersistedException.cs b/Services/Orchestration/MarkerNotPersistedException.cs deleted file mode 100644 index f619255..0000000 --- a/Services/Orchestration/MarkerNotPersistedException.cs +++ /dev/null @@ -1,16 +0,0 @@ -namespace Craft.Orchestration; - -/// -/// The durable "Running" marker for a task did not reach storage — it timed out, or the flush carrying -/// it failed. -/// -/// This is a DEFERRAL, not a task failure, and the distinction is the whole point of the type. Nothing -/// was written, so the task is still Pending in storage; running it anyway would defeat the -/// AttemptCount/MaxRetries bound the marker exists to provide, and failing it would lose work that never -/// ran. The correct response is to release the slot and retry. -/// -public sealed class MarkerNotPersistedException : Exception -{ - public MarkerNotPersistedException(string message) : base(message) { } - public MarkerNotPersistedException(string message, Exception inner) : base(message, inner) { } -} diff --git a/Services/Orchestration/OrchestratorRun.cs b/Services/Orchestration/OrchestratorRun.cs deleted file mode 100644 index 9808b28..0000000 --- a/Services/Orchestration/OrchestratorRun.cs +++ /dev/null @@ -1,40 +0,0 @@ -namespace Craft.Orchestration; - -public class OrchestratorRun -{ - public string Name { get; set; } = string.Empty; - public string? Reference { get; set; } - public string Status { get; set; } = "Pending"; - // 4 matches what every live enqueue path actually passes when a caller sets nothing — a run row - // rehydrated without a stored priority must not come back HIGHER than it originally ran. - public int Priority { get; set; } = 4; - public DateTime StartedUtc { get; set; } - public DateTime? CompletedUtc { get; set; } - public List Tasks { get; set; } = []; - public string? TaskScriptName { get; set; } - public string? PostExecFunctionName { get; set; } - public string? PostExecParametersJson { get; set; } - // null | "Pending" | "Running" | "Completed" | "Failed" | "Abandoned" - // "Failed" is retryable — recovery picks it back up on the next host start. "Abandoned" is the - // terminal one: retries are spent, and the run's result rows have been cleaned up. - public string? PostExecStatus { get; set; } - - /// - /// How many times post-execution has been attempted. Bounds the retry that - /// ResumeInterruptedRunsAsync performs for a "Failed" post-execution, the same way - /// bounds task recovery — without it a - /// permanently-failing aggregation would be retried on every host start forever. - /// - public int PostExecAttemptCount { get; set; } - - public string? ParentRunName { get; set; } - - /// - /// Sequential execution mode. When true the run's tasks are dispatched ONE AT A TIME, in ascending - /// (batch) order: only the current task is ever enqueued, - /// and the next is enqueued when it reaches a terminal state. Runs on any free worker (no pinning) — - /// the durable queue simply never holds more than one of this run's tasks at once. The default (false) - /// is the fan-out behaviour: every task is enqueued up front and drained in parallel by the pool. - /// - public bool Sequential { get; set; } -} diff --git a/Services/Orchestration/OrchestratorService.cs b/Services/Orchestration/OrchestratorService.cs index bc7843f..838dec4 100644 --- a/Services/Orchestration/OrchestratorService.cs +++ b/Services/Orchestration/OrchestratorService.cs @@ -1,29 +1,21 @@ using System.Collections.Concurrent; -using System.Diagnostics.CodeAnalysis; using System.Text.Json; using System.Text.Json.Serialization; using Craft.Configuration; +using Craft.Hosting; using Craft.PowerShellHost; using Craft.Services; using Craft.Storage; namespace Craft.Orchestration; -// OrchestrationResults removed — results are now persisted to the -// CippOrchestratorResults Azure Table via OrchestratorTableStore. - /// -/// Lightweight replacement for Azure Durable Functions orchestration. -/// Manages fan-out/fan-in runs with crash-resilient task tracking. +/// Fan-out/fan-in runs on top of . A run is created durably, its tasks are claimed by +/// and executed here, and each finish is one transaction that also advances the run's +/// counts; the finish that completes the tasks queues the run's PostExecution as one more task. Storage holds +/// all of it, so there is nothing to recover after a restart: unfinished claims lapse and are taken again. /// -/// Flow: -/// 1. Scheduler triggers StartOrResumeRun with a planner script -/// 2. Planner runs on bg pool, returns JSON array of tasks -/// 3. Each task is dispatched through JobManager with priority ordering -/// 4. State is persisted to Azure Table Storage after every state change -/// 5. On restart, interrupted tasks resume from where they left off -/// 6. After 3 interruptions (host crash/reboot), a task is marked Failed -/// 7. PostExecStatus tracks PostExecution lifecycle for crash resilience +/// This process keeps only what is in flight: the pump's claimed buffer and the jobs running from it. /// public class OrchestratorService : IJobDescriptorStateWriter { @@ -31,143 +23,27 @@ public class OrchestratorService : IJobDescriptorStateWriter private readonly PowerShellRunnerService _psRunner; private readonly BackgroundTaskLimiter _limiter; private readonly JobManager _jobManager; - private readonly OrchestratorTableStore _store; - - /// The durable job queue. Its table is created alongside the orchestrator's; nothing - /// dispatches from it yet, so an existing deployment gains an empty table and nothing else. - private readonly JobQueueStore _queue; - private readonly OrchestratorStatusWriter _writer; + private readonly WorkStore _store; + private readonly ResultStore _results; private readonly CraftSettings _settings; - private readonly object _lock = new(); + private readonly FinishBatcher _finisher; private readonly ConcurrentDictionary _activePlanners = new(); - private readonly ConcurrentDictionary _activeRuns = new(); - private readonly ConcurrentDictionary> _childRuns = new(); - - /// - /// Child runs seen in storage at startup but not yet processed by . - /// They count as incomplete: without this, a parent processed BEFORE its child would find the child - /// absent from (nothing has resumed it yet) and conclude it had finished. - /// Each name is cleared as its run is processed, whichever way that goes. - /// - private readonly ConcurrentDictionary _recoveringChildren = new(); - - /// - /// Child runs queued through the bridge but not yet started, keyed by child name with a count - /// of outstanding queue entries. They count as incomplete for the same reason recovering - /// children do: between a task's script enqueuing a sub-orchestration and DrainPending getting - /// it into there is otherwise nothing for the parent's completion - /// check to see — and that window contains the parent's own last-task completion, because the - /// drain runs in the background while the enqueuing task is marked terminal immediately. - /// - private readonly ConcurrentDictionary _pendingChildRuns = new(); - private readonly ConcurrentDictionary _cancelledRuns = new(); - - /// - /// Last status line emitted per run — the (completed, failed, running, pending) tuple and when. Lets - /// skip re-emitting an identical line every 60s for a run that has not - /// changed (the dominant log volume at scale — thousands of runs parked at "0 running / N pending"), - /// while a slow heartbeat still proves a long-lived run is alive. Dropped at finalize. - /// - private readonly ConcurrentDictionary _lastStatusLog = new(); - - /// - /// Per-run re-drive backoff: when the storage verification in may - /// next run, and the interval it grew to. The re-drive is a watchdog for the rare orphaned-Pending task; - /// in steady state it reads storage and finds nothing, so once it does it backs off geometrically instead - /// of paying a full index read (+ a point read per candidate) on every 60s tick for every live run — the - /// dominant per-tick storage cost at scale. Snaps back to the base interval the moment it finds an orphan. - /// Dropped at finalize. - /// - private readonly ConcurrentDictionary _redriveBackoff = new(); - - /// Per-run status/re-drive tick cadence, from Orchestrator:StatusTimerIntervalSeconds. - private readonly TimeSpan _statusInterval; - - /// Re-drive backoff floor — one tick. The interval grows from here to . - private readonly TimeSpan _redriveBase; + private readonly ConcurrentDictionary _scripts = new(StringComparer.OrdinalIgnoreCase); + private readonly ConcurrentDictionary _lastStatusLog = new(); - /// Whether the re-drive backoff is active, from Orchestrator:RedriveBackoff. - private readonly bool _redriveBackoffEnabled; + /// Identifies this process's claims, and is what a lease is checked against. Unique per process start, + /// so a container restarted under the same host name never mistakes its predecessor's claims for its own. + public string Owner { get; } - /// Whether pending tasks shed their Parameters payload, from Orchestrator:ShedPendingParameters. - private readonly bool _shedParameters; + /// {host}/{pid}/{random}: readable in a claim row, unique per process start. + public static string NewOwnerId() => + $"{Environment.GetEnvironmentVariable("HOSTNAME") ?? Environment.MachineName}/{Environment.ProcessId}/{Guid.NewGuid():N}"[..^24]; - private static long _redriveStorageReads; + /// How long a claim is held before anyone may take it back. Longer than any task may run. + public TimeSpan Lease { get; } - /// - /// Count of storage verifications the re-drive has performed (a - /// call: one index-partition read + a point read per candidate). Instrumentation for the perf harness — - /// the backoff's whole purpose is to hold this down at high live-run counts. - /// - public static long RedriveStorageReads => Interlocked.Read(ref _redriveStorageReads); - - /// - /// Resolved task-script path per run. One entry per RUN (not per task), so a 738-task fan-out costs - /// one string reference instead of 738 captured ones. Populated at dispatch, dropped at finalize. - /// - private readonly ConcurrentDictionary _taskScriptPaths = new(); - - /// - /// How many times a run's post-execution may be attempted before it is abandoned. Matches the - /// per-task cap so recovery behaves the same at both levels: retry a few times across restarts, - /// then stop and release the storage rather than retrying forever. - /// - private const int MaxPostExecAttempts = 3; - - /// - /// How often an UNCHANGED run still emits a status line, so a long-lived run proves it is alive without - /// logging the identical line on every 60s tick. A real status change always logs immediately. - /// private static readonly TimeSpan StatusHeartbeat = TimeSpan.FromMinutes(10); - /// - /// Runs whose finalize has been claimed, so it happens once. Claimed in CheckRunCompletion, - /// released on the deferral/failure paths there, and cleared in DispatchPendingTasksAsync when a - /// run becomes live again. In-memory only: after a restart nothing has been finalized yet, so an - /// empty set is the correct starting state. - /// - private readonly ConcurrentDictionary _finalizingRuns = new(); - - /// - /// Sequential runs whose single driver job is currently executing. A sequential run's steps all run on - /// one pinned worker inside one driver, and the not-yet-run steps deliberately have no queue row — so - /// while a driver is active the re-drive must not treat those steps as orphaned and enqueue them (which - /// would spawn a second driver). Set when the driver starts, cleared when it finishes; in-memory only, - /// so after a crash it is empty and the re-drive/resume correctly re-triggers the driver. - /// - private readonly ConcurrentDictionary _activeSequentialDrivers = new(); - - /// - /// Get the Reference for a given run name, or null if not found/no reference set. - /// - public string? GetRunReference(string runName) - { - return _activeRuns.TryGetValue(runName, out var run) ? run.Reference : null; - } - - /// - /// Find a run name by its Reference value (exact match). - /// - public string? FindRunByReference(string reference) - { - return _activeRuns.Values - .FirstOrDefault(r => string.Equals(r.Reference, reference, StringComparison.OrdinalIgnoreCase)) - ?.Name; - } - - /// - /// Completes once startup recovery has finished — or been abandoned, see . - /// claims nothing before it. A claim taken earlier rehydrates the run into - /// _activeRuns ahead of recovery, recovery's own copy then loses the TryAdd, and the live graph - /// keeps the dead process's stale Running markers instead of recovery's reset. - /// - public Task RecoveryDone => _recoveryDone.Task; - private readonly TaskCompletionSource _recoveryDone = new(TaskCreationOptions.RunContinuationsAsynchronously); - - /// Open the claim gate. Called from a finally, so a recovery that throws or never runs - /// (shutdown mid-startup, storage down) still releases the pump rather than wedging it. - public void MarkRecoveryDone() => _recoveryDone.TrySetResult(); - private static readonly JsonSerializerOptions s_jsonOptions = new() { WriteIndented = true, @@ -175,14 +51,22 @@ public class OrchestratorService : IJobDescriptorStateWriter DefaultIgnoreCondition = JsonIgnoreCondition.WhenWritingNull }; + /// Kept for callers that read it; there is no re-drive any more. + public static long RedriveStorageReads => 0; + + /// Completes once startup has initialized storage; claims nothing before it. + public Task RecoveryDone => _recoveryDone.Task; + private readonly TaskCompletionSource _recoveryDone = new(TaskCreationOptions.RunContinuationsAsynchronously); + public void MarkRecoveryDone() => _recoveryDone.TrySetResult(); + public OrchestratorService( ILogger logger, PowerShellRunnerService psRunner, BackgroundTaskLimiter limiter, JobManager jobManager, - OrchestratorTableStore store, - JobQueueStore queue, - OrchestratorStatusWriter writer, + WorkStore store, + ResultStore results, + IConfiguration configuration, CraftSettings settings) { _logger = logger; @@ -190,135 +74,25 @@ public OrchestratorService( _limiter = limiter; _jobManager = jobManager; _store = store; - _queue = queue; - _writer = writer; + _results = results; _settings = settings; + Owner = NewOwnerId(); + Lease = TimeSpan.FromSeconds(Math.Max(60, configuration.GetValue("JobQueueLeaseSeconds", 1800))); + _finisher = new FinishBatcher(store, logger); + _store.AfterFinish = AfterFinishAsync; - // Status/re-drive tick cadence. Configurable so a constrained deployment can slow it (fewer ticks = - // less per-run overhead at high live-run counts) and so the perf harness can compress it to exercise - // the re-drive backoff quickly. The backoff floor is one tick. - _statusInterval = TimeSpan.FromSeconds(Math.Max(1, _settings.Orchestrator.StatusTimerIntervalSeconds)); - _redriveBase = _statusInterval; - _redriveBackoffEnabled = _settings.Orchestrator.RedriveBackoff; - _shedParameters = _settings.Orchestrator.ShedPendingParameters; - - // The queue holds descriptors; this is how they become work again at dispatch time, and how - // operator changes to a queued task are made durable. _jobManager.SetWorkResolver(ResolveTaskWorkAsync); _jobManager.SetDescriptorStateWriter(this); } - // ─── IJobDescriptorStateWriter ─── - // Operator actions on a QUEUED task. Both mutate the live run graph (so the in-memory view and - // CheckRunCompletion stay consistent) and queue a durable write through the same coalescing - // status writer the task lifecycle already uses — which guarantees a flush before the run - // finalizes and a final drain on shutdown. - - /// Persist a per-task priority override so recovery re-queues at the operator's priority. - public void PriorityChanged(JobDescriptor descriptor, int newPriority) - { - if (!TryFindLive(descriptor, out _, out var task)) return; - - // Rehydrate a shed payload before the durable write below (Replace mode) overwrites the stored - // ParametersJson with null. This is a rare, operator-initiated path, so the blocking read is fine. - if (_shedParameters && task.Parameters == null) - { - var p = _store.GetTaskParametersAsync(descriptor.RunName, task.Id).GetAwaiter().GetResult() ?? []; - lock (_lock) { task.Parameters ??= p; } - } - - lock (_lock) task.Priority = newPriority; - _writer.QueueTask(descriptor.RunName, task); - } - - /// - /// Persist a cancellation. Without this the task row stays Pending and - /// re-queues it after a restart — the job comes back. - /// - public void Cancelled(JobDescriptor descriptor) - { - if (!TryFindLive(descriptor, out var run, out var task)) return; - CancelLiveTask(run, task); - } - - /// - /// Mark a live task Cancelled in the run graph and queue the durable write. Returns false when the - /// task is already terminal — nothing to cancel, nothing to persist. With - /// , a Running task is also refused: the callers that pass it - /// found the task via a queue snapshot that may be seconds old, and cancelling a task that has - /// since dispatched would race its own completion write. - /// - private bool CancelLiveTask(OrchestratorRun run, OrchestratorTaskItem task, bool requirePending = false) - { - lock (_lock) - { - if (task.Status is "Completed" or "Failed" or "Cancelled") return false; - if (requirePending && task.Status != "Pending") return false; - task.Status = "Cancelled"; - task.LastError = "Cancelled by user"; - task.CompletedUtc = DateTime.UtcNow; - task.Parameters = null!; - // Cancelling the last outstanding task can complete the run. - CheckRunCompletion(run); - } - _writer.QueueTask(run.Name, task); - return true; - } - - /// - /// Cancel a task that exists only as a durable queue row — queued in the table, not (yet) claimed - /// into this instance's JobManager, which is where the worker-health page now sees most of a - /// backlog. - /// - /// Order is load-bearing: the run graph is marked terminal and the write queued BEFORE the queue - /// row is removed, because the orphan re-drive reads "Pending with no row" as work to re-queue — - /// remove the row first and the cancel un-does itself within a minute. The row removal itself is - /// best-effort: a row left behind is claimed, found terminal at rehydration, skipped and released. - /// - /// Returns false when the run is not live on this node or the task is already terminal; callers - /// should treat that as "nothing cancellable here". - /// - public async Task TryCancelQueuedTaskAsync(string runName, string taskId) - { - if (!TryFindLive(new JobDescriptor(runName, taskId, 0), out var run, out var task)) return false; - if (!CancelLiveTask(run, task, requirePending: true)) return false; - - try - { - await _queue.RemoveTaskAsync(runName, taskId); - } - catch (Exception ex) - { - _logger.LogWarning(ex, - "[Scheduler] Cancelled {Run}/{Task} but could not remove its queue row — it will be skipped at claim", - runName, taskId); - } - - return true; - } - - /// - /// Resolve a descriptor to its LIVE run and task instances. Operator actions only apply to QUEUED - /// jobs, whose run is by definition still active, so a miss means the job is no longer - /// cancellable/reprioritizable and there is nothing to persist. - /// - private bool TryFindLive(JobDescriptor descriptor, - [NotNullWhen(true)] out OrchestratorRun? run, - [NotNullWhen(true)] out OrchestratorTaskItem? task) - { - task = null; - if (!_activeRuns.TryGetValue(descriptor.RunName, out run)) return false; - lock (_lock) task = run.Tasks.FirstOrDefault(t => t.Id == descriptor.TaskId); - return task != null; - } + // ── starting runs ── /// - /// Start a new run or resume an interrupted one. - /// Called by the SchedulerService when an Orchestrator-type task fires. + /// Start a planner-based run: the planner script prints the task list. Skipped while a run of this name is + /// still going, so a timer that fires again does not stack outings. /// public async Task StartOrResumeRun(string name, string plannerPath, string taskPath, int priority, CancellationToken ct) { - // Prevent duplicate concurrent planners for the same run if (!_activePlanners.TryAdd(name, true)) { _logger.LogInformation("[Scheduler] Run {Name} already in progress, skipping", name); @@ -327,77 +101,17 @@ public async Task StartOrResumeRun(string name, string plannerPath, string taskP try { - await _store.InitializeAsync(); - await _queue.InitializeAsync(ct); - var run = await _store.GetRunAsync(name); - - if (run != null && run.Status == "Running") + if (await IsActiveAsync(name, ct)) { - // If tasks are already dispatched for this run, skip - if (_activeRuns.ContainsKey(name)) - { - _logger.LogInformation("[Scheduler] Run {Name} tasks already dispatched, skipping", name); - return; - } - - // Recover interrupted tasks - var recovered = 0; - var tasksToUpdate = new List(); - lock (_lock) - { - foreach (var task in run.Tasks.Where(t => t.Status == "Running")) - { - task.AttemptCount++; - if (task.AttemptCount >= 3) - { - task.Status = "Failed"; - task.LastError = $"Cancelled {task.AttemptCount} times by host interruption"; - _logger.LogWarning("[Scheduler] Task {TaskId} permanently failed after {Attempts} cancellations", - task.Id, task.AttemptCount); - } - else - { - task.Status = "Pending"; - _logger.LogInformation("[Scheduler] Resuming task {TaskId} attempt {Attempt}/3", - task.Id, task.AttemptCount + 1); - } - tasksToUpdate.Add(task); - recovered++; - } - } - - var pendingCount = run.Tasks.Count(t => t.Status == "Pending"); - if (pendingCount > 0) - { - if (recovered > 0) - { - foreach (var t in tasksToUpdate) - await _store.UpsertTaskAsync(run.Name, t); - } - _logger.LogInformation("[Scheduler] Resuming run {Name}: {Pending}/{Total} pending", - name, pendingCount, run.Tasks.Count); - await DispatchPendingTasksAsync(run, taskPath, run.Priority, ct); - return; - } - - // All tasks finished — finalize, and STOP. Finalize dispatches post-execution - // asynchronously, and its success path deletes the run's partitions by name - // (CleanupRunAsync). Falling through to start a fresh same-named outing here raced - // that delete and lost the new run's rows mid-flight; the next scheduler tick starts - // the fresh outing cleanly instead. - await FinalizeRunAsync(run); + _logger.LogInformation("[Scheduler] Run {Name} tasks already dispatched, skipping", name); return; } - // Start a new run _logger.LogInformation("[Scheduler] Starting orchestrator: {Name}", name); - string output; try { - output = await _limiter.RunAsync( - () => _psRunner.ExecuteScriptWithOutput(plannerPath), - $"Planner-{name}", ct); + output = await _limiter.RunAsync(() => _psRunner.ExecuteScriptWithOutput(plannerPath), $"Planner-{name}", ct); } catch (Exception ex) { @@ -413,23 +127,8 @@ public async Task StartOrResumeRun(string name, string plannerPath, string taskP return; } - run = new OrchestratorRun - { - Name = name, - Status = "Running", - Priority = priority, - StartedUtc = DateTime.UtcNow, - Tasks = tasks, - TaskScriptName = Path.GetFileNameWithoutExtension(taskPath) - }; - - await _store.UpsertRunAsync(run); - await _store.UpsertTaskBatchAsync(name, tasks); - // Seed the durable outstanding-task count alongside the tasks themselves, so completion - // is answerable from storage instead of by walking this run graph. - await _store.InitRemainingAsync(name, tasks.Count, ct); - _logger.LogInformation("[Scheduler] Run {Name} created with {Count} tasks at P{Priority}", name, tasks.Count, priority); - await DispatchPendingTasksAsync(run, taskPath, priority, ct); + await CreateAsync(name, tasks, priority, Path.GetFileNameWithoutExtension(taskPath), null, null, null, null, null, + new RunMode(false, 0, false), ct); } finally { @@ -437,19 +136,12 @@ public async Task StartOrResumeRun(string name, string plannerPath, string taskP } } - /// - /// Start a planner-based orchestrator run using the standard naming convention. - /// Resolves planner and task scripts from the command name, then delegates to StartOrResumeRun. - /// Called by OrchestratorBridge.QueuePlannerRun (fire-and-forget from PowerShell). - /// + /// Start a planner run by command name: planner Start-X, tasks Invoke-XTask. public async Task StartPlannerRunAsync(string command, int priority, CancellationToken ct) { var plannerFunc = _psRunner.FindScript(command); - var baseName = command.StartsWith("Start-", StringComparison.OrdinalIgnoreCase) - ? command[6..] - : command; - var taskScriptName = $"Invoke-{baseName}Task"; - var taskFunc = _psRunner.FindScript(taskScriptName); + var baseName = command.StartsWith("Start-", StringComparison.OrdinalIgnoreCase) ? command[6..] : command; + var taskFunc = _psRunner.FindScript($"Invoke-{baseName}Task"); if (plannerFunc != null && taskFunc != null) { @@ -459,1785 +151,732 @@ public async Task StartPlannerRunAsync(string command, int priority, Cancellatio else { _logger.LogWarning("[Orchestrator] Scripts not found for planner run: {Command} planner={Planner} task={Task}", - command, command, taskScriptName); + command, command, $"Invoke-{baseName}Task"); } } /// - /// Resume any runs that were interrupted by a previous process crash. - /// Called once on application startup. + /// Start a run from a pre-built batch (OrchestratorBridge). The batch is a JSON Lines file + /// (, deleted on every path) or a JSON array string. Returns whether a run + /// was created: false when the batch is empty, or when is false and a + /// run of this name is still going. By default runs of one name stack up side by side. + /// above 0 caps how many of the run's tasks run at once; it has no meaning + /// for a sequential run (one step at a time already) and is dropped there. + /// makes a sequential run cancel its remaining steps at the first failure; other runs always carry on. /// - public async Task ResumeInterruptedRunsAsync(CancellationToken ct) + public async Task StartFromBatchAsync(string name, string batchJson, int priority, + string? postExecFunctionName, string? postExecParametersJson, CancellationToken ct, + string? parentRunName = null, string? reference = null, string? batchFilePath = null, + bool sequential = false, string? parentRunKey = null, string? childKey = null, bool allowCollision = true, + int maxConcurrency = 0, bool stopOnFailure = false) { - await _store.InitializeAsync(); - await _queue.InitializeAsync(ct); - - // Rebuild parent→child links BEFORE processing anything. _childRuns is in-memory, so without - // this a resumed parent has no registered children, AllChildRunsComplete answers true, and the - // parent can finalize (and fire PostExecution) while its children are still running — exactly - // what the child-run guard exists to prevent. One partition scan, no task rows. - var summaries = await _store.ListRunSummariesAsync(); - var runningRuns = summaries.Where(s => s.Status == "Running").Select(s => s.Name) - .ToHashSet(StringComparer.Ordinal); - var reattached = 0; - foreach (var child in summaries) - { - if (string.IsNullOrEmpty(child.ParentRunName)) continue; - // Rows written before the self-parent guard can record a run as its own parent; - // reattaching one would rebuild the self-wait this fix removes. - if (child.ParentRunName == child.Name) continue; - if (child.Status is not "Running") continue; // terminal children cannot block a parent - // The parent must itself be resuming. Lineage can point at a run that already - // finalized — a run queued from PostExecution records its finished spawner as parent — - // and reattaching those would build bags no completion check consults, or worse, gate - // the NEXT outing of a recurring parent name on a leftover of the previous one. - if (!runningRuns.Contains(child.ParentRunName)) continue; - _childRuns.GetOrAdd(child.ParentRunName, _ => new ConcurrentBag()).Add(child.Name); - _recoveringChildren.TryAdd(child.Name, true); - reattached++; - } - if (reattached > 0) - _logger.LogInformation("[Scheduler] Reattached {Count} in-flight child runs to their parents", reattached); - - // Recovery emits ONE aggregate line, not a handful per run: at scale a crash-loop replayed thousands - // of per-run "Found/Resuming/Released/Dispatched" lines on every restart. Per-run detail is kept at - // Debug; the counts below carry the summary. Genuine problems (unresumable, post-exec abandoned) still - // log at their own level as they happen. - var resumed = 0; var pendingTotal = 0; var postExecResumed = 0; - var staleReleased = 0; var finalizedNow = 0; var unresumable = 0; var postExecGaveUp = 0; - foreach (var runName in summaries.Select(s => s.Name)) + string? gate = null; + try { - try + name = TableKeys.Sanitize(name); + if (!allowCollision) { - var run = await _store.GetRunAsync(runName); - if (run == null) continue; - - // Check for runs whose PostExec was pending/running when we crashed, or failed outright. - // - // "Failed" belongs here: post-execution IS the aggregation, so a failed one means the - // run's whole point never happened (no cached permissions, no applied standards) and its - // Results rows were never cleaned up, because CleanupRunAsync only runs on success. This - // used to be excluded, which quietly contradicted the catch block's own comment that a - // failure "will be retried on next startup". - if (run.Status is "Completed" or "CompletedWithErrors" - && run.PostExecStatus is "Pending" or "Running" or "Failed") + var family = WorkStore.CollisionFamily(name); + if (_activePlanners.TryAdd(family, true)) gate = family; + if (gate == null || await _store.IsFamilyActiveAsync(family, ct)) { - if (run.PostExecAttemptCount >= MaxPostExecAttempts) - { - // Out of retries. Say so once and release the storage, rather than re-reading - // this run's rows on every start for the life of the deployment. - _logger.LogError( - "[Scheduler] PostExecution for {Name} failed {Count} times — giving up and cleaning up results", - run.Name, run.PostExecAttemptCount); - run.PostExecStatus = "Abandoned"; - await _store.UpsertRunAsync(run); - await _store.CleanupRunAsync(run.Name); - await _queue.RemoveRunAsync(run.Name, ct); - postExecGaveUp++; - continue; - } - - _logger.LogDebug( - "[Scheduler] Resuming interrupted PostExecution for run: {Name} (PostExecStatus={Status}, attempt {Attempt}/{Max})", - run.Name, run.PostExecStatus, run.PostExecAttemptCount + 1, MaxPostExecAttempts); - DispatchPostExecution(run); - postExecResumed++; - continue; - } - - if (run.Status != "Running") continue; - - _logger.LogDebug("[Scheduler] Found interrupted run: {Name}", run.Name); - - // Use the stored task script name, fall back to naming convention - var taskPath = !string.IsNullOrEmpty(run.TaskScriptName) - ? _psRunner.FindScript(run.TaskScriptName) - : FindTaskScript(run.Name); - if (taskPath == null) - { - _logger.LogWarning("[Scheduler] Cannot resume {Name}: task script not found (tried {Script})", - run.Name, run.TaskScriptName ?? $"Invoke-{run.Name}Task"); - unresumable++; - continue; - } - - // Recover interrupted tasks - var tasksToUpdate = new List(); - lock (_lock) - { - foreach (var task in run.Tasks.Where(t => t.Status == "Running")) - { - task.AttemptCount++; - if (task.AttemptCount >= 3) - { - task.Status = "Failed"; - task.LastError = $"Cancelled {task.AttemptCount} times by host interruption"; - } - else - { - task.Status = "Pending"; - } - tasksToUpdate.Add(task); - } - } - - // Persist recovered task state changes - foreach (var t in tasksToUpdate) - await _store.UpsertTaskAsync(run.Name, t); - - var pending = run.Tasks.Count(t => t.Status == "Pending"); - if (pending > 0) - { - // Hand back the claims the dead process was holding. Reaching here means this run was - // interrupted, so every lease on its rows belongs to a process that no longer exists — - // but the lease itself is still live as far as storage is concerned, so nothing can - // claim those rows until it lapses. Re-dispatch below deliberately does not paper over - // that by writing duplicate rows, which is what used to hide it, so without this the - // run stalls for the remainder of the lease: measured as 12 tasks Pending with nothing - // running for the balance of a 30 minute lease after a kill. - try - { - var released = await _queue.ReleaseRunClaimsAsync(run.Name, ct); - if (released > 0) - { - staleReleased += released; - _logger.LogDebug( - "[Scheduler] Released {Count} stale claim(s) held by the previous process for {Name}", - released, run.Name); - } - } - catch (Exception ex) - { - // Not fatal: the leases still lapse on their own, just slowly. - _logger.LogWarning(ex, "[Scheduler] Could not release stale claims for {Name}", run.Name); - } - - _logger.LogDebug("[Scheduler] Resuming interrupted run {Name}: {Pending} pending", run.Name, pending); - resumed++; pendingTotal += pending; - await DispatchPendingTasksAsync(run, taskPath, run.Priority, ct, quiet: true); - } - else - { - await FinalizeRunAsync(run); - finalizedNow++; + _logger.LogWarning("[Orchestrator] Run {Name} skipped: a run of that name is still active and collisions are off", name); + return false; } } - catch (Exception ex) + + var tasks = !string.IsNullOrEmpty(batchFilePath) + ? ParseTasksFromJsonLinesFile(batchFilePath, name) + : ParseTasksFromJson(batchJson, name); + if (tasks.Count == 0) { - _logger.LogError(ex, "[Scheduler] Failed to process interrupted run: {Name}", runName); + _logger.LogWarning("[Orchestrator] Batch for {Name} produced 0 tasks", name); + return false; } - finally + + var genericTaskFunc = _settings.Orchestrator.GenericTaskFunction; + if (string.IsNullOrEmpty(genericTaskFunc) || Script(genericTaskFunc) == null) { - // This run is no longer awaiting recovery, however it went. If it was resumed it is now - // in _activeRuns and still blocks its parent; if it finalized or could not be resumed, - // it must stop blocking. Clearing here covers every path including the failure one. - _recoveringChildren.TryRemove(runName, out _); + _logger.LogError("[Orchestrator] Cannot start {Name}: task function {Func} not found", name, genericTaskFunc); + return false; } - } - if (resumed + finalizedNow + postExecResumed + unresumable + postExecGaveUp > 0) - _logger.LogInformation( - "[Scheduler] Crash recovery: resumed {Resumed} run(s) ({Pending} pending tasks re-dispatched), " + - "{PostExec} post-execution(s), {Finalized} finalized, {Stale} stale claim(s) released" + - "{Unresumable}{GaveUp}", - resumed, pendingTotal, postExecResumed, finalizedNow, staleReleased, - unresumable > 0 ? $", {unresumable} unresumable" : "", - postExecGaveUp > 0 ? $", {postExecGaveUp} post-exec abandoned" : ""); - - // First retention pass, now that every run that could be resumed is back in _activeRuns and so - // exempt from the abandoned-run rule. The scheduler keeps it going on an interval from here. - try - { - await RunRetentionSweepAsync(ct); + if (string.IsNullOrEmpty(postExecFunctionName)) postExecFunctionName = null; + if (string.IsNullOrEmpty(postExecParametersJson)) postExecParametersJson = null; + await CreateAsync(name, tasks, priority, genericTaskFunc, postExecFunctionName, postExecParametersJson, + reference, parentRunKey, childKey, ResolveMode(name, sequential, maxConcurrency, stopOnFailure), ct); + return true; } - catch (Exception ex) + finally { - _logger.LogWarning(ex, "[Scheduler] Startup retention sweep failed"); + if (gate != null) _activePlanners.TryRemove(gate, out _); + if (!string.IsNullOrEmpty(batchFilePath)) + { + try { if (File.Exists(batchFilePath)) File.Delete(batchFilePath); } + catch (Exception ex) { _logger.LogDebug(ex, "[Orchestrator] Failed to delete batch file {Path}", batchFilePath); } + } } } - /// - /// One retention pass over the orchestrator tables: finished runs past - /// Orchestrator:RetentionHours, runs nobody is driving that have not been written to for that - /// long, and Tasks/Results partitions whose Run row is already gone. Runs at the end of startup - /// recovery and then every Orchestrator:CleanupIntervalHours via . - /// - public async Task RunRetentionSweepAsync(CancellationToken ct) - { - var retention = TimeSpan.FromHours(Math.Max(1, _settings.Orchestrator.RetentionHours)); - var active = _activeRuns.Keys.ToHashSet(StringComparer.Ordinal); - var result = await _store.CleanupOldRunsAsync(retention, active, ct); + private async Task IsActiveAsync(string name, CancellationToken ct) => + (await _store.GetActiveRunsAsync(name, ct)).Count > 0; - // An abandoned run can still have rows in the durable queue. The pump would drop each as a - // stale descriptor when it came to claim it — but only after paying for the claim. - foreach (var name in result.AbandonedRuns) - { - try - { - await _queue.RemoveRunAsync(name, ct); - } - catch (Exception ex) - { - _logger.LogWarning(ex, "[Scheduler] Could not remove queue rows for abandoned run {Name}", name); - } - } + /// Whether any run of this name's collision family () is unfinished. + public Task IsRunActiveAsync(string name, CancellationToken ct = default) => + _store.IsFamilyActiveAsync(WorkStore.CollisionFamily(TableKeys.Sanitize(name)), ct); - return result; + /// The runs an operator action names: the run with that key, or every unfinished run of that name. + private async Task> TargetRunsAsync(string keyOrName) + { + keyOrName = TableKeys.Sanitize(keyOrName); + if (keyOrName.Contains('~') && await _store.GetRunAsync(keyOrName) is { IsFinished: false } run) return [run]; + return await _store.GetActiveRunsAsync(keyOrName); } - /// - /// Periodic retention sweeps for the life of the host, started by the scheduler once recovery has - /// run. A sweep that fails is logged and tried again next interval; CleanupIntervalHours of 0 - /// leaves only the startup pass. - /// - public async Task RunRetentionLoopAsync(CancellationToken ct) - { - var hours = _settings.Orchestrator.CleanupIntervalHours; - if (hours <= 0) - { - _logger.LogInformation( - "[Scheduler] Periodic retention sweep disabled (CleanupIntervalHours={Hours}); only the startup pass runs", hours); - return; - } + /// How a run's tasks are scheduled: one at a time on a pinned worker (sequential), at most N at once, + /// or all at once; and whether a sequential run stops at its first failure. + internal readonly record struct RunMode(bool Sequential, int MaxConcurrency, bool StopOnFailure); - var interval = TimeSpan.FromHours(hours); - using var timer = new PeriodicTimer(interval); - try + private RunMode ResolveMode(string name, bool sequential, int maxConcurrency, bool stopOnFailure) + { + maxConcurrency = Math.Max(0, maxConcurrency); + if (sequential && maxConcurrency > 0) { - while (await timer.WaitForNextTickAsync(ct)) - { - try - { - await RunRetentionSweepAsync(ct); - } - catch (OperationCanceledException) when (ct.IsCancellationRequested) - { - return; - } - catch (Exception ex) - { - _logger.LogWarning(ex, "[Scheduler] Retention sweep failed; next attempt in {Interval}", interval); - } - } + _logger.LogWarning("[Orchestrator] Run {Name}: MaxConcurrency {Max} ignored, a sequential run already runs one step at a time", + name, maxConcurrency); + maxConcurrency = 0; } - catch (OperationCanceledException) + if (stopOnFailure && !sequential) { - // Host shutdown. + _logger.LogWarning("[Orchestrator] Run {Name}: StopOnFailure ignored, it applies to sequential runs only", name); + stopOnFailure = false; } + return new RunMode(sequential, maxConcurrency, stopOnFailure); } - /// - /// One loop that ticks every live run's status / re-drive / completion-recheck, at - /// . It replaces the per-run that used to do this: at - /// high live-run counts that meant one Timer object per run and a continuous stream of fire-and-forget - /// callbacks onto the thread pool (~M/interval per second), whereas one sweep over - /// is a single scheduling source that allocates nothing per run. The per-run work stays cheap — - /// skips an unchanged line, is gated by - /// its backoff and returns synchronously when backed off, and - /// short-circuits — so ticking thousands of runs in one pass is fast. A throw for one run is logged and - /// neither stops the sweep nor takes the host down (a Timer callback that threw would have crashed it). - /// Started once by , alongside the retention loop. - /// - public async Task RunStatusSweepLoopAsync(CancellationToken ct) + private async Task CreateAsync(string name, List tasks, int priority, string taskScriptName, + string? postExecFunctionName, string? postExecParametersJson, string? reference, string? parentRunKey, + string? childKey, RunMode mode, CancellationToken ct) { - using var timer = new PeriodicTimer(_statusInterval); - try - { - while (await timer.WaitForNextTickAsync(ct)) - { - foreach (var run in _activeRuns.Values) - { - try - { - LogRunStatus(run); - RedrivePendingTasks(run); - lock (_lock) { CheckRunCompletion(run); } - } - catch (Exception ex) - { - _logger?.LogWarning(ex, "[Scheduler] Run status tick failed for {Name}", run?.Name); - } - } - } - } - catch (OperationCanceledException) + var started = NextStartTime(); + var header = new RunHeader + { + RunKey = WorkStore.RunKeyFor(name, started), + Name = name, + Priority = Math.Clamp(priority, 0, 99), + StartedUtc = started, + TaskScriptName = taskScriptName, + PostExecFunctionName = postExecFunctionName, + PostExecParametersJson = postExecParametersJson, + Reference = reference, + ParentRunKey = parentRunKey, + ParentChildKey = childKey, + Sequential = mode.Sequential, + MaxConcurrency = mode.MaxConcurrency, + StopOnFailure = mode.StopOnFailure, + }; + await _store.CreateRunAsync(header, tasks.Select(t => new WorkStore.NewTask(t.Id, t.Parameters)).ToList(), ct); + _logger.LogInformation("[Orchestrator] Run {Name} created with {Count} tasks at P{Priority}{PostExec}{Mode}", + name, tasks.Count, header.Priority, + postExecFunctionName != null ? $" (PostExec: Push-{postExecFunctionName})" : "", + mode.Sequential ? (mode.StopOnFailure ? " (sequential, stop on failure)" : " (sequential)") + : mode.MaxConcurrency > 0 ? $" (max {mode.MaxConcurrency} at once)" : ""); + } + + private static long s_lastStartTicks; + + /// Strictly increasing start times, so runs of one name started together still get distinct keys. + private static DateTime NextStartTime() + { + while (true) { - // Host shutdown. + var last = Interlocked.Read(ref s_lastStartTicks); + var next = Math.Max(DateTime.UtcNow.Ticks, last + 1); + if (Interlocked.CompareExchange(ref s_lastStartTicks, next, last) == last) return new DateTime(next, DateTimeKind.Utc); } } + // ── child runs ── + /// - /// Start an orchestrator run from a pre-built batch. - /// Called by OrchestratorBridge.DrainPending() when PowerShell's Start-CIPPOrchestrator - /// queues a run on CIPPNG (bypassing the planner script phase). - /// - /// The batch arrives one of two ways. , when set, is a JSON Lines - /// file with one task object per line and is the preferred form: the caller writes it a task at a - /// time and this reads it a line at a time, so no single string ever holds the whole batch. - /// is the original whole-array-in-one-string form, kept for callers - /// that still use it — it costs the full batch as a string here AND again as a JsonDocument. - /// A file path wins when both are given; the file is deleted once parsed. + /// Make wait for a child that is about to be queued. Called at enqueue time, + /// inside the parent's task, so the parent cannot finish first. is the parent's + /// run key (exact, from the stamped context's RunKey) or, from older callers, its name (the newest outing). + /// Returns the parent's run key and the placeholder key the child completes, or null when the parent is + /// not running tasks (a run queued from an aggregation is not a child) or has the child's name (a run + /// re-queueing itself for its next cycle). /// - public async Task StartFromBatchAsync(string name, string batchJson, int priority, - string? postExecFunctionName, string? postExecParametersJson, CancellationToken ct, - string? parentRunName = null, string? reference = null, string? batchFilePath = null, - bool sequential = false) + internal (string ParentRunKey, string ChildKey)? RegisterPendingChild(string parentRun, string childRunName) { - // The batch file is this method's to dispose of, on EVERY path — including the two - // "already running, skipping" returns below, which never look at it. Those are the common - // case for a duplicate enqueue, so leaving cleanup at the parse site would quietly fill the - // container's temp directory with the batches of runs that were skipped rather than started. - try - { - await StartFromBatchCoreAsync(name, batchJson, priority, postExecFunctionName, - postExecParametersJson, parentRunName, reference, batchFilePath, sequential, ct); - } - finally - { - if (!string.IsNullOrEmpty(batchFilePath)) - { - try { if (File.Exists(batchFilePath)) File.Delete(batchFilePath); } - catch (Exception ex) - { - _logger.LogDebug(ex, "[Orchestrator] Failed to delete batch file {Path}", batchFilePath); - } - } - } + if (parentRun == childRunName) return null; + return Task.Run(async () => + { + var parent = await _store.ResolveRunAsync(parentRun); + if (parent is not { Phase: RunPhase.Tasks } || parent.Name == childRunName) return ((string, string)?)null; + var childKey = $"{childRunName}|{Guid.NewGuid():N}"; + if (!await _store.AddChildAsync(parent.RunKey, childKey)) return null; + _logger.LogInformation("[Orchestrator] Registered child run {Child} under parent {Parent}", childRunName, parent.Name); + return (parent.RunKey, childKey); + }).GetAwaiter().GetResult(); } - private async Task StartFromBatchCoreAsync(string name, string batchJson, int priority, - string? postExecFunctionName, string? postExecParametersJson, - string? parentRunName, string? reference, string? batchFilePath, bool sequential, CancellationToken ct) + /// A registered child that was never created: stop the parent waiting for it. + internal async Task AbandonPendingChildAsync(string parentRunKey, string childKey) { - // Run names become PartitionKeys verbatim, and batch names carry user-typed task names - // ("Alert on Entra ID P1/P2 …"). An illegal key character 400s every write for the run — - // run row, task rows, counter, queue rows — identically forever, so the run can neither - // start nor be re-driven. Sanitize at the boundary, like task ids at mint. - name = TableKeys.Sanitize(name); - if (!string.IsNullOrEmpty(parentRunName)) - parentRunName = TableKeys.Sanitize(parentRunName); - - // A run cannot be its own parent. The ambient RunName rides along when a run is re-queued - // from inside its own context; persisting it would feed the reattach loop a self-link on the - // next start and show circular lineage in the status APIs. - if (parentRunName == name) - parentRunName = null; + await _store.FinishAsync(parentRunKey, [new WorkStore.Finish(0, "Completed", ChildKey: childKey)], null); + } - if (!_activePlanners.TryAdd(name, true)) - { - _logger.LogInformation("[Orchestrator] Run {Name} already in progress, skipping", name); - return; - } - - try - { - await _store.InitializeAsync(); - await _queue.InitializeAsync(ct); - - var existing = await _store.GetRunAsync(name); - if (existing != null && existing.Status == "Running" && _activeRuns.ContainsKey(name)) - { - _logger.LogInformation("[Orchestrator] Run {Name} already active, skipping", name); - return; - } - - List tasks; - if (!string.IsNullOrEmpty(batchFilePath)) - { - var batchBytes = File.Exists(batchFilePath) ? new FileInfo(batchFilePath).Length : 0; - _logger.LogDebug( - "[Orchestrator] Batch for {Name} streamed from {Path} ({KB:F1} KB on disk, never held whole)", - name, batchFilePath, batchBytes / 1024.0); - - tasks = ParseTasksFromJsonLinesFile(batchFilePath, name); - } - else - { - // Sizing the legacy inbound batch: the whole run's task list as ONE string, built by - // ConvertTo-Json in the calling PowerShell, held in the bridge queue, and parsed here - // into a JsonDocument that holds it a second time. - _logger.LogDebug( - "[Orchestrator] Batch for {Name} is {Chars} chars (~{ApproxKB:F1} KB in memory)", - name, batchJson.Length, batchJson.Length * 2 / 1024.0); - - tasks = ParseTasksFromJson(batchJson, name); - } - - if (tasks.Count == 0) - { - _logger.LogWarning("[Orchestrator] Batch for {Name} produced 0 tasks", name); - return; - } - - // Stamp payload order. Sequential dispatch reads this to run tasks one at a time in the order - // they were submitted; fan-out ignores it. The parser preserves batch order, so the index is it. - for (var i = 0; i < tasks.Count; i++) tasks[i].Sequence = i; - - var genericTaskFunc = _settings.Orchestrator.GenericTaskFunction; - if (string.IsNullOrEmpty(genericTaskFunc)) - { - _logger.LogError("[Orchestrator] Cannot start {Name}: App:Orchestrator:GenericTaskFunction not configured", name); - return; - } - var taskPath = _psRunner.FindScript(genericTaskFunc); - if (taskPath == null) - { - _logger.LogError("[Orchestrator] Cannot start {Name}: {Func} not found", name, genericTaskFunc); - return; - } - - var run = new OrchestratorRun - { - Name = name, - Reference = reference, - Status = "Running", - Priority = priority, - StartedUtc = DateTime.UtcNow, - Tasks = tasks, - TaskScriptName = genericTaskFunc, - PostExecFunctionName = postExecFunctionName, - PostExecParametersJson = postExecParametersJson, - ParentRunName = parentRunName, - Sequential = sequential - }; - - await _store.UpsertRunAsync(run); - await _store.UpsertTaskBatchAsync(name, tasks); - // Seed the durable outstanding-task count alongside the tasks themselves, so completion - // is answerable from storage instead of by walking this run graph. - await _store.InitRemainingAsync(name, tasks.Count, ct); - // IsNullOrEmpty, not != null: PowerShell marshals a $null argument to an empty string, so a - // run with no post-execution arrives here with "" and a null check logs the meaningless - // "(PostExec: Push-)". Everything that acts on this field already uses IsNullOrEmpty — - // only the log disagreed. - var postExecSuffix = !string.IsNullOrEmpty(postExecFunctionName) - ? $" (PostExec: Push-{postExecFunctionName})" - : ""; - _logger.LogInformation( - "[Orchestrator] Run {Name} created from batch: {Count} tasks P{Priority}{PostExec}", - name, tasks.Count, priority, postExecSuffix); - await DispatchPendingTasksAsync(run, taskPath, priority, ct); - } - finally - { - _activePlanners.TryRemove(name, out _); - } - } + // ── what happens when tasks finish ── - private async Task DispatchPendingTasksAsync(OrchestratorRun run, string taskPath, int priority, - CancellationToken ct, bool quiet = false) + private async Task AfterFinishAsync(WorkStore.FinishOutcome outcome) { - _activeRuns.TryAdd(run.Name, run); - // Registered before anything is enqueued — the resolver reads it on the dispatch side. - _taskScriptPaths[run.Name] = taskPath; - - // This run is live again, so it is allowed to finalize again. Matters for the stable run names - // (CIPPDBCacheOrchestrator, ProcessDeltaQueries) that recur within one process lifetime — without - // this, the finalize claim from their previous outing would strand the next one forever. - _finalizingRuns.TryRemove(run.Name, out _); - - // No per-run timer: the single RunStatusSweepLoopAsync ticks every run in _activeRuns (which this - // run was just added to). One sweep replaces what used to be a System.Threading.Timer per run — at - // high live-run counts that was thousands of Timer objects and a continuous drizzle of fire-and-forget - // callbacks onto the thread pool. The sweep still runs LogRunStatus + RedrivePendingTasks + - // CheckRunCompletion for each run at the same cadence (the last re-run on purpose: a finalize deferred - // while storage showed work outstanding has no task transition left to retrigger it). - - var pending = run.Tasks.Where(t => t.Status == "Pending").ToList(); - - // Skip anything that already has a queue row. - // - // A queue RowKey is BuildRowKey(queuedUtc, run, task), so it is idempotent only for a GIVEN - // timestamp — re-dispatching the same task later writes a SECOND row rather than updating the - // first. That is exactly what crash recovery does: ResumeInterruptedRunsAsync flips interrupted - // tasks back to Pending and calls this, while every pre-crash row for those tasks is still in the - // queue. Measured on a killed 140-task fanout: 102 tasks re-dispatched on top of 102 survivors. - // - // Most duplicates are harmless — the second row resolves to a task that has since finished and is - // dropped as a stale descriptor. But if both rows are claimed while the task is still RUNNING, - // the resolver's terminal-status guard does not apply and the task runs twice. That happened: - // Intune_dev.mspadvisors.com was claimed again 5 minutes into its own execution and ran a second - // time. Not writing the duplicate is the fix; the resolver check below is the backstop. - // - // A read of the run's rows costs one table scan per dispatch, against a write per task avoided. - HashSet alreadyQueued; - try - { - alreadyQueued = await _queue.GetQueuedTaskIdsAsync(run.Name, ct); - } - catch (Exception ex) - { - // Enqueuing a duplicate is recoverable; enqueuing nothing strands the run. Prefer the former. - _logger.LogWarning(ex, - "[Scheduler] Could not read existing queue rows for {Run} — dispatching without de-duplication", - run.Name); - alreadyQueued = []; - } - - List toQueue; - if (run.Sequential) - { - // Sequential mode: a single entry row starts the ONE pinned driver that then runs every step of - // the run inline (see BuildSequentialRunWork), so exactly ONE queue row ever exists for the run — - // the lowest-Sequence task still Pending — and the later steps get no rows at all. Enqueue that - // one task, and only if it does not already have a row. A second row for the same run would let a - // second driver start and break the pinning/ordering. This covers both first dispatch (enqueues - // Sequence 0 to start the driver) and resume (re-enqueues the current step only if its row is - // gone, so a fresh driver picks the run back up where it left off). - var current = pending.OrderBy(t => t.Sequence).FirstOrDefault(); - toQueue = current != null && !alreadyQueued.Contains(current.Id) ? [current] : []; - } - else - { - toQueue = pending.Where(t => !alreadyQueued.Contains(t.Id)).ToList(); - } - - // Shed the Parameters payload BEFORE the tasks become claimable. The caller has already persisted - // them (UpsertTaskBatchAsync), so the Tasks table is authoritative; the live graph keeps each task - // object (its identity + Status drive completion tracking) but drops the payload it does not need - // while it waits, and BuildTaskWork rehydrates it from storage at dispatch. This is what bounds the - // retained memory of a large pending backlog — thousands of runs each holding every task's payload is - // what walks the live-set into the GC heap ceiling. Ordering matters: shedding AFTER the enqueue - // could null a task the pump had already claimed and whose BuildTaskWork had just rehydrated it, so - // shed here, before EnqueueBatchAsync makes the rows claimable. Only toQueue (the rows enqueued in - // THIS call) is touched — a task already queued from a prior call may be mid-dispatch. Sequential - // runs are exempt: they are small (a handful of ordered steps) and their pinned driver holds every - // step's payload in the live graph while it runs, so shedding would buy no memory and only add reads. - if (_shedParameters && !run.Sequential) - { - lock (_lock) - foreach (var t in toQueue) - if (t.Status == "Pending") t.Parameters = null!; - } + var h = outcome.Header; + if (!outcome.Completed) return; + + _lastStatusLog.TryRemove(h.RunKey, out _); + var wall = (h.CompletedUtc ?? DateTime.UtcNow) - h.StartedUtc; + _logger.LogInformation("[Scheduler] Run {Name} finalized: {Status} ({Completed}/{Failed}/{Cancelled}/{Total}) wall={Wall} {Memory}", + h.Name, h.Status, h.Done - h.Failed - h.Cancelled, h.Failed, h.Cancelled, h.Total, + wall.TotalSeconds < 60 ? $"{wall.TotalSeconds:F1}s" : $"{wall.TotalMinutes:F1}min", + BackgroundTaskLimiter.GetMemorySnapshot()); - // One batched write per priority bucket rather than one per task. The queue is the backlog now; - // the JobManager only ever sees the batch JobQueuePump claims from it. - await _queue.EnqueueBatchAsync(run.Name, - toQueue.Select(t => (t.Id, t.Priority ?? priority)).ToList(), - DateTime.UtcNow, ct); + try { await _results.DeleteRunAsync(h.RunKey); } + catch (Exception ex) { _logger.LogDebug(ex, "[Orchestrator] Could not drop results of {Run}", h.Name); } - // quiet = called from crash recovery, where a per-run line per resumed run is the flood the - // aggregate summary replaces — drop to Debug. A normal orchestration start logs it at Info (one line). - var level = quiet ? LogLevel.Debug : LogLevel.Information; - if (toQueue.Count == pending.Count) + if (h.ParentRunKey is { } parentKey && h.ParentChildKey is { } childKey) { - _logger.Log(level, "[Scheduler] Dispatched {Count} tasks for {Name} at P{Priority}", - toQueue.Count, run.Name, priority); - } - else - { - _logger.Log(level, - "[Scheduler] Dispatched {Count} tasks for {Name} at P{Priority} ({Existing} already queued)", - toQueue.Count, run.Name, priority, pending.Count - toQueue.Count); + await _store.FinishAsync(parentKey, + [new WorkStore.Finish(0, h.Status == "Completed" ? "Completed" : "Failed", ChildKey: childKey)], null); } } - /// - /// Build the work for a SEQUENTIAL run: a single delegate that checks out ONE background worker and - /// runs every step of the run on it, in payload (Sequence) order, one at a time, then reclaims the - /// worker once at the end. This is what "pin one worker for the whole run" means — the run starts on a - /// worker and stays there until it finishes, never going back to the pool between steps to be - /// re-scheduled onto a different one. The run is dispatched as a SINGLE queue row (the entry task; see - /// DispatchPendingTasksAsync), so exactly one JobManager slot and one worker are held for the run's - /// whole duration and the mid-run steps never get their own rows. - /// - /// Failure policy is best-effort: a step that throws is recorded Failed and the driver moves on to the - /// next, so one bad step cannot strand the rest and the run ends CompletedWithErrors. Each step runs - /// through InvokeAsync, whose finally resets the runspace after success, failure OR cancellation, so - /// steps stay isolated on the shared worker. Registered in for - /// its whole life so the re-drive leaves the not-yet-reached (deliberately row-less) steps alone while - /// the driver is progressing. - /// - private Func BuildSequentialRunWork(OrchestratorRun run, string taskPath) - { - return async (jobCt) => - { - // One driver per run. Registered for the driver's whole life so the re-drive does not mistake - // the run's row-less pending steps for orphans and enqueue them. TryAdd (not an unconditional - // set) also closes the duplicate-entry-row race: if two rows for the same run are claimed before - // either driver marks the entry step Running, the loser here does nothing and the winner runs - // every step. Because this guard is OUTSIDE the try/finally, the loser never runs the finally - // that would otherwise remove the winner's registration. - if (!_activeSequentialDrivers.TryAdd(run.Name, true)) - { - _logger.LogDebug( - "[Scheduler] Sequential run {Run} already has an active driver — skipping duplicate entry", run.Name); - return; - } - PowerShellWorker? worker = null; - var workerFaulted = false; - try - { - // One worker for the whole run: checked out once here, reclaimed once in the finally. - worker = CheckoutSequentialWorker(jobCt); - - while (true) - { - OrchestratorTaskItem? task; - lock (_lock) - { - task = run.Tasks.Where(t => t.Status == "Pending") - .OrderBy(t => t.Sequence) - .FirstOrDefault(); - } - if (task == null) break; // every step is terminal — the run is done + // ── executing a claimed task ── - // Run cancelled while we were working through it: mark this and every remaining step - // Cancelled and stop. - if (_cancelledRuns.ContainsKey(run.Name)) - { - CancelRemainingSequentialTasks(run); - break; - } - - // Rehydrate a shed payload. Sequential runs are exempt from shedding (see - // DispatchPendingTasksAsync), so this is normally a no-op — kept for a resumed run whose - // in-memory payload was dropped. A missing row means the durable state was lost: fail - // that step closed rather than run it blank, then carry on with the next. - if (_shedParameters && task.Parameters == null) - { - var rehydrated = await _store.GetTaskParametersAsync(run.Name, task.Id, jobCt); - if (rehydrated == null) - { - FailTaskTerminally(run, task, - "Parameters could not be rehydrated at dispatch — the Tasks-table row is missing. " + - "The task's payload was shed from memory and storage no longer has it."); - continue; - } - lock (_lock) { task.Parameters ??= rehydrated; } - } - - lock (_lock) { task.Status = "Running"; task.OwnedHere = true; } - // Durable "Running" marker — awaited before the invoke, same as the parallel path. - try - { - await _writer.MarkRunningAsync(run.Name, task, jobCt); - } - catch (MarkerNotPersistedException ex) - { - // The marker never landed, so storage still has this step Pending. Unlike the - // parallel path we cannot just give a slot back and let a re-drive retry the one - // task — we own the worker and the whole run — and running later steps out of order - // is not allowed. Put the step back to Pending and STOP the driver. The entry job - // then completes with pending work left and no driver active, so the re-drive - // re-enqueues the current step and a fresh driver resumes the run here. - lock (_lock) { task.Status = "Pending"; } - _logger.LogWarning(ex, - "[Scheduler] Sequential run {Run}: could not persist the Running marker for {Task} — " + - "leaving the run for the re-drive to resume", run.Name, task.Id); - break; - } + private string? Script(string? name) => + string.IsNullOrEmpty(name) ? null : _scripts.GetOrAdd(name, n => FindScript(n)); - try - { - // Run the step on the pinned worker. Only a PostExecution run needs the output - // captured and stored; otherwise the seam returns empty and nothing is stored. - var output = await RunSequentialStepAsync(run, task, taskPath, worker); - if (!string.IsNullOrEmpty(run.PostExecFunctionName) - && !string.IsNullOrEmpty(output) - && !_writer.TryQueueResult(run.Name, task.Id, output)) - { - await _store.StoreResultAsync(run.Name, task.Id, output); - } + internal virtual string? FindScript(string name) => _psRunner.FindScript(name); - lock (_lock) - { - task.Status = "Completed"; - task.CompletedUtc = DateTime.UtcNow; - task.Parameters = null!; - CheckRunCompletion(run); - } - PersistTaskAndRunAsync(run, task); - _logger.LogDebug("[Scheduler] Sequential task completed: {TaskId}", task.Id); - } - catch (OperationCanceledException) when (jobCt.IsCancellationRequested) - { - // App shutting down mid-step. Leave this step Running (its durable marker is written) - // for resume on next startup, and let the exception abort the driver. - _logger.LogInformation( - "[Scheduler] Sequential run {Run} interrupted by shutdown at {TaskId}", run.Name, task.Id); - throw; - } - catch (Exception ex) - { - lock (_lock) - { - task.Status = "Failed"; - task.LastError = ex.Message; - task.CompletedUtc = DateTime.UtcNow; - task.Parameters = null!; - CheckRunCompletion(run); - } - PersistTaskAndRunAsync(run, task); - _logger.LogError(ex, - "[Scheduler] Sequential task failed: {TaskId} — continuing with the next step", task.Id); - // best-effort: fall through to the next Pending step - } - } - } - catch (OperationCanceledException) when (jobCt.IsCancellationRequested) - { - // Shutdown (checkout or a step was cancelled). The pipeline may have been Stop()'d, so treat - // the worker as faulted on reclaim; rethrow so the JobManager marks the entry job Cancelled. - workerFaulted = true; - throw; - } - catch (Exception ex) - { - // An unexpected driver-level failure — not a per-step task error, which is handled inline. - workerFaulted = true; - _logger.LogError(ex, "[Scheduler] Sequential driver for {Run} failed", run.Name); - throw; - } - finally - { - ReclaimSequentialWorker(worker, workerFaulted); - _activeSequentialDrivers.TryRemove(run.Name, out _); - } - }; - } - - // ── sequential-driver seams ─────────────────────────────────────────────────────────────────────── - // The driver's loop logic (ordering, best-effort, cancellation, marker recovery) is exercised by unit - // tests through a subclass that overrides these three methods, so the tests need no PowerShell worker - // pool. Production runs the real pool: one worker checked out for the whole run, each step invoked on - // it, reclaimed once at the end. Keep the checkout/run/reclaim split — the whole point is one checkout - // and one reclaim around many step invocations. - - /// Check out the single worker a sequential run is pinned to. Virtual for tests. - internal virtual PowerShellWorker? CheckoutSequentialWorker(CancellationToken ct) - => _psRunner.CheckoutBackgroundWorker(ct); - - /// Return the pinned worker once the run is done. No-op for a null worker (checkout failed). - /// Virtual for tests. - internal virtual void ReclaimSequentialWorker(PowerShellWorker? worker, bool faulted) + /// Run a script on the pool, or on when pinned; returns its output when captured. + internal virtual async Task RunScriptAsync(string path, Dictionary parameters, bool captureOutput, + PowerShellWorker? worker = null) { - if (worker != null) _psRunner.ReclaimBackgroundWorker(worker, faulted: faulted); - } - - /// Run one sequential step on the pinned worker and return its captured output (empty when the - /// run has no PostExecution and so needs no result). Virtual for tests. InvokeAsync resets the runspace - /// afterwards, so successive steps stay isolated on the shared worker. - internal virtual async Task RunSequentialStepAsync( - OrchestratorRun run, OrchestratorTaskItem task, string taskPath, PowerShellWorker? worker) - { - var parameters = new Dictionary - { - { "TaskJson", JsonSerializer.Serialize(task.Parameters, s_jsonOptions) } - }; - if (!string.IsNullOrEmpty(run.PostExecFunctionName)) - return await _psRunner.ExecuteScriptWithOutput(taskPath, parameters, pinnedWorker: worker); - await _psRunner.ExecuteScript(taskPath, parameters, pinnedWorker: worker); + if (captureOutput) return await _psRunner.ExecuteScriptWithOutput(path, parameters, pinnedWorker: worker); + await _psRunner.ExecuteScript(path, parameters, pinnedWorker: worker); return string.Empty; - } - - /// - /// Mark every still-Pending or Running step of a cancelled sequential run Cancelled, in one lock, and - /// persist them. Called by the driver when it notices the run was cancelled between steps. - /// - private void CancelRemainingSequentialTasks(OrchestratorRun run) - { - List remaining; - lock (_lock) - { - remaining = run.Tasks.Where(t => t.Status is "Pending" or "Running").ToList(); - foreach (var t in remaining) - { - t.Status = "Cancelled"; - t.LastError = "Cancelled by user"; - t.CompletedUtc = DateTime.UtcNow; - t.Parameters = null!; - } - CheckRunCompletion(run); - } - foreach (var t in remaining) _writer.QueueTask(run.Name, t); - _writer.QueueRun(run); - } - - /// - /// Put one task back on the durable queue. Fire-and-forget because every caller is on a lock or a - /// timer callback, and a failure is recoverable: the task is still Pending in storage, so the next - /// re-drive finds it again. - /// - private void RequeueToTable(OrchestratorRun run, OrchestratorTaskItem task) - { - var priority = task.Priority ?? run.Priority; - _ = Task.Run(async () => - { - var key = DeferralKey(run.Name, task.Id); - try - { - await _queue.EnqueueAsync(run.Name, task.Id, priority, DateTime.UtcNow); - _requeueFailures.TryRemove(key, out _); - } - catch (Exception ex) - { - // A row storage rejects is rejected identically forever (illegal key remnant, - // oversized property), and the re-drive resets the deferral counter on every pass — - // without this cap the retry loop is infinite and the run it belongs to can never - // finalize. Consecutive failures only: a success above clears the count. - var failures = _requeueFailures.AddOrUpdate(key, 1, (_, c) => c + 1); - if (failures >= MaxRequeueFailures) - { - _requeueFailures.TryRemove(key, out _); - FailTaskTerminally(run, task, - $"Could not re-queue after {failures} consecutive attempts: {ex.Message}"); - return; - } - _logger.LogWarning(ex, - "[Scheduler] Could not re-queue {Task} in {Run} (attempt {Count}/{Max}) — the re-drive will retry", - task.Id, run.Name, failures, MaxRequeueFailures); - } - }); - } - - /// - /// Move a task that can never run to Failed and let its run finish without it. The terminal write - /// flows through the status writer like any other completion, so the remaining counter decrements - /// and finalize proceeds — the alternative is a Pending task retried for the process lifetime, - /// pinning the whole run graph with it. - /// - private void FailTaskTerminally(OrchestratorRun run, OrchestratorTaskItem task, string reason) - { - lock (_lock) - { - if (task.Status is "Completed" or "Failed" or "Cancelled") return; - task.Status = "Failed"; - task.LastError = reason; - task.CompletedUtc = DateTime.UtcNow; - task.Parameters = null!; - CheckRunCompletion(run); - } - PersistTaskAndRunAsync(run, task); - _logger.LogError("[Scheduler] Task {TaskId} in {Run} permanently failed: {Reason}", - task.Id, run.Name, reason); - } - - /// - /// Rebuild the work for a queued task. Registered on the JobManager at startup. - /// - /// Resolution order matters, for correctness before cost: - /// 1. the live run in , if present — this is the steady-state path and - /// costs ZERO storage reads; - /// 2. table storage, if the run is not in memory (crash recovery, or a run finalized and evicted - /// while its tasks were still queued). - /// - /// Step 1 is not an optimization, it is a requirement. decides - /// finalization by inspecting run.Tasks, and the task work mutates task.Status in - /// place. Handing out a freshly-deserialized task object would mutate a copy the live run graph - /// never sees, and no run would ever finalize. Object identity — not the field values — is the one - /// piece of state here that is genuinely not rehydratable. - /// - private async Task?> ResolveTaskWorkAsync(JobDescriptor descriptor, CancellationToken ct) - { - OrchestratorRun? run; - OrchestratorTaskItem? task; - - if (_activeRuns.TryGetValue(descriptor.RunName, out var liveRun)) - { - run = liveRun; - lock (_lock) - { - task = liveRun.Tasks.FirstOrDefault(t => t.Id == descriptor.TaskId); - } - } - else - { - // Run is no longer in memory — rehydrate it. This is the same read the crash-recovery path - // already performs (ResumeInterruptedRunsAsync), which resumed 421 runs in seconds. - run = await _store.GetRunAsync(descriptor.RunName); - - // A run absent from _activeRuns because it FINISHED must not be resurrected. Rehydrating it - // put a finalized run back into the live graph, where the next completion re-finalized it - // and re-dispatched its post-execution; and because terminal task writes are coalesced, the - // rehydrated task could still read Pending and be run a second time. Storage is authoritative - // here — FinalizeRunAsync flushes the run's terminal status before it removes the run from - // _activeRuns, so a terminal Status seen here is not a race. - if (run != null && run.Status is "Completed" or "CompletedWithErrors") - { - _logger.LogDebug( - "[Orchestrator] Descriptor {Run}/{Task} belongs to a run that already finished ({Status}) — dropping", - descriptor.RunName, descriptor.TaskId, run.Status); - return null; - } - - task = run?.Tasks.FirstOrDefault(t => t.Id == descriptor.TaskId); - if (run != null && task != null) - { - // Re-establish the live graph so sibling tasks share this instance and completion - // tracking works, exactly as it does on the resume path. - run = _activeRuns.GetOrAdd(run.Name, run); - task = run.Tasks.FirstOrDefault(t => t.Id == descriptor.TaskId) ?? task; - } - } - - if (run == null || task == null) - { - _logger.LogDebug("[Orchestrator] Stale descriptor {Run}/{Task} — no longer in storage", - descriptor.RunName, descriptor.TaskId); - return null; - } - - // Resolve the task script (a PowerShell function name) for this run. The steady-state path finds - // it in _taskScriptPaths, cached by DispatchPendingTasksAsync when the run was dispatched. A MISS - // is NOT a reason to drop the task. The pump is a BackgroundService that begins claiming persisted - // queue rows at host start, whereas the only writer of _taskScriptPaths does not run for a resumed - // run until ResumeInterruptedRunsAsync reaches it — and that waits on the worker pool first — and - // the rehydration branch above re-adds a run to _activeRuns without a cached path either. In both - // windows the run record still carries TaskScriptName, so rebuild the path from it exactly as the - // resume path does (ScriptRepository is fully loaded before the pump's first claim) and cache it, - // so sibling tasks of the same run cost nothing. Only an empty TaskScriptName with no - // naming-convention match is genuinely unrunnable. Returning null on the miss instead lets the - // JobManager mark the job Skipped and the pump delete its queue row, permanently dropping a task - // that is still Pending in the run — with no queue row left, nothing re-dispatches it. - if (!_taskScriptPaths.TryGetValue(run.Name, out var taskPath)) - { - taskPath = !string.IsNullOrEmpty(run.TaskScriptName) - ? _psRunner.FindScript(run.TaskScriptName) - : FindTaskScript(run.Name); - - if (string.IsNullOrEmpty(taskPath)) - { - _logger.LogWarning( - "[Orchestrator] No task script for run {Run} (TaskScriptName={Script}) — dropping {Task}", - run.Name, run.TaskScriptName, descriptor.TaskId); - return null; - } - - _taskScriptPaths[run.Name] = taskPath; - } - - // Already terminal (e.g. cancelled, or completed by a previous attempt while queued), or already - // executing. - // - // "Running" belongs here as well as the terminal states. A duplicate queue row claimed while its - // task is mid-flight used to pass this check and start a SECOND copy — seen with a five-minute - // Intune collection that was re-claimed four minutes in and ran twice. This does not block crash - // recovery: ResumeInterruptedRunsAsync flips interrupted tasks from Running back to Pending - // before re-dispatching them, so a task that genuinely needs re-running never reaches here as - // Running. - // - // Running only means "a worker HERE has it" when this process wrote it (OwnedHere). A Running read - // from storage is another process's pre-invoke marker, and it reaches the live graph whenever a run - // is rehydrated — by this resolver, or by the pump winning the _activeRuns race with recovery at - // startup, or from a container that outlived this one's recovery. Dropping those left the task - // Running forever: nothing re-drives Running, so the run never finalized. Reaching here means this - // process holds the task's queue claim (only the pump enqueues descriptors, only for rows it - // claimed), so the previous owner is gone: count the interrupted attempt exactly as recovery does, - // and run it. - var poisoned = false; - lock (_lock) - { - if (task.Status is "Completed" or "Failed" or "Cancelled") - return null; - if (task.Status == "Running") - { - if (task.OwnedHere) return null; - poisoned = ++task.AttemptCount >= 3; - if (!poisoned) task.Status = "Pending"; - } - } - if (poisoned) - { - FailTaskTerminally(run, task, $"Cancelled {task.AttemptCount} times by host interruption"); - return null; - } - - // Sequential runs are dispatched as a single entry row; that one claim drives the WHOLE run on one - // pinned worker (BuildSequentialRunWork ignores which step this descriptor named and works through - // every Pending step in order). The "Running" guard above already stops a duplicate row from - // starting a second driver once the entry step is marked Running, and the driver's own TryAdd closes - // the remaining pre-mark race. - if (run.Sequential) - return BuildSequentialRunWork(run, taskPath); - - return BuildTaskWork(run, task, taskPath); - } - - private Func BuildTaskWork(OrchestratorRun run, OrchestratorTaskItem task, string taskPath) - { - return - async (jobCt) => - { - // Rehydrate the Parameters payload shed while this task waited in the backlog. Done BEFORE - // any status write — MarkRunningAsync snapshots Parameters and every task-row write is - // Replace, so a null payload here would overwrite the stored one. One point read, only for a - // task actually being dispatched. The ??= keeps a value another dispatch attempt already set. - if (_shedParameters && task.Parameters == null) - { - // GetTaskParametersAsync returns null only when the Tasks row itself is GONE (a - // present-but-empty payload comes back as an empty dictionary). A missing row means this - // task's durable state was lost out from under a live run — the shed dropped the in-memory - // copy on the promise that storage still had it. Running now would invoke the task with NO - // parameters (its FunctionName and inputs both live in the payload), which for a real task - // is worse than not running it. Fail closed instead of executing blank. Before shedding - // the payload was resident, so a deleted row could not affect an in-flight dispatch; this - // guard restores that safety for the one case shedding introduced. - var rehydrated = await _store.GetTaskParametersAsync(run.Name, task.Id, jobCt); - if (rehydrated == null) - { - FailTaskTerminally(run, task, - "Parameters could not be rehydrated at dispatch — the Tasks-table row is missing. " + - "The task's payload was shed from memory and storage no longer has it."); - return; - } - lock (_lock) { task.Parameters ??= rehydrated; } - } - - // Check if run was cancelled while this job was queued - if (_cancelledRuns.ContainsKey(run.Name)) - { - lock (_lock) - { - if (task.Status != "Cancelled") - { - task.Status = "Cancelled"; - task.LastError = "Cancelled by user"; - task.CompletedUtc = DateTime.UtcNow; - task.Parameters = null!; - CheckRunCompletion(run); - } - } - PersistTaskAndRunAsync(run, task); - return; - } - - lock (_lock) - { - task.Status = "Running"; - task.OwnedHere = true; - } - // Pre-script "Running" write is awaited — the durability marker for crash recovery. Batched - // across concurrently-starting tasks by the status writer, but still durable before the invoke. - try - { - await _writer.MarkRunningAsync(run.Name, task, jobCt); - } - catch (MarkerNotPersistedException ex) - { - // DEFERRAL, not failure. The marker never landed, so storage still has this task - // Pending — running it now would break the poison-task bound that the marker exists - // to provide. Put the in-memory copy back to Pending so it agrees with storage, give - // the slot up, and retry. Nothing is lost: even if this process dies first, recovery - // re-queues it from the Pending row. - lock (_lock) { task.Status = "Pending"; } - DeferTask(run, task, ex); - return; - } - _deferrals.TryRemove(DeferralKey(run.Name, task.Id), out _); - - try - { - var parameters = new Dictionary - { - { "TaskJson", JsonSerializer.Serialize(task.Parameters, s_jsonOptions) } - }; - - if (!string.IsNullOrEmpty(run.PostExecFunctionName)) - { - // Capture output so C# can store results for PostExecution - var output = await _psRunner.ExecuteScriptWithOutput(taskPath, parameters); - - // Sizing the result payload. This string is the whole task result held in one - // piece, and BgPoolSize of them can be live at once — each at roughly two bytes - // per char since it is UTF-16. - // - // Measured on a 16-tenant instance across 92 real task results (mailbox and - // calendar permission batches, the widest fan-out CIPP has): median 5.9K chars, - // p95 42.5K, max 43.3K — 0.08 MB in memory for the largest. Eight of those - // concurrently is under 1 MB against a 2398 MB heap cap, so this site is not - // where the memory goes; the aggregate built from these at post-execution is - // (see DispatchPostExecution). Reported in KB because MB rounds every real - // result to 0.0 and hides exactly that conclusion. - _logger.LogDebug( - "[Scheduler] Task {TaskId} in {Run} returned {Chars} chars (~{ApproxKB:F1} KB in memory)", - task.Id, run.Name, output?.Length ?? 0, (output?.Length ?? 0) * 2 / 1024.0); - - if (!string.IsNullOrEmpty(output) && !_writer.TryQueueResult(run.Name, task.Id, output)) - { - // Large result, or result-batching off: keep the directly-awaited chunked path. - // Either way this awaits BEFORE the task is marked Completed below, so the result - // is durable before the run can finalize. Small results instead ride the status - // writer (TryQueueResult == true), which writes them before this task's terminal - // marker in the same flush — same guarantee, off the slot-held critical path. - await _store.StoreResultAsync(run.Name, task.Id, output); - } - } - else - { - await _psRunner.ExecuteScript(taskPath, parameters); - } - - lock (_lock) - { - task.Status = "Completed"; - task.CompletedUtc = DateTime.UtcNow; - // Release parameters — they are persisted in Table Storage and no longer - // needed in-memory. For 738-task runs this frees significant Gen2 memory. - task.Parameters = null!; - CheckRunCompletion(run); - } - // Post-script writes are fire-and-forget so the JobManager slot releases - // immediately and the dispatch loop can hand the worker to the next task. - // Crash recovery still works: the next startup re-reads task state from the - // table and re-runs anything not marked Completed (idempotent). - PersistTaskAndRunAsync(run, task); - - _logger.LogDebug("[Scheduler] Task completed: {TaskId}", task.Id); - } - catch (OperationCanceledException) when (jobCt.IsCancellationRequested) - { - // App shutting down — leave task as Running for resume on next startup - _logger.LogInformation("[Scheduler] Task {TaskId} interrupted by shutdown", task.Id); - throw; // Let JobManager mark as Cancelled - } - catch (Exception ex) - { - lock (_lock) - { - task.Status = "Failed"; - task.LastError = ex.Message; - task.CompletedUtc = DateTime.UtcNow; - // Release parameters on failure too — Table Storage has the full state - task.Parameters = null!; - CheckRunCompletion(run); - } - PersistTaskAndRunAsync(run, task); - - _logger.LogError(ex, "[Scheduler] Task failed: {TaskId}", task.Id); - throw; // Let JobManager also track the failure - } - }; - } - - /// - /// Fire-and-forget persistence of task + run state. Callers do not await this — it lets the - /// JobManager slot release immediately so the dispatch loop can hand the worker to the next - /// task. Errors are logged; on host crash, ResumeInterruptedRunsAsync re-derives state from - /// whatever made it to the table (writes are idempotent). - /// - private void PersistTaskAndRunAsync(OrchestratorRun run, OrchestratorTaskItem task) - { - // Non-blocking: the status writer coalesces these terminal task/run writes and flushes them in batches - // (guaranteed flushed before the run finalizes). Previously two individual fire-and-forget writes. - _writer.QueueTask(run.Name, task); - _writer.QueueRun(run); - } - - /// How many times a task has been deferred, and when it last was. The timestamp is what lets - /// the re-drive tell an exhausted task that has been sitting for minutes from one that deferred a - /// moment ago and is still legitimately retrying. - private sealed record DeferralState(int Count, DateTime LastUtc); - - /// Deferrals per task while storage is unable to accept the durable marker. In-memory and - /// intentionally so — it bounds retries within one process life, nothing more. - private readonly ConcurrentDictionary _deferrals = new(); - - /// Cap on in-process retries before a task is left for the next recovery pass to pick up. - private const int MaxDeferrals = 3; - - /// Consecutive finalize checks where storage still reported outstanding work for a run whose - /// in-memory tasks are all terminal. At the counter is recounted - /// from the task rows — a lost decrement otherwise defers finalize forever. - private readonly ConcurrentDictionary _finalizeDeferrals = new(); - private const int ReconcileAfterDeferrals = 3; - - /// Consecutive re-queue failures per task. Storage rejecting the same entity is not - /// transient — the write fails identically forever (see ) — so past - /// the task is failed terminally instead of re-driven again. - private readonly ConcurrentDictionary _requeueFailures = new(); - private const int MaxRequeueFailures = 5; - - /// - /// Re-queue a task whose durable marker could not be written, so it retries once storage recovers - /// instead of waiting for a restart. Bounded: after the task is simply - /// left Pending, which is already the durable state — recovery re-queues it on the next startup. - /// - private static string DeferralKey(string runName, string taskId) => $"{runName}{taskId}"; - - private void DeferTask(OrchestratorRun run, OrchestratorTaskItem task, Exception cause) - { - var state = _deferrals.AddOrUpdate(DeferralKey(run.Name, task.Id), - _ => new DeferralState(1, DateTime.UtcNow), - (_, s) => new DeferralState(s.Count + 1, DateTime.UtcNow)); - var count = state.Count; - - if (count > MaxDeferrals) - { - // One exhausted deferral cycle counts as one attempt on the task, mirroring startup - // recovery's 3-attempts rule. The re-drive resets the deferral counter when it re-queues, - // so without this the marker-fail → re-queue → marker-fail cycle repeats for the process - // lifetime and the run never finalizes. Exactly-once per cycle: only the call that - // crosses the cap increments (a duplicate queue row can push count past it again). - if (count == MaxDeferrals + 1 && ++task.AttemptCount >= 3) - { - FailTaskTerminally(run, task, - $"Durable Running marker rejected across {task.AttemptCount} deferral cycles: {cause.Message}"); - return; - } - - // Left Pending on purpose — storage already says Pending, so nothing is lost. It is no longer - // terminal though: RedrivePendingTasks picks it up once it has aged, so recovery is not - // gated on a restart the way it used to be. - _logger.LogError(cause, - "[Scheduler] Task {TaskId} in {Run} deferred {Count} times — left Pending for the re-drive", - task.Id, run.Name, count); - return; - } - - _logger.LogWarning( - "[Scheduler] Task {TaskId} in {Run} could not be marked Running (attempt {Count}/{Max}) — re-queued, slot released", - task.Id, run.Name, count, MaxDeferrals); - - // Back to the QUEUE, not to memory. The pump drops a claimed row once the JobManager is done with - // the job, so an in-memory re-queue here would leave the retry with no durable row behind it — and - // nothing to pick it up again if this instance went away. - RequeueToTable(run, task); - } - - /// - /// How long a task must have sat Pending-and-unowned before the re-drive claims it. Long enough that a - /// task mid-deferral (each attempt can take up to the barrier timeout) is not stolen out from under the - /// attempt already in progress. - /// - private static readonly TimeSpan RedriveAge = TimeSpan.FromMinutes(5); - - /// - /// Re-drive backoff bounds. The first verification for a run runs at the status-timer cadence; each time - /// it confirms nothing orphaned the interval doubles up to , so a run stuck for - /// hours costs a handful of index reads rather than one per minute. The cap bounds how long a genuinely - /// orphaned task can wait to be caught (worst case ~RedriveMax), which the watchdog trades for the cost. - /// - private static readonly TimeSpan RedriveMax = TimeSpan.FromMinutes(15); - - /// - /// Re-queue tasks that are Pending in memory but that nothing owns — no queued job, no running job. - /// - /// This is the safety net for the state a deferral leaves behind. A task whose durable "Running" marker - /// could not be written is rolled back to Pending and retried, but only - /// times; after that it used to sit Pending with nothing to pick it up, because the only other retry - /// path was startup recovery. A healthy instance never restarts, so in production that meant 658 tasks - /// pending and 0 running for three days, with "Run X already active, skipping" preventing a fresh run - /// from doing the work instead. - /// - /// Ownership is decided by rather than by a timestamp on the - /// task, so a task queued normally is never double-dispatched. The age gate only applies to tasks with - /// a deferral history — anything Pending and unowned with no deferral record was lost some other way - /// and there is nothing to wait for. - /// - private void RedrivePendingTasks(OrchestratorRun run) => _ = RedrivePendingTasksAsync(run); + } - /// - /// Re-queue tasks whose durable queue row went missing — a task Pending forever with nothing left to - /// dispatch it. Runs off the 60s status timer. - /// - /// "Orphaned" has to mean "storage has no row for it". It used to mean "the JobManager does not have - /// it queued or running", which was true in the world where dispatch enqueued every task into the - /// JobManager immediately. Under the pump that is simply what a BACKLOG looks like: the pump holds a - /// worker-pool-sized buffer and leaves the rest in storage, so a 124-task run against eight workers - /// has most of its tasks Pending and absent from the JobManager for minutes. - /// - /// The consequence was severe and silent. Every 60 seconds this re-queued the entire un-started - /// backlog — measured live at 92, then 60, 60, 52, 44, 36 tasks on consecutive ticks — and because - /// RequeueToTable stamps UtcNow into the RowKey, each pass created an ADDITIONAL row for the same - /// task instead of updating the existing one. Every copy was independently claimable, so tasks ran - /// once per copy: one Intune collection executed six times, from six rows exactly 60s apart. - /// - private async Task RedrivePendingTasksAsync(OrchestratorRun run) + /// Turn a claimed task into work. Registered on the JobManager; runs on the dispatched job. + private async Task?> ResolveTaskWorkAsync(JobDescriptor descriptor, CancellationToken ct) { - var now = DateTime.UtcNow; + if (descriptor.RunKey is not { } runKey) return null; + var header = await _store.GetRunAsync(runKey, ct); + if (header == null || header.IsFinished) return null; + + if (descriptor.Seq == WorkStore.AggregateSeq) return jobCt => RunPostExecutionAsync(header, descriptor, jobCt); - // Backoff gate. Once the verification below has confirmed a run has nothing orphaned, it need not run - // again for a while: orphaning is caused by specific rare events (a removed/expired queue row, a - // crash/migration), not something that spontaneously arises every 60s. Skipping here avoids the whole - // tick body — the candidates scan AND the storage read — for a run that verified clean, which for a - // large stuck backlog is nearly every run on nearly every tick. - if (_redriveBackoffEnabled && _redriveBackoff.TryGetValue(run.Name, out var st) && now < st.NextUtc) return; + var taskPath = Script(header.TaskScriptName); + if (taskPath == null) + { + await FinishAsync(header, descriptor.Seq, "Failed", $"Task script {header.TaskScriptName} not found"); + return null; + } + + return header.Sequential + ? BuildSequentialRunWork(header, descriptor, taskPath) + : jobCt => RunTaskAsync(header, descriptor.Seq, descriptor.TaskId, taskPath, jobCt); + } - List candidates; + private Task FinishAsync(RunHeader h, int seq, string status, string? error = null) => + _finisher.FinishAsync(h.RunKey, new WorkStore.Finish(seq, status, error, Owner)); - lock (_lock) + private async Task RunTaskAsync(RunHeader header, int seq, string taskId, string taskPath, CancellationToken ct) + { + if (header.CancelRequested) { - candidates = run.Tasks - .Where(t => t.Status == "Pending") - .Where(t => !_jobManager.IsQueuedOrRunning($"{run.Name}-{t.Id}")) - .Where(t => !_deferrals.TryGetValue(DeferralKey(run.Name, t.Id), out var s) - || now - s.LastUtc >= RedriveAge) - .ToList(); - - // Sequential run: one pinned driver runs every step, so the ONLY queue row that ever exists is - // the entry row that started the driver — the not-yet-reached steps deliberately have none. - // Applying the generic "Pending with no queue row = orphaned" rule to them would re-drive them - // all and spawn a second driver. The run is progressing whenever a driver is registered for it - // OR its entry job is still queued/running (that job's identity is one of this run's tasks) — - // re-drive nothing in either case. The driver registration closes the gap the entry-job check - // alone leaves open between steps (no task queued, none marked Running for an instant). Only when - // neither holds is the driver truly gone (never started, or died with the process): re-enqueue - // ONLY the current step (lowest Sequence still Pending) so a fresh driver resumes the run. - if (run.Sequential) - { - var driverActive = _activeSequentialDrivers.ContainsKey(run.Name) - || run.Tasks.Any(t => _jobManager.IsQueuedOrRunning($"{run.Name}-{t.Id}")); - if (driverActive) - { - candidates.Clear(); - } - else - { - var next = candidates.OrderBy(t => t.Sequence).FirstOrDefault(); - candidates = next != null ? [next] : []; - } - } + await FinishAsync(header, seq, "Cancelled", "Cancelled by user"); + return; } - if (candidates.Count == 0) + var parameters = await _store.GetPayloadAsync(header.RunKey, seq, ct); + if (parameters == null) { - // No candidates to verify (all Pending tasks are queued/running, or none are Pending). Don't grow - // the backoff — a run mid-drain legitimately produces no candidates and should stay responsive — - // just clear any prior backoff so the next real candidate is checked promptly. - _redriveBackoff.TryRemove(run.Name, out _); + await FinishAsync(header, seq, "Failed", "The task's payload row is missing"); return; } - // Storage decides — but the queue TABLE decides, not the index. Asking the index (the old - // GetQueuedTaskIdsAsync here) reports a task queued whenever its index row exists, and an index - // row can outlive the queue row it points at. Such a task is invisible to the pump yet looks - // "queued" to this check, so it is never re-driven and its run stalls indefinitely with the task - // Pending — this watchdog keeps ticking and finds nothing orphaned. GetDispatchableTaskIdsAsync - // verifies each candidate against the queue table (one point read apiece; the candidate set is - // small), returning only tasks the pump can actually still claim. Anything else is a ghost to - // re-enqueue. - HashSet dispatchable; try { - Interlocked.Increment(ref _redriveStorageReads); - dispatchable = await _queue.GetDispatchableTaskIdsAsync( - run.Name, candidates.Select(t => t.Id).ToList()); + var output = await RunScriptAsync(taskPath, TaskInvocation(parameters), header.HasPostExec); + if (!string.IsNullOrEmpty(output)) await _results.StoreResultAsync(header.RunKey, taskId, output); + } + catch (OperationCanceledException) when (ct.IsCancellationRequested) + { + await _store.ReleaseAsync(header.RunKey, seq, Owner, refundAttempt: true, CancellationToken.None); + throw; } catch (Exception ex) { - // Without this answer every candidate looks orphaned, which is the failure being fixed. - // Skip this tick; the timer comes back in 60s. - _logger.LogWarning(ex, "[Scheduler] Could not read queued tasks for {Run} — skipping re-drive", run.Name); - return; + await FinishAsync(header, seq, "Failed", ex.Message); + _logger.LogError(ex, "[Scheduler] Task failed: {TaskId}", taskId); + throw; } - var orphaned = candidates.Where(t => !dispatchable.Contains(t.Id)).ToList(); - if (orphaned.Count == 0) + await FinishAsync(header, seq, "Completed"); + _logger.LogDebug("[Scheduler] Task completed: {TaskId}", taskId); + } + + private async Task RunPostExecutionAsync(RunHeader header, JobDescriptor descriptor, CancellationToken ct) + { + var postExecScript = Script(_settings.Orchestrator.PostExecFunction); + if (postExecScript == null) { - // Verified clean: grow the interval (double, capped) so this run's next storage read is further - // out. A run stuck for hours thus costs O(log) reads, not one per minute. - if (_redriveBackoffEnabled) - { - var next = _redriveBackoff.TryGetValue(run.Name, out var cur) - ? TimeSpan.FromTicks(Math.Min(cur.Interval.Ticks * 2, RedriveMax.Ticks)) - : _redriveBase; - _redriveBackoff[run.Name] = (now + next, next); - } + _logger.LogError("[Orchestrator] PostExec function '{Func}' not found, cannot run PostExecution for {Name}", + _settings.Orchestrator.PostExecFunction, header.Name); + await FinishAsync(header, WorkStore.AggregateSeq, "Failed", "PostExec function not found"); return; } - // Found orphans — something is wrong with this run's queue rows, so snap back to close watch and - // re-drive them. - if (_redriveBackoffEnabled) - _redriveBackoff[run.Name] = (now + _redriveBase, _redriveBase); - foreach (var task in orphaned) + _logger.LogInformation("[Orchestrator] Dispatching PostExecution Push-{Function} for run {Name} {Memory}", + header.PostExecFunctionName, header.Name, BackgroundTaskLimiter.GetMemorySnapshot()); + var tempFile = Path.Combine(Path.GetTempPath(), $"craft-postexec-{Guid.NewGuid():N}.jsonl"); + try { - // Clear the exhausted counter, or DeferTask would abandon it again on its first attempt. - _deferrals.TryRemove(DeferralKey(run.Name, task.Id), out _); - RequeueToTable(run, task); - } - - _logger.LogWarning( - "[Scheduler] Re-drove {Count} orphaned Pending task(s) in {Run} — no runnable queue row and not queued or running", - orphaned.Count, run.Name); - } + var count = await _results.StreamResultsToJsonLinesAsync(header.RunKey, tempFile, ct); + _logger.LogInformation("[Orchestrator] PostExec results for {Name}: {Count} results, {SizeMB:F1}MB streamed to temp file", + header.Name, count, new FileInfo(tempFile).Length / (1024.0 * 1024.0)); - private void LogRunStatus(OrchestratorRun run) - { - // Nothing consumes this Info line at a higher level, and the flood of them is itself a measured - // cost, so do no work at all when Info is disabled. - if (!_logger.IsEnabled(LogLevel.Information)) return; + var parameters = new Dictionary + { + ["FunctionName"] = header.PostExecFunctionName!, + ["ResultsPath"] = tempFile, + }; + if (await _store.GetPostExecParametersAsync(header.RunKey, ct) is { Length: > 0 } postParameters) + parameters["ParametersJson"] = postParameters; - // One pass, not four Count(predicate) calls. Enumerable.Count over the List boxes an enumerator per - // call, and this runs on every run's 60s timer — four boxed enumerators × M runs per minute. - int completed = 0, failed = 0, running = 0, pending = 0; - lock (_lock) + await RunScriptAsync(postExecScript, parameters, captureOutput: false); + await OrchestratorBridge.DrainPendingAsync(); + _logger.LogInformation("[Orchestrator] PostExecution Push-{Function} completed for run {Name}", + header.PostExecFunctionName, header.Name); + await FinishAsync(header, WorkStore.AggregateSeq, "Completed"); + } + catch (OperationCanceledException) when (ct.IsCancellationRequested) + { + await _store.ReleaseAsync(header.RunKey, WorkStore.AggregateSeq, Owner, refundAttempt: true, CancellationToken.None); + throw; + } + catch (Exception ex) { - foreach (var t in run.Tasks) + _logger.LogError(ex, "[Orchestrator] PostExecution Push-{Function} failed for run {Name} (attempt {Attempt})", + header.PostExecFunctionName, header.Name, descriptor.Attempt); + if (descriptor.Attempt < Math.Max(1, _settings.Orchestrator.MaxRetries)) + await _store.ReleaseAsync(header.RunKey, WorkStore.AggregateSeq, Owner, refundAttempt: false, CancellationToken.None); + else { - switch (t.Status) - { - case "Completed": completed++; break; - case "Failed": failed++; break; - case "Running": running++; break; - case "Pending": pending++; break; - } + _logger.LogError("[Scheduler] PostExecution for {Name} failed {Count} times — giving up and cleaning up results", + header.Name, descriptor.Attempt); + await FinishAsync(header, WorkStore.AggregateSeq, "Failed", ex.Message); } + throw; + } + finally + { + try { if (File.Exists(tempFile)) File.Delete(tempFile); } + catch (Exception ex) { _logger.LogDebug(ex, "[Orchestrator] Failed to delete temp file {Path}", tempFile); } } - - // Skip the line — and the string format, the nine boxed args, and the memory snapshot it needs — - // when nothing has changed since the last tick. A run parked at "0 running / N pending" for hours - // re-emitted the identical line every 60s (M of them per minute at scale). Log on a real transition, - // plus a slow heartbeat so a long-lived run still shows it is alive. - var now = DateTime.UtcNow; - if (_lastStatusLog.TryGetValue(run.Name, out var prev) - && prev.C == completed && prev.F == failed && prev.R == running && prev.P == pending - && now - prev.LoggedUtc < StatusHeartbeat) - return; - _lastStatusLog[run.Name] = (completed, failed, running, pending, now); - - var elapsed = now - run.StartedUtc; - _logger.LogInformation( - "[Scheduler] Run {Name} T+{Elapsed:F1}min: {Completed}/{Total} done {Running} running {Pending} pending {Failed} failed jobs={Active}a/{Queued}q {Memory}", - run.Name, elapsed.TotalMinutes, completed, run.Tasks.Count, running, pending, failed, - _jobManager.ActiveCount, _jobManager.QueuedCount, - BackgroundTaskLimiter.GetMemorySnapshot()); } - private void CheckRunCompletion(OrchestratorRun run) - { - // Already locked by caller - if (run.Tasks.All(t => t.Status is "Completed" or "Failed" or "Cancelled")) - { - // Don't finalize if child runs (sub-orchestrators spawned by tasks) are still active - if (!AllChildRunsComplete(run.Name)) - { - _logger.LogInformation( - "[Scheduler] Run {Name} tasks complete but waiting for child runs to finish", - run.Name); - return; - } + // ── sequential runs ── - // Cannot await inside lock — schedule finalization - _ = Task.Run(async () => + /// + /// A sequential run: one driver checks out one worker and runs every step on it in payload order, claiming + /// each step itself under the run's driver lease so no other worker can take a step meanwhile. A failed + /// step is recorded and the next one runs. Cancelling the run stops it at the next step. + /// + private Func BuildSequentialRunWork(RunHeader header, JobDescriptor first, string taskPath) => + async jobCt => + { + PowerShellWorker? worker = null; + var faulted = false; + var step = new WorkStore.ClaimedTask(header.RunKey, first.Seq, first.TaskId, first.Attempt); + // The job this driver runs in carries each step's name in turn, and keeps a record of each step it finishes. + var jobId = WorkPump.JobId(step); + void StepDone(string status, string? error, WorkStore.ClaimedTask? next) => + _jobManager.AdvanceStep(jobId, status, error, next is { } n ? WorkPump.JobName(header.Name, n) : null); + RunHeader? known = null; + Dictionary? nextPayload = null; + try { - try + worker = CheckoutSequentialWorker(jobCt); + while (true) { - // Flush before the counter read below. The batched status writer decrements the counter - // only when a terminal write flushes, so an unflushed read sees the pre-decrement value - // and defers a finalize that is due — a single GET beats the drain+decrement every time, - // stalling every run until the 60s timer. Same flush FinalizeRunCoreAsync relies on, just - // ahead of the veto read; bounded by the barrier timeout, so it cannot hang. - await _writer.FlushAsync(); - - // The in-memory graph proposes, storage disposes. Finalizing is irreversible - it - // writes the aggregate and cleans the run up - so it must not run while storage still - // shows work outstanding, which is exactly the case when terminal writes have not yet - // flushed. A null count means the run predates the counter and cannot veto anything. - var remaining = await _store.GetRemainingAsync(run.Name); - if (remaining is > 0) + var current = known ?? await _store.GetRunAsync(header.RunKey, jobCt); + known = null; + if (current == null || current.IsFinished) break; + if (current.CancelRequested) + { + await FinishAsync(header, step.Seq, "Cancelled", "Cancelled by user"); + StepDone("Cancelled", "Cancelled by user", null); + await _store.CancelPendingAsync(header.RunKey, ct: jobCt); + break; + } + if (step.Seq == WorkStore.AggregateSeq) + { + var aggregate = new JobDescriptor(header.Name, step.TaskId, header.Priority) { RunKey = header.RunKey, Seq = step.Seq, Attempt = step.Attempt }; + await RunPostExecutionAsync(current, aggregate, jobCt); + break; + } + + var parameters = nextPayload ?? await _store.GetPayloadAsync(header.RunKey, step.Seq, jobCt); + nextPayload = null; + WorkStore.Finish finish; + try + { + if (parameters == null) throw new InvalidOperationException("The task's payload row is missing"); + var job = OperationContext.Current; + string output; + using (OperationContext.Set(new OperationContext.Invocation(WorkPump.JobName(header.Name, step)) + { + WorkerId = job?.WorkerId, + RunName = job?.RunName, + RunKey = job?.RunKey, + Priority = job?.Priority, + })) + { + output = await RunScriptAsync(taskPath, TaskInvocation(parameters), current.HasPostExec, worker); + } + if (current.HasPostExec && !string.IsNullOrEmpty(output)) + await _results.StoreResultAsync(header.RunKey, step.TaskId, output); + finish = new WorkStore.Finish(step.Seq, "Completed", Owner: Owner); + } + catch (OperationCanceledException) when (jobCt.IsCancellationRequested) + { + throw; + } + catch (Exception ex) { - // A counter that keeps contradicting a fully-terminal graph is drifted, not - // busy — a decrement that exhausted its retries is never re-applied, and - // without a recount this deferral repeats on every 60s tick for the process - // lifetime, pinning the run graph with it. Give in-flight terminal writes a - // few checks to land, then recount the partition the counter summarizes. - var misses = _finalizeDeferrals.AddOrUpdate(run.Name, 1, (_, c) => c + 1); - if (misses >= ReconcileAfterDeferrals) + if (current.StopOnFailure) { - _finalizeDeferrals.TryRemove(run.Name, out _); - if (await _store.ReconcileRemainingAsync(run.Name) is 0) - { - await FinalizeRunAsync(run); - return; - } + await FinishAsync(header, step.Seq, "Failed", ex.Message); + StepDone("Failed", ex.Message, null); + _logger.LogError(ex, "[Scheduler] Sequential task failed: {TaskId} — stopping the run", step.TaskId); + await _store.CancelPendingAsync(header.RunKey, WorkStore.StoppedReason(step.TaskId), ct: jobCt); + break; } - _logger.LogInformation( - "[Scheduler] Run {Name} complete in memory but storage shows {Remaining} outstanding - deferring finalize", - run.Name, remaining); - return; + _logger.LogError(ex, "[Scheduler] Sequential task failed: {TaskId} — continuing with the next step", step.TaskId); + finish = new WorkStore.Finish(step.Seq, "Failed", ex.Message, Owner); } - _finalizeDeferrals.TryRemove(run.Name, out _); - await FinalizeRunAsync(run); + // Finish this step and claim the next in one transaction; if that cannot be written, the batcher + // keeps retrying the finish and the next step is claimed on its own. + WorkStore.StepResult result; + try { result = await _store.FinishStepAsync(header.RunKey, finish, Owner, Lease, jobCt); } + catch (Exception ex) when (ex is not OperationCanceledException) + { + _logger.LogWarning(ex, "[Scheduler] Could not finish step {TaskId} of {Run} with its successor; recording it on its own", + step.TaskId, header.Name); + await _finisher.FinishAsync(header.RunKey, finish); + var claimed = await _store.ClaimSequentialAsync(header.RunKey, Owner, Lease, continuing: true, ct: jobCt); + StepDone(finish.Status, finish.Error, claimed); + if (claimed == null) break; + step = claimed; + continue; + } + StepDone(finish.Status, finish.Error, result.Next); + if (result.Next is not { } next) + { + if (result.Outcome?.Header is { CancelRequested: true, IsFinished: false }) + await _store.CancelPendingAsync(header.RunKey, ct: jobCt); + break; + } + step = next; + known = result.Outcome?.Header; + nextPayload = result.Payload; } - catch (Exception ex) { _logger.LogError(ex, "[Scheduler] FinalizeRun failed for {Name}", run.Name); } - }); - } - } + } + catch (OperationCanceledException) when (jobCt.IsCancellationRequested) + { + faulted = true; + await _store.ReleaseAsync(header.RunKey, step.Seq, Owner, refundAttempt: true, CancellationToken.None); + throw; + } + catch (Exception ex) + { + faulted = true; + _logger.LogError(ex, "[Scheduler] Sequential driver for {Run} failed", header.Name); + throw; + } + finally + { + ReclaimSequentialWorker(worker, faulted); + await _store.ReleaseDriverAsync(header.RunKey, Owner, CancellationToken.None); + } + }; - /// - /// Register a child run under a parent at ENQUEUE time — while the parent task's script is - /// still executing, so the parent cannot pass its completion check before the link exists. The - /// parent will not finalize while any child is pending dispatch, active, or recovering. Only - /// registers if the parent is still active; returns whether a pending gate was taken (the - /// bridge releases exactly what was taken via ). - /// - internal bool TryRegisterPendingChildRun(string parentRunName, string childRunName) + internal virtual PowerShellWorker? CheckoutSequentialWorker(CancellationToken ct) => _psRunner.CheckoutBackgroundWorker(ct); + + internal virtual void ReclaimSequentialWorker(PowerShellWorker? worker, bool faulted) { - // A run re-queued from inside its own context arrives with itself as parent — the - // recurring-run pattern, or a duplicate enqueue of an already-active run. Linking it would - // deadlock finalization: the run stays in _activeRuns until it finalizes, so - // AllChildRunsComplete would wait on the run itself forever. Observed live as runs stuck - // "Running" for days with every task terminal and Remaining=0. - if (parentRunName == childRunName) - return false; - - if (!_activeRuns.ContainsKey(parentRunName)) - return false; // Parent no longer active (e.g. queued from PostExec context) - - // Gate before link: the moment the link is visible to AllChildRunsComplete the pending - // mark must already hold, or a completion check could slip between the two writes. - _pendingChildRuns.AddOrUpdate(childRunName, 1, (_, n) => n + 1); - _childRuns.GetOrAdd(parentRunName, _ => new ConcurrentBag()).Add(childRunName); - _logger.LogInformation("[Orchestrator] Registered child run {Child} under parent {Parent}", - childRunName, parentRunName); - return true; + if (worker != null) _psRunner.ReclaimBackgroundWorker(worker, faulted: faulted); } + private static Dictionary TaskInvocation(Dictionary parameters) => + new() { ["TaskJson"] = JsonSerializer.Serialize(parameters, s_jsonOptions) }; + + // ── operator actions ── + /// - /// Lift the enqueue-time gate for ONE queued entry of this child. Called by the bridge after - /// the start attempt finishes, whatever the outcome: a started child is in _activeRuns by then - /// (which takes over blocking the parent), and one that failed to start must stop blocking — a - /// leaked gate would defer the parent's finalize for the process lifetime, re-checked every - /// 60s. Counted rather than boolean so two queued entries under the same child name cannot - /// release each other's gate. + /// Cancel the pending tasks of a run (by key) or of every unfinished run of a name; running ones finish. + /// Returns whether any run was found and how many tasks were cancelled. /// - internal void ReleasePendingChildRun(string childRunName) + public async Task<(bool found, int cancelledCount)> CancelRunAsync(string name) { - while (_pendingChildRuns.TryGetValue(childRunName, out var n)) - { - if (n <= 1) - { - if (_pendingChildRuns.TryRemove(new KeyValuePair(childRunName, n))) - return; - } - else if (_pendingChildRuns.TryUpdate(childRunName, n - 1, n)) - { - return; - } - } + var runs = await TargetRunsAsync(name); + var total = 0; + foreach (var header in runs) + { + await _store.RequestCancelAsync(header.RunKey); + await _store.CancelPendingAsync(header.RunKey); + // The pump cancels pages of a cancelled run too, so count what the run records, not what this call did. + var cancelled = ((await _store.GetRunAsync(header.RunKey))?.Cancelled ?? header.Cancelled) - header.Cancelled; + total += cancelled; + _logger.LogInformation("[Scheduler] Run {Name} cancelled: {Cancelled} pending tasks cancelled", header.Name, cancelled); + } + return (runs.Count > 0, total); } - private bool AllChildRunsComplete(string runName) + /// Cancel one pending task by its row (from a queue listing): a point read. False when it is not pending. + public async Task TryCancelQueuedTaskAsync(string runKey, int seq) { - if (!_childRuns.TryGetValue(runName, out var children)) - return true; - // A run is never its own blocker. TryRegisterPendingChildRun refuses self-links, but ones - // registered before that guard existed can still be sitting in the bag of a long-lived - // process. - return !children.Any(childName => - childName != runName && - (_pendingChildRuns.ContainsKey(childName) || - _activeRuns.ContainsKey(childName) || - _recoveringChildren.ContainsKey(childName))); + if (await _store.GetPendingAsync(runKey, seq) == null) return false; + var outcome = await _store.FinishAsync(runKey, [new WorkStore.Finish(seq, "Cancelled", "Cancelled by user")], 'P'); + return outcome?.Applied > 0; } - /// - /// Declare a run finished: write its terminal status, drop it from the live graph, and hand off to - /// post-execution. Runs at most once per run — see the claim below. - /// - private async Task FinalizeRunAsync(OrchestratorRun run) + /// Cancel one task that is still pending in storage, found by run name and task id. Reads the run's + /// pending range to find it (task ids are not keys), so prefer the run-key overload when the row is known. + public async Task TryCancelQueuedTaskAsync(string runName, string taskId) { - // Finalize once. This is not an idempotent method: it re-arms PostExecStatus and dispatches the - // post-execution again, so entering it twice runs the run's aggregation twice. Observed live - // before this guard: one 13-task run finalized 7 times and dispatched Push-StoreMailboxRules 7 - // times. Idempotent consumers hid it; Push-ScheduledTaskPostExecution did not, because it - // advances a recurring task by one interval per invocation. - // - // The claim lives here rather than at the call sites because there are four of them — normal - // completion, two resume paths, and cancellation — and cancellation can race a completion. - // DispatchPendingTasksAsync releases it when a run becomes live again, so a recurring run name - // can finalize on its next outing. - if (!_finalizingRuns.TryAdd(run.Name, true)) + foreach (var header in await TargetRunsAsync(runName)) { - _logger.LogDebug("[Scheduler] Run {Name} has already been finalized — ignoring duplicate", run.Name); - return; + var task = (await _store.GetTasksAsync(header.RunKey, 'P')).FirstOrDefault(t => t.TaskId == taskId); + if (task == null) continue; + var outcome = await _store.FinishAsync(header.RunKey, [new WorkStore.Finish(task.Seq, "Cancelled", "Cancelled by user")], 'P'); + if (outcome?.Applied > 0) return true; } + return false; + } - try - { - await FinalizeRunCoreAsync(run); - } - catch - { - // Still needs finalizing, so it must stay claimable — the 60s status timer retriggers it. - _finalizingRuns.TryRemove(run.Name, out _); - throw; - } + /// Move a run (or every unfinished run of a name) to another priority band. Applies to whole runs: + /// a run's tasks share one queue position. + public async Task ReprioritizeRunAsync(string runName, int priority) + { + var moved = false; + foreach (var header in await TargetRunsAsync(runName)) + moved |= await _store.SetPriorityAsync(header.RunKey, Math.Clamp(priority, 0, 99)); + return moved; } - private async Task FinalizeRunCoreAsync(OrchestratorRun run) + /// Cancel every pending task of every run — the whole backlog. Returns how many were cancelled. + public async Task ClearQueueAsync(CancellationToken ct = default) { - var failed = run.Tasks.Count(t => t.Status == "Failed"); - var cancelled = run.Tasks.Count(t => t.Status == "Cancelled"); - var completed = run.Tasks.Count(t => t.Status == "Completed"); - - run.Status = (failed > 0 || cancelled > 0) ? "CompletedWithErrors" : "Completed"; - run.CompletedUtc = DateTime.UtcNow; - var wallClock = run.CompletedUtc.Value - run.StartedUtc; - - // Set PostExecStatus before persisting, so crash between here and DispatchPostExecution is recoverable - if (!string.IsNullOrEmpty(run.PostExecFunctionName)) - run.PostExecStatus = "Pending"; - - // Flush-before-finalize: queue the run's final status, then await a full drain so every task's terminal - // state + this run status are durable BEFORE we declare the run done and dispatch PostExecution. - _writer.QueueRun(run); - await _writer.FlushAsync(); - - // Removed from _activeRuns first, so the status sweep stops ticking it before its per-run maps go. - _activeRuns.TryRemove(run.Name, out _); - _cancelledRuns.TryRemove(run.Name, out _); - _taskScriptPaths.TryRemove(run.Name, out _); - _lastStatusLog.TryRemove(run.Name, out _); - _redriveBackoff.TryRemove(run.Name, out _); - _finalizeDeferrals.TryRemove(run.Name, out _); - // Deferral and re-queue tracking is keyed per task and nothing else removes entries for tasks - // that ended without passing through their happy-path cleanup — without this sweep the residue - // of every run that ever deferred outlives the run. - foreach (var t in run.Tasks) + var total = 0; + var entries = new List(); + await foreach (var e in _store.ReadReadyAsync(200, ct)) entries.Add(e); + foreach (var e in entries) { - var key = DeferralKey(run.Name, t.Id); - _deferrals.TryRemove(key, out _); - _requeueFailures.TryRemove(key, out _); + await _store.RequestCancelAsync(e.RunKey, ct); + total += (await _store.CancelPendingAsync(e.RunKey, ct: ct)).Cancelled; } + _logger.LogWarning("[JobQueue] Durable queue cleared — {Count} queued task(s) cancelled", total); + return total; + } - var wallDisplay = wallClock.TotalSeconds < 60 - ? $"{wallClock.TotalSeconds:F1}s" - : $"{wallClock.TotalMinutes:F1}min"; + /// A claimed job reprioritized in the local buffer: it is already claimed, so nothing is stored. + public void PriorityChanged(JobDescriptor descriptor, int newPriority) { } - _logger.LogInformation( - "[Scheduler] Run {Name} finalized: {Status} ({Completed}/{Failed}/{Cancelled}/{Total}) wall={Wall} {Memory}", - run.Name, run.Status, completed, failed, cancelled, run.Tasks.Count, wallDisplay, - BackgroundTaskLimiter.GetMemorySnapshot()); + /// A claimed job cancelled before it started: record it, so its claim does not lapse and run it. + public void Cancelled(JobDescriptor descriptor) + { + if (descriptor.RunKey is not { } runKey) return; + _ = _finisher.FinishAsync(runKey, new WorkStore.Finish(descriptor.Seq, "Cancelled", "Cancelled by user", Owner)); + } - // If this was a child run, re-check parent's completion — it may have been - // waiting for this child to finish before it can finalize and run PostExecution - if (!string.IsNullOrEmpty(run.ParentRunName) && - _activeRuns.TryGetValue(run.ParentRunName, out var parentRun)) - { - lock (_lock) { CheckRunCompletion(parentRun); } - } + // ── diagnosis ── - // Cleanup child run tracking for this run - _childRuns.TryRemove(run.Name, out _); - - // Anything of this run's still queued is now moot; leaving rows behind has the pump claim work - // for a run that is already finished. That applies to EVERY finalized run — this used to sit in - // the else below, so a run WITH post-execution kept its queue rows from finalize until - // post-execution succeeded, and the pump spent that window re-claiming them. Observed live: one - // 13-task run finalized 7 times, dispatched its Push-* aggregation 7 times, and re-ran - // individual tasks up to 4 times each. The durable queue only ever carries TASKS — the - // post-execution job is enqueued in-memory on the JobManager and, after a crash, is re-derived - // from PostExecStatus — so dropping these rows here cannot cost the post-execution its retry. - _ = _queue.RemoveRunAsync(run.Name); - - // Dispatch PostExecution if configured - if (!string.IsNullOrEmpty(run.PostExecFunctionName)) - { - DispatchPostExecution(run); - } - else - { - // No PostExec — cleanup results table (if any stray entries exist) - _ = _store.CleanupRunAsync(run.Name); - } - } + public sealed record ClaimView(string TaskId, int Seq, string? Owner, DateTimeOffset? LeaseUntil, int Attempt, bool HeldHere); - private void DispatchPostExecution(OrchestratorRun run) + public sealed record RunView(string RunKey, string Name, string Status, string Phase, int Priority, DateTime StartedUtc, + DateTime? CompletedUtc, string Mode, int Total, int Done, int Failed, int Cancelled, bool Listed, string Pending, + IReadOnlyList Running, IReadOnlyList WaitingOnChildren, string? PostExecStatus, string? Driver, + IReadOnlyList Diagnosis); + + public sealed record Inspection(string Query, string ThisProcess, string? LockHolder, DateTimeOffset? LockLeaseUntil, + IReadOnlyList Runs); + + /// + /// Everything needed to see why a run is (or is not) moving, from storage, in a few bounded reads per run: + /// its counts and mode, whether the scheduler can see it, its claims and who holds them, the child runs it + /// waits for, its aggregation, and the instance lock, plus a plain-language diagnosis. Looks a run up by key, + /// or every unfinished run of a name, or else the latest finished one. + /// + public async Task InspectRunAsync(string nameOrKey, CancellationToken ct = default) { - var postExecFunc = _settings.Orchestrator.PostExecFunction; - var postExecScript = !string.IsNullOrEmpty(postExecFunc) ? _psRunner.FindScript(postExecFunc) : null; - if (postExecScript == null) - { - _logger.LogError("[Orchestrator] PostExec function '{Func}' not found, cannot run PostExecution for {Name}", - postExecFunc, run.Name); - return; + var runs = await TargetRunsAsync(nameOrKey); + if (runs.Count == 0 && await _store.ResolveRunAsync(TableKeys.Sanitize(nameOrKey), ct) is { } byKey) runs = [byKey]; + if (runs.Count == 0 && await _store.GetRunByNameAsync(TableKeys.Sanitize(nameOrKey), ct) is { } latest) runs = [latest]; + var lockRow = await _store.GetInstanceLockAsync(ct); + var lockLive = lockRow != null && lockRow.LeaseUntil > DateTimeOffset.UtcNow; + + var views = new List(); + foreach (var h in runs) + { + var pending = await _store.GetTasksAsync(h.RunKey, 'P', 1001, ct); + var running = (await _store.GetTasksAsync(h.RunKey, 'R', 500, ct)) + .Select(t => new ClaimView(t.TaskId, t.Seq, t.Owner, t.LeaseUntil, t.Attempt, t.Owner == Owner)).ToList(); + var children = await _store.GetChildWaitsAsync(h.RunKey, ct: ct); + var listed = h.IsFinished || await _store.IsListedAsync(h, ct); + var tasksPending = pending.Count(t => t.Seq != WorkStore.AggregateSeq); + var pendingText = tasksPending > 1000 ? "1000+" : tasksPending.ToString(System.Globalization.CultureInfo.InvariantCulture); + var mode = h.Sequential ? (h.StopOnFailure ? "sequential, stop on failure" : "sequential") + : h.MaxConcurrency > 0 ? $"at most {h.MaxConcurrency} at once" : "fan-out"; + + var why = new List(); + if (h.IsFinished) + { + why.Add($"Finished {h.Status} at {h.CompletedUtc:O}."); + } + else + { + if (!listed) why.Add("Not on the Ready list, so the scheduler cannot see it. RepairIndexes (or a restart) relists it."); + if (!lockLive) why.Add("Nobody holds the instance lock: no process is claiming work."); + else if (lockRow!.Owner != Owner) why.Add($"The instance lock is held by {lockRow.Owner}, not this process; that process is the one claiming."); + var stale = running.Where(r => !r.HeldHere).ToList(); + var here = running.Count - stale.Count; + if (here > 0) why.Add($"{here} task(s) running in this process."); + if (stale.Count > 0) + why.Add(lockLive && lockRow!.Owner == Owner + ? $"{stale.Count} claim(s) held by a process that no longer works the queue ({string.Join(", ", stale.Select(s => s.Owner).Distinct())}); taken back the next time the run is read." + : $"{stale.Count} claim(s) held by {string.Join(", ", stale.Select(s => s.Owner).Distinct())}."); + if (children.Count > 0) why.Add($"Waiting for {children.Count} child run(s): {string.Join(", ", children.Select(c => c.Split('|')[0]))}."); + if (h.MaxConcurrency > 0 && !h.Sequential && tasksPending > 0 && here >= h.MaxConcurrency) + why.Add($"At its concurrency limit of {h.MaxConcurrency}; the next task starts when one finishes."); + else if (tasksPending > 0 && running.Count == 0) + why.Add($"{pendingText} task(s) pending, waiting for a worker in band P{h.Priority} (lower bands, and older runs in this band, go first)."); + if (h.Phase == RunPhase.Aggregate) + why.Add($"Every task is done; its aggregation (Push-{h.PostExecFunctionName}) is {(running.Any(r => r.Seq == WorkStore.AggregateSeq) ? "running" : "waiting to be claimed")}."); + if (h.CancelRequested) why.Add("Cancel requested: pending tasks are cancelled and running ones finish."); + } + + views.Add(new RunView(h.RunKey, h.Name, h.Status, h.Phase.ToString(), h.Priority, h.StartedUtc, h.CompletedUtc, mode, + h.Total, h.Done, h.Failed, h.Cancelled, listed, pendingText, running, children, h.PostExecStatus, + h.DriverOwner == null ? null : $"{h.DriverOwner} until {h.DriverLease:O}", why)); } + return new Inspection(nameOrKey, Owner, lockRow?.Owner, lockRow?.LeaseUntil, views); + } - _logger.LogInformation( - "[Orchestrator] Dispatching PostExecution Push-{Function} for run {Name} {Memory}", - run.PostExecFunctionName, run.Name, BackgroundTaskLimiter.GetMemorySnapshot()); - - _jobManager.Enqueue( - name: $"{run.Name}-PostExec", - priority: run.Priority, - // Post-exec commonly starts follow-up runs (baseline → cache refresh); they should land - // at this run's priority, not the enqueue default. - inheritPriority: run.Priority, - runName: run.Name, - work: async (jobCt) => - { - // Mark PostExec as Running, and count the attempt. Incremented BEFORE the work so a - // crash mid-post-execution still burns an attempt — otherwise a post-execution that - // kills the host would be retried forever, which is precisely what the bound is for. - run.PostExecStatus = "Running"; - run.PostExecAttemptCount++; - await _store.UpsertRunAsync(run); - - // Stream results to a temp file rather than building the aggregate in memory. For large - // runs (738+ tasks) that aggregate is 50-150 MB, and StreamResultsToJsonLinesAsync holds - // one chunk at a time rather than the whole entity set (it used to buffer every result - // row into a dictionary before writing a byte, so "streaming" still peaked at the full - // payload in UTF-16 — roughly 2x the stored size — before this ran). - // - // The path — not the content — is what goes to PowerShell. This used to read the file - // back with File.ReadAllTextAsync and pass it as a ResultsJson string, which put the - // whole aggregate on the Large Object Heap and had PowerShell copy it a second time on - // ConvertFrom-Json. Handing over the path instead lets Invoke-CraftPostExecution walk - // the file one line at a time, so neither copy is ever made. The file is JSON Lines for - // exactly that reason; see StreamResultsToJsonLinesAsync. - // - // Consequence for the file's lifetime: it must now survive until PowerShell has read - // it, so it is deleted in the finally below rather than immediately after streaming. - var tempFile = Path.Combine(Path.GetTempPath(), $"craft-postexec-{Guid.NewGuid():N}.jsonl"); - try - { - var resultCount = await _store.StreamResultsToJsonLinesAsync(run.Name, tempFile, jobCt); - var fileSize = new FileInfo(tempFile).Length; - var fileSizeMB = fileSize / (1024.0 * 1024.0); - _logger.LogInformation( - "[Orchestrator] PostExec results for {Name}: {Count} results, {SizeMB:F1}MB streamed to temp file {Memory}", - run.Name, resultCount, fileSizeMB, BackgroundTaskLimiter.GetMemorySnapshot()); - - var parameters = new Dictionary - { - { "FunctionName", run.PostExecFunctionName! }, - { "ResultsPath", tempFile } - }; - if (!string.IsNullOrEmpty(run.PostExecParametersJson)) - parameters["ParametersJson"] = run.PostExecParametersJson; + /// Rebuild the Ready and Finished indexes from the active-run list now (the pump does this at startup). + public Task RepairIndexesAsync(CancellationToken ct = default) => + _store.RepairIndexesAsync(TimeSpan.FromMinutes(10), ct); - await _psRunner.ExecuteScript(postExecScript, parameters); + // ── lookups ── - // PostExecution functions may call Start-CIPPOrchestrator (Phase 2) - await OrchestratorBridge.DrainPendingAsync(); + public string? GetRunReference(string runName) => Task.Run(async () => + (await _store.ResolveRunAsync(runName) ?? await _store.GetRunByNameAsync(runName))?.Reference).GetAwaiter().GetResult(); - // Mark PostExec as Completed - run.PostExecStatus = "Completed"; - await _store.UpsertRunAsync(run); + public string? FindRunByReference(string reference) => Task.Run(async () => + { + await foreach (var e in _store.ReadReadyAsync(200)) + if (string.Equals(e.Reference, reference, StringComparison.OrdinalIgnoreCase)) return e.Name; + return null; + }).GetAwaiter().GetResult(); - _logger.LogInformation("[Orchestrator] PostExecution Push-{Function} completed for run {Name}", - run.PostExecFunctionName, run.Name); + // ── startup, retention, status lines ── - // Cleanup after successful PostExec - await _store.CleanupRunAsync(run.Name); - await _queue.RemoveRunAsync(run.Name, jobCt); - } - catch (Exception ex) + /// + /// Startup. Storage is the state, so there is nothing to recover: create the tables, drop the previous + /// design's tables once, and sweep expired runs. + /// + public async Task ResumeInterruptedRunsAsync(CancellationToken ct) + { + await _store.InitializeAsync(ct); + await _store.DropLegacyTablesAsync(ct); + try { await RunRetentionSweepAsync(ct); } + catch (Exception ex) { _logger.LogWarning(ex, "[Scheduler] Startup retention sweep failed"); } + } + + public async Task RunRetentionSweepAsync(CancellationToken ct) + { + var removed = await _store.SweepFinishedAsync(TimeSpan.FromHours(Math.Max(1, _settings.Orchestrator.RetentionHours)), ct); + if (removed > 0) _logger.LogInformation("[OrchestratorStore] Retention sweep removed {Count} finished run(s)", removed); + return removed; + } + + public async Task RunRetentionLoopAsync(CancellationToken ct) + { + var hours = _settings.Orchestrator.CleanupIntervalHours; + if (hours <= 0) return; + using var timer = new PeriodicTimer(TimeSpan.FromHours(hours)); + try + { + while (await timer.WaitForNextTickAsync(ct)) + { + try { await RunRetentionSweepAsync(ct); } + catch (Exception ex) when (ex is not OperationCanceledException) { - // Mark PostExec as Failed. ResumeInterruptedRunsAsync picks "Failed" back up on the - // next startup, up to MaxPostExecAttempts; the run's Results rows stay in storage - // until it either succeeds or is abandoned, because they are the retry's input. - run.PostExecStatus = "Failed"; - try { await _store.UpsertRunAsync(run); } catch { /* best effort */ } - _logger.LogError(ex, "[Orchestrator] PostExecution Push-{Function} failed for run {Name}", - run.PostExecFunctionName, run.Name); - throw; + _logger.LogWarning(ex, "[Scheduler] Retention sweep failed; next attempt in {Hours}h", hours); } - finally + } + } + catch (OperationCanceledException) { } + } + + /// + /// One status line per active run on a change, or every ten minutes when unchanged: + /// T+{min}min: {done}/{total} done {running} running {pending} pending {failed} failed. Health checks parse it to + /// spot runs that sit with work pending and nothing running. + /// + public async Task RunStatusSweepLoopAsync(CancellationToken ct) + { + using var timer = new PeriodicTimer(TimeSpan.FromSeconds(Math.Max(1, _settings.Orchestrator.StatusTimerIntervalSeconds))); + try + { + while (await timer.WaitForNextTickAsync(ct)) + { + try { await LogRunStatusAsync(ct); } + catch (Exception ex) when (ex is not OperationCanceledException) { - // The only delete. PowerShell reads the file during ExecuteScript above, so it - // cannot be freed any earlier — and it must still be freed when that throws. - try { if (File.Exists(tempFile)) File.Delete(tempFile); } - catch (Exception ex) { _logger.LogDebug(ex, "[Orchestrator] Failed to delete temp file {Path}", tempFile); } + _logger.LogWarning(ex, "[Scheduler] Run status sweep failed"); } } - ); + } + catch (OperationCanceledException) { } + } + + internal async Task LogRunStatusAsync(CancellationToken ct) + { + if (!_logger.IsEnabled(LogLevel.Information)) return; + // Running jobs are counted per run name, so runs sharing a name take them oldest first. + var running = _jobManager.GetJobs(status: "Running").Where(j => j.RunName != null) + .GroupBy(j => j.RunName!).ToDictionary(g => g.Key, g => g.Count()); + var now = DateTime.UtcNow; + await foreach (var e in _store.ReadReadyAsync(200, ct)) + { + var r = Math.Min(Math.Max(0, e.Total - e.Done), running.GetValueOrDefault(e.Name)); + running[e.Name] = running.GetValueOrDefault(e.Name) - r; + var p = Math.Max(0, e.Total - e.Done - r); + if (_lastStatusLog.TryGetValue(e.RunKey, out var prev) && prev.C == e.Done && prev.R == r && prev.P == p + && now - prev.LoggedUtc < StatusHeartbeat) continue; + _lastStatusLog[e.RunKey] = (e.Done, r, p, now); + _logger.LogInformation( + "[Scheduler] Run {Name} T+{Elapsed:F1}min: {Completed}/{Total} done {Running} running {Pending} pending {Failed} failed jobs={Active}a/{Queued}q {Memory}", + e.Name, (now - e.StartedUtc).TotalMinutes, e.Done - e.Failed - e.Cancelled, e.Total, r, p, e.Failed, + _jobManager.ActiveCount, _jobManager.QueuedCount, BackgroundTaskLimiter.GetMemorySnapshot()); + } } + // ── parsing batches into tasks ── + private List ParseTasksFromJson(string json, string runName) { var tasks = new List(); @@ -2457,113 +1096,4 @@ internal static void AddTaskFromElement(List tasks, HashSe Status = "Pending" }); } - - private string? FindTaskScript(string runName) - { - // Convention: strip "Start-" prefix → "Invoke-{rest}Task" - var baseName = runName.StartsWith("Start-", StringComparison.OrdinalIgnoreCase) - ? runName[6..] - : runName; - return _psRunner.FindScript($"Invoke-{baseName}Task"); - } - - /// - /// Empty the durable job queue (maintenance/reset). Delegates to - /// — see its remarks: in-flight work is unaffected and Pending tasks may be re-driven, so pair this - /// with when the intent is to STOP work rather than clear a wedged queue. - /// Returns the number of queue rows removed. - /// - public Task ClearQueueAsync(CancellationToken ct = default) => _queue.ClearAllAsync(ct); - - /// - /// Cancel a running orchestrator run. Pending tasks are marked Cancelled immediately. - /// Already-running tasks are allowed to finish (no force-kill). - /// Queued jobs in the JobManager will be skipped when they are dequeued. - /// - public async Task<(bool found, int cancelledCount)> CancelRunAsync(string name) - { - var run = await _store.GetRunAsync(name); - if (run == null) return (false, 0); - - // Mark this run as cancelled so dispatched-but-not-yet-started tasks skip execution - _cancelledRuns.TryAdd(name, true); - - int cancelled; - var tasksToUpdate = new List(); - lock (_lock) - { - var pendingTasks = run.Tasks.Where(t => t.Status == "Pending").ToList(); - cancelled = pendingTasks.Count; - foreach (var task in pendingTasks) - { - task.Status = "Cancelled"; - task.LastError = "Cancelled by user"; - task.CompletedUtc = DateTime.UtcNow; - tasksToUpdate.Add(task); - } - } - - // Persist cancelled task states through the status-guarded cancel, not a plain upsert. Two - // things ride on that. Cancelled is terminal, so the write must decrement the run counter — - // skip it and CheckRunCompletion's finalize veto reads "{cancelled count} still outstanding" - // forever once the running tasks drain. And the write must only land while storage still shows - // Pending: dispatch can move a task to Running between our read above and this write, and - // clobbering that would have the task's real completion decrement a second time. A task that - // moved on is un-cancelled in our copy and left to finish. - foreach (var t in tasksToUpdate) - { - var result = await _store.CancelPendingTaskAsync(run.Name, t); - if (result.Cancelled) continue; - - cancelled--; - lock (_lock) - { - t.Status = result.CurrentStatus ?? "Running"; - t.LastError = null; - t.CompletedUtc = null; - } - } - - // Check if the run is now fully done (Running tasks will finalize themselves) - var remaining = run.Tasks.Count(t => t.Status is "Running"); - if (remaining == 0) - { - await FinalizeRunAsync(run); - } - else - { - await _store.UpsertRunAsync(run); - } - - _logger.LogInformation("[Scheduler] Run {Name} cancelled: {Cancelled} pending tasks cancelled, {Running} still running", - name, cancelled, run.Tasks.Count(t => t.Status == "Running")); - - return (true, cancelled); - } - - /// - /// Check whether a run has been cancelled (used by dispatch to skip queued tasks). - /// - public bool IsRunCancelled(string runName) => _cancelledRuns.ContainsKey(runName); - - /// - /// Get the current state of a run (or null if it doesn't exist). - /// Used by the API status endpoint. - /// - public async Task GetRunStatusAsync(string name) - { - await _store.InitializeAsync(); - await _queue.InitializeAsync(); - return await _store.GetRunAsync(name); - } - - /// - /// List all known run names from table storage. - /// - public async Task> ListRunsAsync() - { - await _store.InitializeAsync(); - await _queue.InitializeAsync(); - return await _store.ListRunsAsync(); - } } diff --git a/Services/Orchestration/OrchestratorStatusWriter.cs b/Services/Orchestration/OrchestratorStatusWriter.cs deleted file mode 100644 index 1208e8b..0000000 --- a/Services/Orchestration/OrchestratorStatusWriter.cs +++ /dev/null @@ -1,388 +0,0 @@ -using System.Text.Json; -using Craft.Configuration; -using Craft.Storage; - -namespace Craft.Orchestration; - -/// -/// Coalescing, batched, durable writer for orchestrator TASK and RUN status transitions. It removes the -/// per-task Azure Table write from the fan-out critical path (that write was the throughput ceiling — see -/// docs/orch-analysis.md) by coalescing many transitions and flushing them in ≤100-entity, byte-budgeted -/// transactions. -/// -/// Durability is preserved: -/// - the pre-invoke "Running" marker is written under a synchronous barrier (batched across concurrently -/// starting tasks, but still durable-BEFORE-invoke, so AttemptCount/MaxRetries still bound poison tasks); -/// - guarantees all pending terminal states are persisted before a run finalizes; -/// - a final drain runs on shutdown. -/// -/// RESULTS are deliberately NOT handled here — OrchestratorTableStore.StoreResultAsync keeps its -/// property-chunking / multi-row large-payload path completely untouched. -/// -public sealed class OrchestratorStatusWriter : IDisposable -{ - private readonly OrchestratorTableStore _store; - private readonly ILogger _logger; - private readonly bool _enabled; - private readonly bool _durableBarrier; - private readonly bool _batchResults; - private readonly int _flushIntervalMs; - - /// Results at or below this many chars fit a single Azure Table property and can be coalesced - /// here; larger ones need the chunked multi-row path and are written directly by the caller. Matches - /// OrchestratorTableStore's single-property fast-path bound. - private const int SmallResultMaxChars = 30_000; - private readonly TimeSpan _barrierTimeout; - private readonly TimeSpan _flushTimeout; - private readonly int _flushConcurrency; - - // Match the read path (OrchestratorTableStore serializes/deserializes ParametersJson camelCase). - private static readonly JsonSerializerOptions s_json = new() { PropertyNamingPolicy = JsonNamingPolicy.CamelCase }; - - private readonly object _lock = new(); - private Dictionary _pendingTasks = new(); // key: runName  taskId (last-wins coalesce) - private Dictionary _pendingRuns = new(); // key: runName - private Dictionary _pendingResults = new(); // key: run + task - private TaskCompletionSource _barrier = new(TaskCreationOptions.RunContinuationsAsynchronously); - private readonly SemaphoreSlim _signal = new(0, int.MaxValue); - private readonly CancellationTokenSource _cts = new(); - private readonly Task _drainLoop; - - /// Set first in so a status enqueue racing shutdown drops its wake - /// instead of throwing ObjectDisposedException into a job that already succeeded. - private volatile bool _disposed; - - public bool Enabled => _enabled; - - public OrchestratorStatusWriter(OrchestratorTableStore store, ILogger logger, CraftSettings settings) - { - _store = store; - _logger = logger; - _enabled = settings.Orchestrator.BatchStatusWrites; - _durableBarrier = settings.Orchestrator.DurableRunningBarrier; - _batchResults = settings.Orchestrator.BatchResultWrites; - _flushIntervalMs = Math.Max(5, settings.Orchestrator.StatusFlushIntervalMs); - _barrierTimeout = TimeSpan.FromSeconds(Math.Max(1, settings.Orchestrator.RunningBarrierTimeoutSeconds)); - _flushTimeout = TimeSpan.FromSeconds(Math.Max(1, settings.Orchestrator.StatusFlushTimeoutSeconds)); - _flushConcurrency = Math.Max(1, settings.Orchestrator.StatusFlushConcurrency); - _drainLoop = _enabled ? Task.Run(DrainLoopAsync) : Task.CompletedTask; - _logger.LogInformation( - "[Orchestrator] StatusWriter: enabled={E} durableBarrier={B} flushMs={F} barrierTimeout={BT}s flushTimeout={FT}s concurrency={C}", - _enabled, _durableBarrier, _flushIntervalMs, _barrierTimeout.TotalSeconds, _flushTimeout.TotalSeconds, - _flushConcurrency); - } - - private static string Key(string run, string task) => run + "" + task; - private static TaskStatusWrite Snap(string run, OrchestratorTaskItem t) => new( - run, t.Id, t.Status, JsonSerializer.Serialize(t.Parameters, s_json), t.AttemptCount, t.LastError, - t.CompletedUtc, t.Priority, t.Sequence); - - /// Persist the pre-invoke "Running" marker durably before the task runs. Under the barrier it is - /// batched with other concurrently-starting tasks (N tasks → ~1 transaction) yet still lands before the - /// invoke. Disabled → the original per-task awaited write. - /// - /// The marker did not persist within RunningBarrierTimeoutSeconds. The caller must treat this - /// as a DEFERRAL, not a failure: nothing was written, so storage still has the task Pending and it - /// is safe — and necessary — to retry it. - /// - public async Task MarkRunningAsync(string runName, OrchestratorTaskItem task, CancellationToken ct = default) - { - if (!_enabled) { await _store.UpsertTaskAsync(runName, task); return; } - if (!_durableBarrier) { QueueTask(runName, task); return; } // eventual mode (weaker poison guarantee) - - Task barrier; - lock (_lock) - { - _pendingTasks[Key(runName, task.Id)] = Snap(runName, task); - barrier = _barrier.Task; - } - Signal(); - - // Bounded. This wait sits between dispatch and worker checkout while holding a JobManager slot, - // so waiting forever converts a slow flush into a whole-host outage — observed in production as - // 8/8 slots held, every BG worker idle, 1,919 jobs queued and nothing moving until a restart. - try - { - if (await CompletesWithinAsync(barrier, _barrierTimeout, ct)) return; - } - catch (Exception ex) when (ex is not OperationCanceledException) - { - // The flush carrying this marker failed. It has been requeued, but THIS task must not start: - // the marker is not in storage, so its retry bound would not hold. - throw new MarkerNotPersistedException( - $"Durable 'Running' marker for {runName}/{task.Id} did not persist: {ex.Message}", ex); - } - - throw new MarkerNotPersistedException( - $"Durable 'Running' marker for {runName}/{task.Id} did not persist within " + - $"{_barrierTimeout.TotalSeconds:F0}s. The task was not started and remains Pending."); - } - - /// - /// Await with a ceiling. True if it finished (and any exception it carried - /// is rethrown), false if the ceiling was hit first. - /// - private static async Task CompletesWithinAsync(Task work, TimeSpan limit, CancellationToken ct) - { - if (work.IsCompleted) { await work; return true; } - - using var cts = CancellationTokenSource.CreateLinkedTokenSource(ct); - var timer = Task.Delay(limit, cts.Token); - var winner = await Task.WhenAny(work, timer); - cts.Cancel(); // stop the timer whichever way this went, so it cannot outlive the call - - if (winner != work) - { - ct.ThrowIfCancellationRequested(); - return false; - } - - await work; // observe a flush failure as an exception rather than a silent success - return true; - } - - /// Queue a task's (usually terminal) status — non-blocking, coalesced, flushed by the drain loop. - /// Disabled → the original fire-and-forget per-task write. - public void QueueTask(string runName, OrchestratorTaskItem task) - { - if (!_enabled) { _ = _store.UpsertTaskAsync(runName, task); return; } - lock (_lock) { _pendingTasks[Key(runName, task.Id)] = Snap(runName, task); } - Signal(); - } - - /// Queue a run's status — non-blocking, coalesced, flushed by the drain loop. - public void QueueRun(OrchestratorRun run) - { - if (!_enabled) { _ = _store.UpsertRunAsync(run); return; } - lock (_lock) { _pendingRuns[run.Name] = run; } - Signal(); - } - - /// - /// Try to coalesce a task result. Returns true if it was queued (written before this task's terminal - /// marker in the next flush, so it is durable before the task is counted done); false if the caller - /// must write it directly — because batching or result-batching is off, or the result is too large - /// for a single table property and needs the chunked path. - /// - public bool TryQueueResult(string runName, string taskId, string resultJson) - { - if (!_enabled || !_batchResults || resultJson.Length > SmallResultMaxChars) return false; - lock (_lock) { _pendingResults[Key(runName, taskId)] = new ResultWrite(runName, taskId, resultJson); } - Signal(); - return true; - } - - /// Flush all currently-pending writes and await their persistence. Call before finalizing a run - /// so terminal task states + run state are durable before post-execution reads results. - public async Task FlushAsync(CancellationToken ct = default) - { - if (!_enabled) return; - Task barrier; - lock (_lock) { barrier = _barrier.Task; } - Signal(); - - // Bounded for the same reason as the Running marker: this is awaited by FinalizeRunAsync, and an - // unbounded wait meant no run could finalize while a flush was stuck. Requeue-on-failure means a - // timeout here does not lose the writes — they persist on a later flush or on the shutdown drain, - // so finalization proceeds rather than blocking on transient storage trouble. - try - { - if (await CompletesWithinAsync(barrier, _barrierTimeout, ct)) return; - _logger.LogWarning("[Orchestrator] Flush barrier not met within {Sec}s — pending writes remain queued", - _barrierTimeout.TotalSeconds); - } - catch (Exception ex) when (ex is not OperationCanceledException) - { - _logger.LogWarning(ex, "[Orchestrator] Flush barrier reported a failed write — requeued for retry"); - } - } - - private async Task DrainLoopAsync() - { - while (!_cts.IsCancellationRequested) - { - try { await _signal.WaitAsync(_flushIntervalMs, _cts.Token); } - catch (OperationCanceledException) { break; } - - // FlushOnceAsync used to be called outside any try. Anything escaping it — an OOM while - // formatting the error log is enough — killed this loop silently, and a dead loop means no - // barrier is ever completed again, so every task hangs at its durable marker forever. - // This loop must not be able to die while the process lives. - try { await FlushOnceAsync(_flushTimeout); } - catch (Exception ex) { LogSafely(ex, "status flush"); } - } - - // Final drain on shutdown. Deliberately NOT bounded by _cts (which is already cancelled) and - // given a generous ceiling: this is the last chance for terminal task states to reach storage. - try { await FlushOnceAsync(TimeSpan.FromSeconds(30), ignoreShutdown: true); } - catch (Exception ex) { LogSafely(ex, "final status drain"); } - } - - /// Log without letting the logger itself take the drain loop down (it allocates, and this - /// path runs under exactly the memory pressure that makes allocation fail). - private void LogSafely(Exception ex, string what) - { - try { _logger.LogError(ex, "[Orchestrator] Unhandled error during {What} — drain loop continues", what); } - catch { /* nothing useful left to do; staying alive matters more than reporting */ } - } - - private async Task FlushOnceAsync(TimeSpan timeout, bool ignoreShutdown = false) - { - Dictionary tasks; - Dictionary runs; - Dictionary results; - TaskCompletionSource done; - lock (_lock) - { - done = _barrier; - _barrier = new(TaskCreationOptions.RunContinuationsAsynchronously); - if (_pendingTasks.Count == 0 && _pendingRuns.Count == 0 && _pendingResults.Count == 0) - { - // Nothing to flush — still release barrier waiters (e.g. FlushAsync on an already-drained run). - done.TrySetResult(); - return; - } - tasks = _pendingTasks; _pendingTasks = new(); - runs = _pendingRuns; _pendingRuns = new(); - results = _pendingResults; _pendingResults = new(); - } - - using var cts = ignoreShutdown - ? new CancellationTokenSource(timeout) - : CancellationTokenSource.CreateLinkedTokenSource(_cts.Token); - if (!ignoreShutdown) cts.CancelAfter(timeout); - - var unwritten = new List(); - Exception? failure = null; - - try - { - // Results FIRST, and a run whose result did not land withholds that run's terminal task and - // run-status writes THIS flush. A terminal marker must never persist ahead of its result: a - // crash in between would leave a task counted done with no result for post-execution to read. - var failedResultRuns = new HashSet(StringComparer.Ordinal); - if (results.Count > 0) - { - var failedResults = await _store.WriteResultBatchAsync(results.Values.ToList(), _flushConcurrency, cts.Token); - foreach (var run in failedResults) failedResultRuns.Add(run); - unwritten.AddRange(failedResults); - } - - if (tasks.Count > 0) - { - var toWrite = failedResultRuns.Count == 0 - ? tasks.Values.ToList() - : tasks.Values.Where(t => !failedResultRuns.Contains(t.RunName)).ToList(); - unwritten.AddRange(await _store.WriteTaskStatusBatchAsync(toWrite, _flushConcurrency, cts.Token)); - // Tasks held back because their run's result failed: requeue so they retry with the result. - if (failedResultRuns.Count > 0) - unwritten.AddRange(tasks.Values.Where(t => failedResultRuns.Contains(t.RunName)).Select(t => t.RunName)); - } - - // Batched, not one await per run. Run rows all share the "Run" partition key, so N of them - // cost ceil(N/100) transactions; the previous per-run loop cost N round-trips inside a flush - // bounded by _flushTimeout, which is how a busy instance (100+ live runs) blew the flush - // budget and then the durable barrier. The store isolates a poison row by retrying a failed - // chunk individually, so a single bad run no longer discards the other 99. - if (runs.Count > 0) - { - var toWrite = failedResultRuns.Count == 0 - ? runs.Values.ToList() - : runs.Values.Where(r => !failedResultRuns.Contains(r.Name)).ToList(); - unwritten.AddRange(await _store.WriteRunStatusBatchAsync(toWrite, cts.Token)); - if (failedResultRuns.Count > 0) - unwritten.AddRange(runs.Values.Where(r => failedResultRuns.Contains(r.Name)).Select(r => r.Name)); - } - } - catch (Exception ex) - { - // Timed out or the store threw wholesale — treat EVERYTHING in this snapshot as unwritten. - failure = ex; - unwritten.AddRange(results.Values.Select(r => r.RunName)); - unwritten.AddRange(tasks.Values.Select(t => t.RunName)); - unwritten.AddRange(runs.Keys); - _logger.LogError(ex, "[Orchestrator] Status flush failed ({Results} results, {Tasks} tasks, {Runs} runs) — requeued for retry", - results.Count, tasks.Count, runs.Count); - } - - // Durability: anything that did not reach storage goes back on the pending set. Dropping it - // (the previous behaviour on any exception) silently lost terminal task states — a completed - // task would look Pending forever and be re-run by the next recovery. - if (unwritten.Count > 0) Requeue(results, tasks, runs, unwritten); - - // The barrier may ONLY report success when this batch actually persisted. Waiters cannot tell - // which run in the batch was theirs, so any un-persisted write has to fail all of them — a - // waiter that returns successfully goes on to run its task, and running a task whose "Running" - // marker never landed is exactly the poison-retry hole the marker exists to close. Failing here - // is cheap: the writes are requeued, and the waiters defer and retry. - if (failure != null) - done.TrySetException(failure); - else if (unwritten.Count > 0) - done.TrySetException(new InvalidOperationException( - $"{unwritten.Count} status write(s) did not persist and were requeued")); - else - done.TrySetResult(); - } - - /// - /// Put un-persisted writes back. , not assignment: a - /// NEWER state for the same task may have arrived while the flush was in flight, and the retry of a - /// stale snapshot must never overwrite it. - /// - private void Requeue(Dictionary results, Dictionary tasks, - Dictionary runs, List unwrittenRuns) - { - var retry = new HashSet(unwrittenRuns, StringComparer.OrdinalIgnoreCase); - var restored = 0; - - lock (_lock) - { - // Results before tasks, mirroring the flush order: a result put back must be in the pending - // set before the terminal marker it guards is retried. - foreach (var (key, write) in results) - { - if (!retry.Contains(write.RunName)) continue; - if (_pendingResults.TryAdd(key, write)) restored++; - } - foreach (var (key, write) in tasks) - { - if (!retry.Contains(write.RunName)) continue; - if (_pendingTasks.TryAdd(key, write)) restored++; - } - foreach (var (name, run) in runs) - { - if (!retry.Contains(name)) continue; - if (_pendingRuns.TryAdd(name, run)) restored++; - } - } - - if (restored > 0) - { - Signal(); // make sure the next flush actually runs - _logger.LogWarning("[Orchestrator] Requeued {Count} un-persisted status writes across {Runs} runs", - restored, retry.Count); - } - } - - /// - /// Wake the drain loop. Guarded so a status enqueue that races — an in-flight - /// job finishing after shutdown began — drops the wake instead of throwing ObjectDisposedException. - /// By the time a task queues its terminal status its data is already persisted, so a lost coalesced - /// wake on the way down is harmless; a task falsely marked Failed because the release threw is not. - /// - private void Signal() - { - if (_disposed) return; - try { _signal.Release(); } - catch (ObjectDisposedException) { /* shutting down; the drain loop has already stopped */ } - } - - public void Dispose() - { - _disposed = true; - _cts.Cancel(); - try { _drainLoop.Wait(TimeSpan.FromSeconds(5)); } catch { /* best effort final drain */ } - _cts.Dispose(); - _signal.Dispose(); - } -} diff --git a/Services/Orchestration/OrchestratorTaskItem.cs b/Services/Orchestration/OrchestratorTaskItem.cs index 3df1994..59a1bf5 100644 --- a/Services/Orchestration/OrchestratorTaskItem.cs +++ b/Services/Orchestration/OrchestratorTaskItem.cs @@ -1,36 +1,9 @@ namespace Craft.Orchestration; +/// A task parsed from a batch or planner output, before it is stored. public class OrchestratorTaskItem { public string Id { get; set; } = string.Empty; public string Status { get; set; } = "Pending"; public Dictionary Parameters { get; set; } = []; - public int AttemptCount { get; set; } - public string? LastError { get; set; } - public DateTime? CompletedUtc { get; set; } - - /// - /// Dispatch priority override for this task alone. Null — the normal case — means "inherit the - /// run's priority", so existing rows and everything the planner emits behave exactly as before - /// with no backfill needed. - /// - /// Set only when an operator reprioritizes one queued job (JobManager.ChangePriority). - /// Persisted so the override survives a restart, instead of silently reverting to the run's - /// priority when ResumeInterruptedRunsAsync re-queues the task. - /// - public int? Priority { get; set; } - - /// - /// Position of this task in the batch as submitted (0-based). Only meaningful for a run marked - /// , where tasks are dispatched one at a time in ascending - /// Sequence order. Non-sequential runs leave it 0 and ignore it. - /// - public int Sequence { get; set; } - - /// - /// Set when a worker in THIS process marks the task Running. Never persisted, so a Running status - /// rehydrated from storage (another process's pre-invoke marker) reads false: it is not proof that - /// anything here is executing the task. See OrchestratorService.ResolveTaskWorkAsync. - /// - internal bool OwnedHere; } diff --git a/Services/Orchestration/WorkPump.cs b/Services/Orchestration/WorkPump.cs new file mode 100644 index 0000000..ff886a5 --- /dev/null +++ b/Services/Orchestration/WorkPump.cs @@ -0,0 +1,424 @@ +using Craft.Configuration; +using Craft.Storage; + +namespace Craft.Orchestration; + +/// +/// Keeps the JobManager's buffer topped up from storage. Each refill reads the Ready list (best band first, +/// oldest run first), claims from those runs' partitions until the batch is full, and hands the claims to the +/// JobManager as descriptors. A run with nothing claimable is skipped for a while rather than read every tick. +/// +/// One process works the queue: the pump claims nothing until it holds the instance lock (a single row, renewed +/// every few seconds and released on shutdown), so a recycle never has two processes claiming at once. While it +/// holds the lock, any claim in storage owned by another process belongs to one that has stopped, and is taken +/// back the first time its run is read. The pump holds no other state that matters after a crash. +/// +public class WorkPump : BackgroundService +{ + private readonly ILogger _logger; + private readonly WorkStore _store; + private readonly JobManager _jobs; + private readonly OrchestratorService? _orchestrator; + private readonly string _owner; + private readonly int _batchSize; + private readonly int _lowWater; + private readonly TimeSpan _lease; + private readonly TimeSpan _pollInterval; + private readonly TimeSpan _idlePollInterval; + private readonly Task? _claimGate; + + /// + /// Runs the pump has written off for now, so they cost no storage read until something can have changed. + /// + /// A run whose pending tasks ran out (its remaining work is running, or it waits on child runs) can only + /// have claimable work again when its counts move (a task or child finishes, its aggregation falls due), + /// when a claim is released back to pending, or when a claim lapses unrenewed. So it is skipped until its + /// Ready counts change, a release names it, or its earliest claim's lease runs out. Without this, runs + /// waiting at the head of a large queue spend the per-refill read budget over and over and starve every run + /// behind them. + /// + /// A run that came back empty for another reason (a lost race, a busy sequential driver) is skipped for + /// 30 s, doubling each time it is found empty again, up to 15 minutes, until its counts move. + /// + private readonly Dictionary _skip = new(StringComparer.Ordinal); + private readonly System.Collections.Concurrent.ConcurrentQueue _changed = new(); + private static readonly TimeSpan EmptyBackoff = TimeSpan.FromSeconds(30); + private static readonly TimeSpan MaxEmptyBackoff = TimeSpan.FromMinutes(15); + + /// Runs read from storage per refill. Skipped runs cost nothing. + private const int MaxRunsReadPerRefill = 32; + + /// Ready rows per page: the whole list of a normal instance in one request, and few requests when a + /// large backlog has to be scanned past. + private const int ReadyPageSize = 1000; + + /// The instance lock: how long it is held for without renewal, and how often it is renewed. + private readonly TimeSpan _lockLease; + private readonly TimeSpan _lockRenewEvery; + private DateTime _lockRenewedAt; + private volatile bool _holdsLock; + + /// Whether this pump holds the instance lock (and so may treat other owners' claims as dead). + internal bool HoldsLock => _holdsLock; + + /// How long a run may sit with its creation unfinished before the startup repair removes it. + private static readonly TimeSpan AbandonUnfinishedCreation = TimeSpan.FromMinutes(10); + + /// Claims handed to the JobManager, by job id, with when their lease runs out. + private readonly Dictionary _inFlight = new(StringComparer.Ordinal); + + public WorkPump(ILogger logger, WorkStore store, JobManager jobs, IConfiguration configuration, + CraftSettings settings, OrchestratorService? orchestrator = null) + { + _logger = logger; + _store = store; + _jobs = jobs; + _orchestrator = orchestrator; + _claimGate = orchestrator?.RecoveryDone; + _owner = orchestrator?.Owner ?? OrchestratorService.NewOwnerId(); + _batchSize = Math.Max(1, configuration.GetValue("JobQueueBatchSize", Math.Max(1, settings.Worker.BgPoolSize))); + _lowWater = Math.Max(0, configuration.GetValue("JobQueueLowWaterMark", 2)); + _lease = orchestrator?.Lease ?? TimeSpan.FromSeconds(Math.Max(60, configuration.GetValue("JobQueueLeaseSeconds", 1800))); + _pollInterval = TimeSpan.FromMilliseconds(Math.Max(100, configuration.GetValue("JobQueuePollIntervalMs", 1000))); + _idlePollInterval = TimeSpan.FromMilliseconds(Math.Max(_pollInterval.TotalMilliseconds, + configuration.GetValue("JobQueueIdlePollIntervalMs", 10_000))); + _lockLease = TimeSpan.FromSeconds(Math.Max(5, configuration.GetValue("InstanceLockSeconds", 30))); + _lockRenewEvery = _lockLease / 3; + _store.RunChanged += runKey => + { + _changed.Enqueue(runKey); + Wake(); + }; + _jobs.Dispatched += () => + { + if (_jobs.QueuedCount <= _lowWater) Wake(); + }; + } + + /// + /// Refill now rather than at the next poll: the JobManager's buffer has drained to the low-water mark, or a + /// run was created or moved on in this process. Wakes are coalesced to one refill per + /// , so a burst of finishes does not re-read the Ready list for each one. + /// + private void Wake() + { + try + { + if (_wake.CurrentCount == 0) _wake.Release(); + } + catch (SemaphoreFullException) { } + } + + private readonly SemaphoreSlim _wake = new(0, 1); + private static readonly TimeSpan MinRefillGap = TimeSpan.FromMilliseconds(50); + + protected override async Task ExecuteAsync(CancellationToken stoppingToken) + { + _logger.LogInformation("[WorkPump] Started: owner={Owner} batch={Batch} lowWater={Low} lease={Lease}s", + _owner, _batchSize, _lowWater, _lease.TotalSeconds); + + try + { + if (_claimGate is { IsCompleted: false }) await _claimGate.WaitAsync(stoppingToken); + await AcquireLockAsync(stoppingToken); + } + catch (OperationCanceledException) { return; } + _ = RepairAsync(stoppingToken); + + var idleTicks = 0; + while (!stoppingToken.IsCancellationRequested) + { + var claimed = 0; + try + { + if (!await KeepLockAsync(stoppingToken)) await AcquireLockAsync(stoppingToken); + claimed = await RefillAsync(stoppingToken); + await RenewAsync(stoppingToken); + } + catch (OperationCanceledException) when (stoppingToken.IsCancellationRequested) { break; } + catch (Exception ex) + { + try { _logger.LogError(ex, "[WorkPump] Cycle failed; continuing"); } catch { } + } + + idleTicks = claimed > 0 || _inFlight.Count > 0 ? 0 : idleTicks + 1; + var delay = idleTicks == 0 + ? _pollInterval + : TimeSpan.FromMilliseconds(Math.Min(_idlePollInterval.TotalMilliseconds, + _pollInterval.TotalMilliseconds * (1L << Math.Min(idleTicks, 20)))); + if (delay > _lockRenewEvery) delay = _lockRenewEvery; + try + { + if (await _wake.WaitAsync(delay, stoppingToken)) + { + idleTicks = 0; + await Task.Delay(MinRefillGap, stoppingToken); + } + } + catch (OperationCanceledException) { break; } + } + } + + internal void ForgetBackoff() => _skip.Clear(); + + /// The job a claimed task runs as: its id, and the name job lists and worker stats show. + internal static string JobId(WorkStore.ClaimedTask c) => $"{c.RunKey}|{c.Seq}"; + + internal static string JobName(string runName, WorkStore.ClaimedTask c) => + c.Seq == WorkStore.AggregateSeq ? $"{runName}-PostExec" : $"{runName}-{c.TaskId}"; + + /// + /// Wait until this process holds the instance lock. A predecessor that shut down cleanly released it, so this + /// is immediate after a normal recycle; one that crashed holds it until its lease runs out. + /// + internal async Task AcquireLockAsync(CancellationToken ct) + { + string? waitingOn = null; + while (true) + { + var (held, holder) = await _store.TryHoldInstanceLockAsync(_owner, _lockLease, ct); + if (held) + { + _holdsLock = true; + _lockRenewedAt = DateTime.UtcNow; + _logger.LogInformation("[WorkPump] Holding the instance lock as {Owner}; claims of any other process are taken back on sight", + _owner); + return; + } + if (holder?.Owner != waitingOn) + { + waitingOn = holder?.Owner; + _logger.LogInformation("[WorkPump] Waiting for the instance lock, held by {Holder} until {Until:O}", + holder?.Owner, holder?.LeaseUntil); + } + var wait = (holder?.LeaseUntil ?? DateTimeOffset.UtcNow) - DateTimeOffset.UtcNow; + await Task.Delay(wait < TimeSpan.FromSeconds(1) ? TimeSpan.FromSeconds(1) : wait > _lockRenewEvery ? _lockRenewEvery : wait, ct); + } + } + + /// Renew the instance lock when due. False when it was lost (a renewal failed long enough for + /// another process to take it): claiming stops until it is held again. + internal async Task KeepLockAsync(CancellationToken ct) + { + if (!_holdsLock) return false; + if (DateTime.UtcNow - _lockRenewedAt < _lockRenewEvery) return true; + try + { + var (held, holder) = await _store.TryHoldInstanceLockAsync(_owner, _lockLease, ct); + if (held) + { + _lockRenewedAt = DateTime.UtcNow; + return true; + } + _holdsLock = false; + _logger.LogCritical("[WorkPump] Lost the instance lock to {Holder}; claiming stops until it is held again", holder?.Owner); + return false; + } + catch (Exception ex) when (ex is not OperationCanceledException) + { + // A storage blip: keep claiming while the lease we hold is still good, and say so if it is not. + if (DateTime.UtcNow - _lockRenewedAt < _lockLease) return true; + _holdsLock = false; + _logger.LogCritical(ex, "[WorkPump] Could not renew the instance lock before it ran out; claiming stops until it is held again"); + return false; + } + } + + private int _repairing; + private int _indexFailuresSeen; + + /// + /// Rebuild the indexes from the active-run list: once when the lock is first held, and again whenever an index + /// write has failed for good since the last pass (so a run whose Ready entry never landed is relisted now, not + /// at the next restart). Runs beside claiming; one pass at a time. + /// + private async Task RepairAsync(CancellationToken ct) + { + if (Interlocked.Exchange(ref _repairing, 1) == 1) return; + _indexFailuresSeen = _store.IndexFailures; + try + { + var r = await _store.RepairIndexesAsync(AbandonUnfinishedCreation, ct); + _logger.LogInformation("[WorkPump] Index repair: {Active} active run(s), {Relisted} relisted, {Retired} retired, {Removed} unfinished creation(s) removed, {Young} still being created", + r.Active, r.Relisted, r.Retired, r.Removed, r.Young); + } + catch (Exception ex) when (ex is not OperationCanceledException) + { + _logger.LogError(ex, "[WorkPump] Index repair failed; it runs again at the next start, or via RepairIndexes"); + } + finally + { + Volatile.Write(ref _repairing, 0); + } + } + + /// The background repair started by the last refill that saw a failed index write; tests await it. + internal Task? LastRepair { get; private set; } + + /// The pump's notion of now, for its backoff and renewal timing; tests replace it. + internal Func Clock { get; set; } = () => DateTime.UtcNow; + + /// Stop tracking claims the JobManager is done with. Their finish (or release) was written by the job. + private void Forget() + { + foreach (var id in _inFlight.Keys.Where(id => !_jobs.IsQueuedOrRunning(id)).ToList()) + _inFlight.Remove(id); + } + + /// + /// Claim until the buffer holds a batch. Returns how many tasks were claimed. + /// + /// A run's concurrency limit is applied here, against the claims this process holds: only one Craft + /// instance works the tables, so what it holds is what is running. A claim left by a process that has + /// since died is not running, so it rightly does not count. During an overlapping restart, two processes + /// could briefly run up to the limit each. + /// + internal async Task RefillAsync(CancellationToken ct) + { + Forget(); + while (_changed.TryDequeue(out var changedRun)) _skip.Remove(changedRun); + if (_store.IndexFailures != _indexFailuresSeen && Volatile.Read(ref _repairing) == 0) LastRepair = RepairAsync(ct); + if (_jobs.QueuedCount > _lowWater) return 0; + var need = _batchSize - _jobs.QueuedCount; + var claimed = 0; + var now = Clock(); + var read = 0; + Dictionary? held = null; + + await foreach (var entry in _store.ReadReadyAsync(ReadyPageSize, ct)) + { + if (need <= 0 || read >= MaxRunsReadPerRefill) break; + var progressed = true; + if (_skip.TryGetValue(entry.RunKey, out var skip) && skip.Done == entry.Done && skip.Total == entry.Total) + { + if (skip.Until > now) continue; + progressed = false; + } + + var want = need; + if (entry.MaxConcurrency > 0 && !entry.Sequential) + { + held ??= _inFlight.Values.GroupBy(v => v.Claim.RunKey).ToDictionary(g => g.Key, g => g.Count(), StringComparer.Ordinal); + want = Math.Min(need, entry.MaxConcurrency - held.GetValueOrDefault(entry.RunKey)); + if (want <= 0) continue; + } + read++; + + var header = await _store.GetRunAsync(entry.RunKey, ct); + if (header == null || header.IsFinished) + { + await _store.DropReadyAsync(entry, ct); + continue; + } + + // A run being cancelled: claiming its pending tasks would only race the cancel. Cancel a page of them instead, + // so a cancel interrupted part-way still finishes; its aggregation runs once it falls due. + if (header.CancelRequested && header.Phase == RunPhase.Tasks) + { + var (cancelled, _) = _store.IsCancelling(header.RunKey) + ? (0, null) + : await _store.CancelPendingAsync(header.RunKey, maxPages: 1, ct: ct); + if (cancelled == 0) _skip[header.RunKey] = (entry.Done, entry.Total, DateTime.MaxValue, 0); + continue; + } + + IReadOnlyList claims; + var probe = new WorkStore.ClaimProbe(); + if (header.Sequential) + { + var step = await _store.ClaimSequentialAsync(header.RunKey, _owner, _lease, othersAreDead: _holdsLock, ct: ct); + claims = step == null ? [] : [step]; + } + else + { + claims = await _store.ClaimAsync(header.RunKey, want, _owner, _lease, reclaimExpired: true, probe, othersAreDead: _holdsLock, ct); + } + + if (probe.PendingExhausted) + { + // A lapsing claim is the one change that bumps no count; look again when the earliest lease is due. + _skip[header.RunKey] = (entry.Done, entry.Total, probe.EarliestLeaseUntil?.UtcDateTime ?? DateTime.MaxValue, 0); + } + else if (claims.Count == 0) + { + var strikes = progressed ? 0 : skip.Strikes + 1; + var backoff = TimeSpan.FromTicks(Math.Min(MaxEmptyBackoff.Ticks, EmptyBackoff.Ticks << Math.Min(strikes, 10))); + _skip[header.RunKey] = (entry.Done, entry.Total, now + backoff, strikes); + continue; + } + else + { + _skip.Remove(header.RunKey); + } + if (claims.Count == 0) continue; + + foreach (var c in claims) + { + var descriptor = new JobDescriptor(header.Name, c.TaskId, header.Priority) { RunKey = c.RunKey, Seq = c.Seq, Attempt = c.Attempt }; + var jobId = _jobs.Enqueue(descriptor, JobName(header.Name, c), id: JobId(c)); + _inFlight[jobId] = (c, now + _lease); + if (held != null) held[c.RunKey] = held.GetValueOrDefault(c.RunKey) + 1; + } + need -= claims.Count; + claimed += claims.Count; + } + + // ponytail: forgetting every mark costs one read per run on the next pass; prune by Ready membership if that ever shows up. + if (_skip.Count > 50_000) _skip.Clear(); + + return claimed; + } + + /// Renew claims in their last third, so a long buffer wait or a long task never loses its lease. + internal async Task RenewAsync(CancellationToken ct) + { + var now = Clock(); + var due = _inFlight.Where(kv => kv.Value.LeaseUntil - now < _lease / 3).ToList(); + if (due.Count == 0) return; + + var lost = await _store.RenewAsync(due.Select(kv => kv.Value.Claim).ToList(), _owner, _lease, ct); + if (lost.Count > 0) + _logger.LogWarning("[WorkPump] {Count} claim(s) were taken back before renewal — their leases had lapsed", lost.Count); + foreach (var (id, v) in due) _inFlight[id] = (v.Claim, now + _lease); + } + + /// On shutdown, hand back claims that never started so another process can run them now. + public override async Task StopAsync(CancellationToken cancellationToken) + { + await base.StopAsync(cancellationToken); + _logger.LogInformation("[WorkPump] Stopping with {Count} claim(s) in flight; lock held={Held}", _inFlight.Count, _holdsLock); + foreach (var (id, v) in _inFlight.ToList()) + { + if (!_jobs.WithdrawJob(id)) continue; + try { await _store.ReleaseAsync(v.Claim.RunKey, v.Claim.Seq, _owner, refundAttempt: true, cancellationToken); } + catch (Exception ex) { _logger.LogDebug(ex, "[WorkPump] Could not release {Job} on shutdown", id); } + } + if (_holdsLock) + { + // Hold the lock while this process still runs claimed tasks: a successor holding it would take those + // claims back as dead and run them a second time. If shutdown is cut short, the lock simply lapses. + try + { + while (_jobs.GetJobs(status: "Running").Any(j => _inFlight.ContainsKey(j.Id))) + { + await KeepLockAsync(cancellationToken); + await Task.Delay(250, cancellationToken); + } + } + catch (OperationCanceledException) + { + _logger.LogWarning("[WorkPump] Shutdown cut short with tasks still running; the instance lock lapses at its lease"); + return; + } + try + { + if (await _store.ReleaseInstanceLockAsync(_owner, cancellationToken)) + _logger.LogInformation("[WorkPump] Released the instance lock"); + else + _logger.LogWarning("[WorkPump] Could not release the instance lock on shutdown; it lapses at its lease"); + } + catch (Exception ex) { _logger.LogWarning(ex, "[WorkPump] Could not release the instance lock on shutdown; it lapses at its lease"); } + _holdsLock = false; + } + } +} diff --git a/Services/PowerShellHost/PowerShellRunnerService.cs b/Services/PowerShellHost/PowerShellRunnerService.cs index a08a8c1..080e43d 100644 --- a/Services/PowerShellHost/PowerShellRunnerService.cs +++ b/Services/PowerShellHost/PowerShellRunnerService.cs @@ -502,6 +502,7 @@ public async Task ExecuteScript(string functionName, Dictionary? { WorkerId = $"W{worker.Id}", RunName = parentRun, + RunKey = OperationContext.Current?.RunKey, Priority = OperationContext.Current?.Priority, Category = "Job" }; @@ -644,6 +645,7 @@ public async Task ExecuteScriptWithOutput(string functionName, Dictionar { WorkerId = $"W{worker.Id}", RunName = parentRun, + RunKey = OperationContext.Current?.RunKey, Priority = OperationContext.Current?.Priority, Category = "Planner" }; diff --git a/Services/PowerShellHost/PowerShellWorker.cs b/Services/PowerShellHost/PowerShellWorker.cs index 33a0be5..f57ba9c 100644 --- a/Services/PowerShellHost/PowerShellWorker.cs +++ b/Services/PowerShellHost/PowerShellWorker.cs @@ -337,13 +337,16 @@ public async Task> InvokeAsync(string functionName, Diction if (prof) buildTicks = System.Diagnostics.Stopwatch.GetTimestamp() - bStart; var rStart = prof ? System.Diagnostics.Stopwatch.GetTimestamp() : 0; - var asyncResult = _pwsh.BeginInvoke(); - var results = await Task.Factory.FromAsync(asyncResult, _pwsh.EndInvoke); + // A fresh output collection per invocation: after an invocation that threw, the PowerShell object's + // own output buffer comes back null from the next EndInvoke, silently dropping that call's output. + using var outputs = new PSDataCollection(); + var asyncResult = _pwsh.BeginInvoke(null, outputs); + await Task.Factory.FromAsync(asyncResult, _pwsh.EndInvoke); ct.ThrowIfCancellationRequested(); if (prof) runTicks = System.Diagnostics.Stopwatch.GetTimestamp() - rStart; var cpStart = prof ? System.Diagnostics.Stopwatch.GetTimestamp() : 0; - var coll = new Collection(results?.ToList() ?? new List()); + var coll = new Collection(outputs.ReadAll()); if (prof) copyTicks = System.Diagnostics.Stopwatch.GetTimestamp() - cpStart; return coll; } @@ -382,11 +385,13 @@ public async Task> InvokeScriptAsync(ScriptBlock scriptBloc if (ct.CanBeCanceled) registration = ct.Register(() => _pwsh.Stop()); - var asyncResult = _pwsh.BeginInvoke(); - var results = await Task.Factory.FromAsync(asyncResult, _pwsh.EndInvoke); + // A fresh output collection per invocation; see InvokeAsync. + using var outputs = new PSDataCollection(); + var asyncResult = _pwsh.BeginInvoke(null, outputs); + await Task.Factory.FromAsync(asyncResult, _pwsh.EndInvoke); ct.ThrowIfCancellationRequested(); - return new Collection(results?.ToList() ?? new List()); + return new Collection(outputs.ReadAll()); } catch (PipelineStoppedException) when (ct.IsCancellationRequested) { diff --git a/Services/Storage/AzureTableStore.cs b/Services/Storage/AzureTableStore.cs index 2bd6a25..f9e0d3e 100644 --- a/Services/Storage/AzureTableStore.cs +++ b/Services/Storage/AzureTableStore.cs @@ -80,6 +80,7 @@ private TableClient Client(string table) => private TableServiceClient Service => _service ??= new TableServiceClient(_connectionString.Value, _clientOptions); private static readonly string[] select = new[] { "PartitionKey", "RowKey" }; + private static readonly string[] PartLookupProperties = ["PartitionKey", "RowKey", EntitySplitter.OriginalEntityIdKey]; public async Task PingAsync(CancellationToken ct = default) { @@ -269,6 +270,45 @@ public async Task TryReplaceBatchAsync(string table, string partitionKey, return true; } + public async Task DeleteTableAsync(string table, CancellationToken ct = default) + { + try { await Service.DeleteTableAsync(table, ct); } + catch (RequestFailedException ex) when (ex.Status == 404) { } + } + + public async Task TrySubmitAsync(string table, string partitionKey, IReadOnlyList ops, + CancellationToken ct = default) + { + if (ops.Count == 0) return true; + if (ops.Count > MaxBatch) + throw new ArgumentException($"A transaction is limited to {MaxBatch} ops, got {ops.Count}.", nameof(ops)); + + var actions = ops.Select(op => op.Kind switch + { + StoreOpKind.Insert => new TableTransactionAction(TableTransactionActionType.Add, ToEntity(op.Row)), + StoreOpKind.Replace => new TableTransactionAction(TableTransactionActionType.UpdateReplace, ToEntity(op.Row), + new ETag(op.Row.ETag ?? throw new ArgumentException($"Replace of {op.Row.RowKey} needs an ETag.", nameof(ops)))), + StoreOpKind.Delete => new TableTransactionAction(TableTransactionActionType.Delete, + new TableEntity(op.Row.PartitionKey, op.Row.RowKey), op.Row.ETag is { } etag ? new ETag(etag) : ETag.All), + _ => new TableTransactionAction(TableTransactionActionType.UpsertReplace, ToEntity(op.Row)), + }).ToList(); + + try + { + await Client(table).SubmitTransactionAsync(actions, ct); + return true; + } + catch (RequestFailedException ex) when (IsTableNotFound(ex)) + { + await RecreateTableAsync(table, ct); + return false; + } + catch (RequestFailedException ex) when (ex.Status is 412 or 404 or 409) + { + return false; + } + } + private async Task SubmitAsync(string table, TableClient client, List batch, CancellationToken ct) { try @@ -462,6 +502,23 @@ public async IAsyncEnumerable QueryTableAsync(string table, string? fi yield return ToRow(entity); } + public async IAsyncEnumerable QueryTableAsync(string table, string? filter, int maxPerPage, + [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) + { + await foreach (var entity in StreamReassembledAsync(table, () => Client(table).QueryAsync(filter: filter, maxPerPage: maxPerPage, cancellationToken: ct), ct)) + yield return ToRow(entity); + } + + public async IAsyncEnumerable QueryRowKeyRangeAsync(string table, string partitionKey, string fromRowKey, + string toRowKey, IReadOnlyList? properties = null, int? maxPerPage = null, + [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) + { + var filter = $"PartitionKey eq '{Escape(partitionKey)}' and RowKey ge '{Escape(fromRowKey)}' and RowKey lt '{Escape(toRowKey)}'"; + await foreach (var entity in EnumerateAsync(table, () => Client(table).QueryAsync(filter: filter, maxPerPage: maxPerPage, + select: properties, cancellationToken: ct), ct)) + yield return ToRow(entity); + } + public async Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) { try @@ -623,7 +680,10 @@ private async Task> RecoverMissingPartRowsAsync(string table, var filter = $"{partitionClause} and {BuildRowKeyPrefixClause(entityId)}"; await foreach (var row in Client(table).QueryAsync(filter: filter, cancellationToken: ct)) { - if (seen.Add((row.PartitionKey, row.RowKey))) + // The prefix range also holds unrelated keys ("task1-partner"); only this entity's rows belong. + var owned = row.RowKey == entityId || + (row.TryGetValue(EntitySplitter.OriginalEntityIdKey, out var owner) && owner?.ToString() == entityId); + if (owned && seen.Add((row.PartitionKey, row.RowKey))) rows.Add(row); } } @@ -675,11 +735,14 @@ private async Task> RecoverMissingPartRowsAsync(string table, private async Task RemoveStalePartRowsAsync(string table, string partitionKey, string originalRowKey, HashSet live, CancellationToken ct) { - var filter = $"PartitionKey eq '{Escape(partitionKey)}' and {EntitySplitter.OriginalEntityIdKey} eq '{Escape(originalRowKey)}'"; + // A RowKey range is an index seek; filtering on the marker alone scans the whole partition, which + // every plain delete paid. The range also catches other keys sharing the prefix, so confirm the owner. + var filter = $"PartitionKey eq '{Escape(partitionKey)}' and {BuildRowKeyPrefixClause($"{originalRowKey}-part")}"; var stale = new List(); - await foreach (var row in Client(table).QueryAsync(filter: filter, select: select, cancellationToken: ct)) + await foreach (var row in Client(table).QueryAsync(filter: filter, select: PartLookupProperties, cancellationToken: ct)) { - if (!live.Contains(row.RowKey)) + if (row.TryGetValue(EntitySplitter.OriginalEntityIdKey, out var owner) && owner?.ToString() == originalRowKey + && !live.Contains(row.RowKey)) stale.Add(row.RowKey); } diff --git a/Services/Storage/ICraftTableStore.cs b/Services/Storage/ICraftTableStore.cs index 2d7ab0d..f74ac07 100644 --- a/Services/Storage/ICraftTableStore.cs +++ b/Services/Storage/ICraftTableStore.cs @@ -83,6 +83,62 @@ IAsyncEnumerable QueryTableAsync(string table, string? filter, IReadOn CancellationToken ct = default) => QueryTableAsync(table, filter, ct); + /// The filtered scan, fetched rows per request, for callers that + /// stop after the first few matches. Same contract as the filter: a backend may ignore both. + IAsyncEnumerable QueryTableAsync(string table, string? filter, int maxPerPage, + CancellationToken ct = default) + => QueryTableAsync(table, filter, ct); + + /// + /// Rows of one partition with <= RowKey < + /// (ordinal), optionally projected (name the keys too if you read them). Split entities are not + /// reassembled, so use it only on tables whose rows are never split. is + /// the page size asked of the service ($top); a caller that needs a few rows should pass it, or each + /// request returns up to 1,000. + /// + async IAsyncEnumerable QueryRowKeyRangeAsync(string table, string partitionKey, string fromRowKey, + string toRowKey, IReadOnlyList? properties = null, int? maxPerPage = null, + [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) + { + await foreach (var row in QueryPartitionAsync(table, partitionKey, ct)) + if (string.CompareOrdinal(row.RowKey, fromRowKey) >= 0 && string.CompareOrdinal(row.RowKey, toRowKey) < 0) + yield return row; + } + + /// + /// Apply to one partition as a single all-or-nothing transaction (at most 100 + /// ops, rows never split). Insert fails if the row exists; Replace and a Delete carrying an ETag fail if + /// the row changed or is gone. Returns false, with nothing written, when any guard fails. + /// + /// This default checks every guard and then applies the ops one by one, which is atomic only for a + /// single-threaded caller; submits a real transaction. + /// + async Task TrySubmitAsync(string table, string partitionKey, IReadOnlyList ops, + CancellationToken ct = default) + { + foreach (var op in ops) + { + var current = await GetAsync(table, op.Row.PartitionKey, op.Row.RowKey, ct); + var ok = op.Kind switch + { + StoreOpKind.Insert => current == null, + StoreOpKind.Replace => current != null && current.ETag == op.Row.ETag, + StoreOpKind.Delete => op.Row.ETag == null || (current != null && current.ETag == op.Row.ETag), + _ => true, + }; + if (!ok) return false; + } + foreach (var op in ops) + { + if (op.Kind == StoreOpKind.Delete) await DeleteAsync(table, op.Row.PartitionKey, op.Row.RowKey, ct); + else await UpsertAsync(table, op.Row, ct); + } + return true; + } + + /// Delete a whole table if it exists. The default does nothing. + Task DeleteTableAsync(string table, CancellationToken ct = default) => Task.CompletedTask; + /// Delete a single row. A missing row is not an error. Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default); @@ -104,3 +160,15 @@ async Task DeleteBatchAsync(string table, string partitionKey, IReadOnlyListDelete every row in a partition. Task DeletePartitionAsync(string table, string partitionKey, CancellationToken ct = default); } + +public enum StoreOpKind { Insert, Replace, Delete, Upsert } + +/// One write in a transaction. Replace and Delete are +/// guarded by (a Delete without one is unconditional). +public readonly record struct StoreOp(StoreOpKind Kind, StoreRow Row) +{ + public static StoreOp Insert(StoreRow row) => new(StoreOpKind.Insert, row); + public static StoreOp Replace(StoreRow row) => new(StoreOpKind.Replace, row); + public static StoreOp Upsert(StoreRow row) => new(StoreOpKind.Upsert, row); + public static StoreOp Delete(StoreRow row) => new(StoreOpKind.Delete, row); +} diff --git a/Services/Storage/JobQueueStore.cs b/Services/Storage/JobQueueStore.cs deleted file mode 100644 index 0dd5f93..0000000 --- a/Services/Storage/JobQueueStore.cs +++ /dev/null @@ -1,910 +0,0 @@ -using System.Globalization; -using Craft.Configuration; - -namespace Craft.Storage; - -/// -/// The durable job queue: one row per queued task, claimed in batches under a lease. -/// -/// This exists so the dispatch side can hold a worker-pool-sized buffer instead of the whole backlog, -/// and so an edit to a queued task in storage is what actually runs — the in-memory copy is a buffer, -/// not the truth. -/// -/// Key design, and it is doing real work: -/// -/// PartitionKey "P04" — the priority bucket, zero-padded so it sorts numerically. -/// RowKey "{queuedTicks:D19}-{run}-{task}" — time-ordered within a bucket, unique by construction. -/// -/// Azure Table returns rows ordered by partition key then row key, so a single unfiltered read yields -/// the highest-priority, oldest-first work FIRST, across every run. That is the cross-run priority -/// ordering the in-memory PriorityQueue provides today, for one round-trip and without probing each -/// bucket in turn — probing 16 buckets per refill would have cost more round-trips than the whole -/// batching exercise saves. -/// -/// A claim is one conditional transaction over rows sharing a bucket: one round-trip per BATCH, not per -/// task. Guarded by each row's ETag, so an external edit — or another instance claiming first — makes -/// the claim fail rather than silently overwrite, and the caller re-reads. -/// -/// THE RUN INDEX, and why the queue table alone is not enough: -/// -/// The key design above is right for the claim path and wrong for everything else. RunName is not part -/// of the key, so every run-scoped read — "which tasks does this run still have queued", "release this -/// run's claims", "drop this run's rows" — can only be answered by scanning every partition. Those run -/// on hot paths: once per finalize, once per resumed run, once per orphan re-drive. -/// -/// Measured on a production instance whose queue reached ~743,000 rows: 61,939 such scans in ten -/// minutes, p50 80 seconds, p95 100 seconds, then TaskCanceledException. The timeouts fell on the task -/// status writes, so tasks never reached a terminal state, so runs never finalized, so their rows were -/// never deleted — and the table that made the scans slow could only grow. 65% of the log file was the -/// resulting HTTP-SLOW warnings; 0.3% was actual task execution. -/// -/// A server-side $filter does not fix this and previously appeared to: filtering on a non-key property -/// narrows what crosses the wire, not what the backend reads. The scan is the cost. -/// -/// So run-scoped access gets its own index table, keyed the way those reads actually ask: -/// -/// PartitionKey the run name, escaped for the key charset -/// RowKey "{bucket}|{queue row key}" — enough to address the queue row directly -/// -/// Every run-scoped method below is now a single-partition read followed by point operations, and the -/// claim path is untouched. The index is maintained by the enqueue/remove paths, and built (and, from -/// schema v2, re-keyed) once for a pre-existing queue by . -/// -public sealed class JobQueueStore : IDisposable -{ - private readonly ILogger _logger; - private readonly ICraftTableStore _store; - private readonly string _queueTable; - private readonly string _indexTable; - private bool _initialized; - - /// - /// Wakes the the moment claimable rows appear, instead - /// of it discovering them only on its next poll tick. Bounded at one pending permit: many enqueues - /// between two pump cycles coalesce into a single wake, because one refill claims a whole batch anyway. - /// The pump keeps polling on its (idle-backing-off) interval as the backstop — this signal only - /// removes the wait, it does not replace the loop. Same-instance only, which is all that is needed: - /// the pump and the enqueue paths share this singleton, and a lease keeps cross-instance work safe. - /// - private readonly SemaphoreSlim _pumpWake = new(0, 1); - - /// Signal the pump that new claimable rows exist. Never throws and never exceeds one permit. - private void WakePump() - { - try { _pumpWake.Release(); } - catch (SemaphoreFullException) { /* a wake is already pending; the pump will claim the batch */ } - catch (ObjectDisposedException) { /* shutting down; the pump loop has already stopped */ } - } - - public void Dispose() => _pumpWake.Dispose(); - - /// - /// Block until the pump is woken by an enqueue or elapses, whichever - /// comes first. Returns true if woken (new work signalled), false on the poll timeout. - /// - public Task WaitForWorkAsync(TimeSpan pollInterval, CancellationToken ct = default) => - _pumpWake.WaitAsync(pollInterval, ct); - - /// Priorities above this share the lowest bucket. Callers use 0-6; the cap only bounds the key. - private const int MaxPriorityBucket = 99; - - /// Width of the zero-padded bucket key ("P04"), so the index row key splits at a fixed offset. - private const int BucketKeyLength = 3; - - /// - /// Where the schema marker lives. '$' is legal in a key and no run name starts with it — run names - /// are "{OrchestratorName}-{tenant}-{guid}" or "{OrchestratorName}_{...}". - /// - private const string SchemaPartition = "$schema"; - private const string SchemaRowKey = "queue-index"; - - /// - /// Current on-disk schema version, applied once per storage account by : - /// 1 — the run index exists (see the RUN INDEX note above). - /// 2 — queue RowKeys are deterministic per (run, task) — {run}|{task} — instead of - /// time-prefixed, so re-dispatching a task UPDATES its row instead of writing a second one - /// (the duplicate-execution class). Enqueue time moves to the QueuedUtc property. - /// A single forward migration takes any older account straight to this version; there is no - /// backward-compatible dual-read — after the migration only the new key scheme is used. - /// - private const int SchemaVersion = 2; - - /// - /// Rows buffered before the backfill flushes. Bounds peak memory on a very large queue — the - /// instance that motivated this was at 85% of a 2398MB heap cap before the backfill even started, - /// and buffering 743,000 rows to group them perfectly would have been the thing that OOMed it. - /// Rows for one run are written adjacently, so a window this size still groups them in practice. - /// - private const int BackfillFlushThreshold = 5_000; - - public JobQueueStore(ILogger logger, CraftSettings settings, ICraftTableStore store) - { - _logger = logger; - _store = store; - _queueTable = $"{settings.Orchestrator.TablePrefix}Queue"; - _indexTable = $"{settings.Orchestrator.TablePrefix}QueueIndex"; - } - - public async Task InitializeAsync(CancellationToken ct = default) - { - if (_initialized) return; - await _store.EnsureTableAsync(_queueTable, ct); - await _store.EnsureTableAsync(_indexTable, ct); - await MigrateSchemaAsync(ct); - _initialized = true; - } - - internal static string Bucket(int priority) => - "P" + Math.Clamp(priority, 0, MaxPriorityBucket).ToString("D2", CultureInfo.InvariantCulture); - - /// - /// The queue RowKey: deterministic per (run, task), so re-dispatching a task upserts its one row - /// instead of writing a second, time-prefixed one — the cause of a task running 2-6× (schema v2). - /// - /// Both components are escaped so the key is legal and the '|' separator is unambiguous; the row - /// itself carries RunName/TaskId as properties, so nothing needs to parse the key back apart. - /// - /// The trade: within a priority bucket Azure now returns rows in run|task order rather than - /// oldest-first. Priority still orders across buckets, and a fan-out enqueues its tasks together, so - /// sub-priority FIFO fairness is the only thing given up — cheap next to never running a task twice. - /// - internal static string BuildRowKey(string runName, string taskId) => - $"{EscapeKeyComponent(runName)}|{EscapeKeyComponent(taskId)}"; - - /// - /// Escape a run or task id for use inside a queue RowKey: the Azure-illegal key characters plus '|' - /// (the separator) and '%' (the escape marker itself), percent-encoded. Reversible and injective, so - /// distinct (run, task) pairs never collide, and ordinary names pass through untouched. - /// - private static string EscapeKeyComponent(string value) - { - var needsEscape = false; - foreach (var c in value) - { - if (c is '/' or '\\' or '#' or '?' or '%' or '|' || char.IsControl(c)) { needsEscape = true; break; } - } - if (!needsEscape) return value; - - var sb = new System.Text.StringBuilder(value.Length + 8); - foreach (var c in value) - { - if (c is '/' or '\\' or '#' or '?' or '%' or '|' || char.IsControl(c)) - sb.Append('%').Append(((int)c).ToString("X2", CultureInfo.InvariantCulture)); - else - sb.Append(c); - } - return sb.ToString(); - } - - /// The enqueue time embedded in a legacy (v1) row key — {ticks:D19}-{run}-{task} — or - /// null for a key that is not in that format. Used only by the one-time migration to carry the old - /// key's timestamp into the new row's QueuedUtc property. - internal static DateTime? ParseLegacyQueuedUtc(string rowKey) - { - if (rowKey.Length < 20 || rowKey[19] != '-') return null; - if (!long.TryParse(rowKey.AsSpan(0, 19), NumberStyles.None, CultureInfo.InvariantCulture, out var ticks)) - return null; - if (ticks <= 0 || ticks > DateTime.MaxValue.Ticks) return null; - return new DateTime(ticks, DateTimeKind.Utc); - } - - /// - /// A run name as an index partition key. Azure Tables rejects '/', '\', '#', '?' and control - /// characters in a key, and a run name carries a user-supplied scheduled-task name — "Alert on - /// Huntress Rogue Apps detected" is a real one, and nothing stops the next one containing a slash - /// or a question mark. Percent-escaping is reversible and leaves ordinary names untouched, so the - /// table stays readable in the portal, which is where anyone debugging this will be looking. - /// - internal static string IndexPartition(string runName) - { - var needsEscape = false; - foreach (var c in runName) - { - if (c is '/' or '\\' or '#' or '?' or '%' || char.IsControl(c)) { needsEscape = true; break; } - } - if (!needsEscape) return runName; - - var sb = new System.Text.StringBuilder(runName.Length + 8); - foreach (var c in runName) - { - // '%' first, or the escape sequences themselves would be ambiguous. - if (c is '/' or '\\' or '#' or '?' or '%' || char.IsControl(c)) - sb.Append('%').Append(((int)c).ToString("X2", CultureInfo.InvariantCulture)); - else - sb.Append(c); - } - return sb.ToString(); - } - - /// Index row key: the bucket (fixed width) plus the queue row key it points at. - internal static string IndexRowKey(string bucket, string queueRowKey) => $"{bucket}|{queueRowKey}"; - - /// The inverse of . Split at a fixed offset — the bucket is always - /// three characters, so a '|' inside the queue row key cannot confuse this. - internal static (string Bucket, string QueueRowKey)? SplitIndexRowKey(string indexRowKey) - { - if (indexRowKey.Length < BucketKeyLength + 2 || indexRowKey[BucketKeyLength] != '|') return null; - return (indexRowKey[..BucketKeyLength], indexRowKey[(BucketKeyLength + 1)..]); - } - - private static StoreRow IndexRow(string runName, string taskId, string bucket, string queueRowKey) => - new(IndexPartition(runName), IndexRowKey(bucket, queueRowKey)) - { - Properties = { ["TaskId"] = taskId, ["RunName"] = runName } - }; - - /// Add one task to the queue. Idempotent for a given (queuedUtc, run, task). - /// - /// Queue row first, index row second. The queue row is what makes the task actually run; the index - /// only accelerates lookups. If the process dies between the two the task still executes, and the - /// missing index entry is repaired by the next enqueue of the same (queuedUtc, run, task), which - /// rewrites both keys unchanged. The other order would leave the index claiming a task is queued - /// when no row exists — the orphan re-drive trusts the index, would decline to re-queue, and the - /// run would sit Pending with nothing running. - /// - public async Task EnqueueAsync(string runName, string taskId, int priority, DateTime queuedUtc, - CancellationToken ct = default) - { - var bucket = Bucket(priority); - var rowKey = BuildRowKey(runName, taskId); - - await _store.UpsertAsync(_queueTable, new StoreRow(bucket, rowKey) - { - Properties = - { - ["RunName"] = runName, - ["TaskId"] = taskId, - ["Priority"] = priority, - ["Owner"] = "", - ["LeaseUntil"] = (DateTimeOffset?)null, - // Enqueue time is a property now that it is no longer in the key (schema v2), so age and - // status reporting keep working while the key stays deterministic per (run, task). - ["QueuedUtc"] = new DateTimeOffset(queuedUtc, TimeSpan.Zero), - } - }, ct); - - await _store.UpsertAsync(_indexTable, IndexRow(runName, taskId, bucket, rowKey), ct); - - WakePump(); - } - - /// Queue many tasks for one run. Chunked by the caller's priority into per-bucket batches. - /// - /// The index rows for one run all share a partition, so however many buckets the tasks span the - /// index costs exactly one transaction. Ordering is as . - /// - public async Task EnqueueBatchAsync(string runName, IReadOnlyList<(string TaskId, int Priority)> tasks, - DateTime queuedUtc, CancellationToken ct = default) - { - var indexRows = new List(tasks.Count); - - foreach (var byBucket in tasks.GroupBy(t => Bucket(t.Priority))) - { - var queuedOffset = new DateTimeOffset(queuedUtc, TimeSpan.Zero); - var rows = byBucket.Select(t => new StoreRow(byBucket.Key, BuildRowKey(runName, t.TaskId)) - { - Properties = - { - ["RunName"] = runName, - ["TaskId"] = t.TaskId, - ["Priority"] = t.Priority, - ["Owner"] = "", - ["LeaseUntil"] = (DateTimeOffset?)null, - ["QueuedUtc"] = queuedOffset, - } - }).ToList(); - - await _store.UpsertBatchAsync(_queueTable, byBucket.Key, rows, ct); - - indexRows.AddRange(byBucket.Select(t => - IndexRow(runName, t.TaskId, byBucket.Key, BuildRowKey(runName, t.TaskId)))); - } - - if (indexRows.Count > 0) - await _store.UpsertBatchAsync(_indexTable, IndexPartition(runName), indexRows, ct); - - if (tasks.Count > 0) WakePump(); - } - - /// A queued task this worker now owns, with the row key needed to release it. - public sealed record ClaimedJob(string RunName, string TaskId, int Priority, string Bucket, string RowKey); - - /// - /// Claim up to of the highest-priority, oldest queued tasks for - /// , for . - /// - /// One read plus one conditional transaction. The read stops as soon as it has a batch, so it costs - /// a single page however deep the queue is; the transaction covers one bucket, because that is the - /// unit a backend transaction can span. - /// - /// Returns empty when there is nothing claimable, and ALSO when another worker won the race — the - /// caller simply tries again rather than forcing the write, which is what stops two workers running - /// the same task. - /// - public async Task> ClaimBatchAsync(string owner, int max, TimeSpan leaseFor, - CancellationToken ct = default) - { - if (max <= 0) return []; - - var now = DateTimeOffset.UtcNow; - var candidates = new List(max); - string? bucket = null; - - // Ordered partition-then-row, so this walks highest priority first, oldest first within it. - // - // The filter is the same predicate as IsClaimable, pushed to the service so a backlog is not - // paged to the client on every pump tick just to find the few free rows at its head. It is an - // optimisation ONLY — a store that ignores it still returns everything — so IsClaimable below - // stays as the authority. Nothing here may assume the filter was applied. - await foreach (var row in _store.QueryTableAsync(_queueTable, ClaimableFilter(now), ct)) - { - if (!IsClaimable(row, now)) continue; - - // A transaction cannot span partitions, so the batch is whatever the top bucket offers. - bucket ??= row.PartitionKey; - if (row.PartitionKey != bucket) break; - - candidates.Add(row); - if (candidates.Count == max) break; - } - - if (candidates.Count == 0) return []; - - var leaseUntil = now.Add(leaseFor); - foreach (var row in candidates) - { - row["Owner"] = owner; - row["LeaseUntil"] = leaseUntil; - } - - if (!await _store.TryReplaceBatchAsync(_queueTable, bucket!, candidates, ct)) - { - // Someone else got there first, or a row changed underneath us. Not an error: the caller - // retries and takes whatever is genuinely free. - _logger.LogDebug("[JobQueue] Claim of {Count} from {Bucket} lost the race", candidates.Count, bucket); - return []; - } - - return candidates.Select(r => new ClaimedJob( - r.GetString("RunName") ?? "", - r.GetString("TaskId") ?? "", - r.GetInt32("Priority") ?? 0, - r.PartitionKey, - r.RowKey)).ToList(); - } - - /// - /// Claimable means unowned, or owned under a lease that has expired. - /// - /// Lease expiry is what replaces the age-based re-drive: a worker that dies holding a claim gives the - /// task back on its own, without anything having to notice the worker is gone. - /// - private static bool IsClaimable(StoreRow row, DateTimeOffset now) - { - if (string.IsNullOrEmpty(row.GetString("Owner"))) return true; - - var lease = row.GetDateTimeOffset("LeaseUntil"); - return lease == null || lease <= now; - } - - /// - /// The server-side half of : free rows, plus rows whose lease has run out. - /// - /// Enqueue writes Owner as an empty string and LeaseUntil as null, and a null property is simply - /// absent from an Azure Tables entity — so a free row is matched by the Owner clause rather than by - /// anything about LeaseUntil, which is why this does not try to express "LeaseUntil is null". - /// - /// One case is deliberately narrower than IsClaimable: a row with an Owner but NO LeaseUntil, which - /// IsClaimable treats as claimable, is not matched here. No write path produces one — Owner and - /// LeaseUntil are always set together, by the claim, the renewal and the release alike — so this - /// costs nothing in practice, and IsClaimable keeps the defensive reading for anything that - /// arrives through the unfiltered path. - /// - private static string ClaimableFilter(DateTimeOffset now) => - $"Owner eq '' or LeaseUntil lt datetime'{now.UtcDateTime:yyyy-MM-ddTHH:mm:ss.fffffffZ}'"; - - /// Remove a finished task from the queue. A missing row is not an error — it is the normal - /// result of a retry after the removal already landed. - /// - /// Index row first, mirroring the enqueue rationale from the other side. A crash between the two - /// leaves a queue row for a task that has finished; it gets claimed once more and the resolver drops - /// it as a stale descriptor, which is already a handled path. Deleting the queue row first would - /// instead leave the index advertising queued work that does not exist, which stalls the run. - /// - public async Task RemoveAsync(ClaimedJob job, CancellationToken ct = default) - { - await _store.DeleteAsync(_indexTable, IndexPartition(job.RunName), IndexRowKey(job.Bucket, job.RowKey), ct); - await _store.DeleteAsync(_queueTable, job.Bucket, job.RowKey, ct); - } - - /// - /// Remove many finished tasks at once. The pump releases a whole claimed batch per cycle, so this - /// turns what was 2 point deletes per task (index + queue, one each) into - /// one transaction per partition: index rows share a run's partition, queue rows share a bucket. - /// - /// Ordering matches at the batch level: ALL index rows first, then the - /// queue rows. A crash in between leaves queue rows whose tasks are finished — claimed once more and - /// dropped as stale descriptors, an already-handled path — whereas deleting the queue rows first - /// would leave the index advertising work that no longer exists and stall those runs. - /// - public async Task RemoveBatchAsync(IReadOnlyList jobs, CancellationToken ct = default) - { - if (jobs.Count == 0) return; - - foreach (var byRun in jobs.GroupBy(j => IndexPartition(j.RunName))) - await _store.DeleteBatchAsync(_indexTable, byRun.Key, - byRun.Select(j => IndexRowKey(j.Bucket, j.RowKey)).ToList(), ct); - - foreach (var byBucket in jobs.GroupBy(j => j.Bucket)) - await _store.DeleteBatchAsync(_queueTable, byBucket.Key, - byBucket.Select(j => j.RowKey).ToList(), ct); - } - - /// - /// This run's index rows. One single-partition read — the operation every run-scoped method below - /// used to perform as a full-table scan. - /// - private async Task> - ReadIndexAsync(string runName, CancellationToken ct) - { - var entries = new List<(string, string, string, string)>(); - - await foreach (var row in _store.QueryPartitionAsync(_indexTable, IndexPartition(runName), ct)) - { - var split = SplitIndexRowKey(row.RowKey); - if (split == null) continue; - - var taskId = row.GetString("TaskId"); - if (string.IsNullOrEmpty(taskId)) continue; - - entries.Add((taskId, split.Value.Bucket, split.Value.QueueRowKey, row.RowKey)); - } - - return entries; - } - - /// - /// Extend the lease on jobs still in flight. One transaction per bucket, so a full buffer costs one - /// round-trip rather than one per job. Returns false if any renewal was rejected, which means the - /// lease had already lapsed and the work may have been taken. - /// - public async Task RenewAsync(IReadOnlyList jobs, string owner, TimeSpan leaseFor, - CancellationToken ct = default) - { - if (jobs.Count == 0) return true; - - var leaseUntil = DateTimeOffset.UtcNow.Add(leaseFor); - var ok = true; - - foreach (var group in jobs.GroupBy(j => j.Bucket)) - { - var rows = new List(); - foreach (var job in group) - { - var row = await _store.GetAsync(_queueTable, job.Bucket, job.RowKey, ct); - // Gone means finished and removed; still ours means renewable. Anything else is not ours. - if (row == null) continue; - if (row.GetString("Owner") != owner) { ok = false; continue; } - - row["LeaseUntil"] = leaseUntil; - rows.Add(row); - } - - if (rows.Count > 0 && !await _store.TryReplaceBatchAsync(_queueTable, group.Key, rows, ct)) - ok = false; - } - - return ok; - } - - /// Drop every queued row for a run — used when a run is cancelled or cleaned up. - public async Task RemoveRunAsync(string runName, CancellationToken ct = default) - { - foreach (var e in await ReadIndexAsync(runName, ct)) - await _store.DeleteAsync(_queueTable, e.Bucket, e.QueueRowKey, ct); - - // One call, and it also takes any entry whose queue row was already gone. - await _store.DeletePartitionAsync(_indexTable, IndexPartition(runName), ct); - } - - /// - /// Empty the durable queue: delete every queue row and every index row (keeping only the schema - /// marker). Returns the number of queue rows removed. - /// - /// A maintenance/reset primitive. It drops the BACKLOG, not in-flight work — a row a worker is already - /// running finishes, and the pump's later removal of it simply 404s. Tasks still Pending in the - /// orchestrator's own tables can be re-driven onto the queue by recovery, so pair this with cancelling - /// the runs when the intent is to STOP work rather than to clear a wedged or corrupted queue. - /// Deletes are streamed in bounded windows, so this holds a fixed amount of memory on any queue size. - /// - public async Task ClearAllAsync(CancellationToken ct = default) - { - await InitializeAsync(ct); - - var removed = await ClearTableAsync(_queueTable, keepSchema: false, ct); - await ClearTableAsync(_indexTable, keepSchema: true, ct); - - _logger.LogWarning("[JobQueue] Durable queue cleared — {Count} queue row(s) removed", removed); - return removed; - } - - /// Delete every row of one table (optionally sparing the schema marker), batched per - /// partition and flushed in bounded windows so a huge table never lands in memory at once. - private async Task ClearTableAsync(string table, bool keepSchema, CancellationToken ct) - { - var removed = 0; - var pending = new Dictionary>(StringComparer.Ordinal); - var buffered = 0; - - async Task FlushAsync() - { - foreach (var (partition, keys) in pending) - await _store.DeleteBatchAsync(table, partition, keys, ct); - removed += buffered; - pending.Clear(); - buffered = 0; - } - - await foreach (var row in _store.QueryTableAsync(table, ct)) - { - if (keepSchema && row.PartitionKey == SchemaPartition) continue; - if (!pending.TryGetValue(row.PartitionKey, out var keys)) pending[row.PartitionKey] = keys = []; - keys.Add(row.RowKey); - if (++buffered >= BackfillFlushThreshold) await FlushAsync(); - } - - await FlushAsync(); - return removed; - } - - /// - /// Hand back every claim on a run's rows, making them immediately claimable again. Returns how many - /// were released. - /// - /// For crash recovery only, where "this run was interrupted" already means the process that held - /// these claims is gone. Without it a crash strands the run for up to the full lease: the rows are - /// owned with a live LeaseUntil, so nothing can claim them, while re-dispatch correctly declines to - /// write duplicates for tasks that already have rows. Seen on a killed 140-task fanout — 12 tasks sat - /// Pending with 0 running for the remainder of a 30 minute lease. - /// - /// Rows are updated in place (same PartitionKey/RowKey), so this frees the existing row rather than - /// adding another one. - /// - public async Task ReleaseRunClaimsAsync(string runName, CancellationToken ct = default) - { - var released = 0; - - foreach (var e in await ReadIndexAsync(runName, ct)) - { - var row = await _store.GetAsync(_queueTable, e.Bucket, e.QueueRowKey, ct); - if (row == null) continue; // finished and removed - if (string.IsNullOrEmpty(row.GetString("Owner"))) continue; // already free - - row["Owner"] = ""; - row["LeaseUntil"] = (DateTimeOffset?)null; - await _store.UpsertAsync(_queueTable, row, ct); - released++; - } - - // Freed claims are claimable again — wake the pump to pick them up rather than waiting for the - // recovery-path re-drive on its own timer. - if (released > 0) WakePump(); - - return released; - } - - /// A queued row as the status APIs see it: identity, priority, age and claim state. - public sealed record QueuedRow(string RunName, string TaskId, int Priority, DateTime QueuedUtc, - bool Claimed, string Owner, string Bucket, string RowKey); - - /// - /// A row's enqueue time: the QueuedUtc property (schema v2), falling back to the timestamp a - /// legacy v1 key was built from, then to now. For age/status reporting only. - /// - private static DateTime QueuedUtcOf(StoreRow row) => - row.GetDateTimeOffset("QueuedUtc")?.UtcDateTime - ?? ParseLegacyQueuedUtc(row.RowKey) - ?? DateTime.UtcNow; - - /// - /// Every row currently in the queue, in storage order (highest priority bucket first, oldest first - /// within it). This is the durable backlog the status APIs report — the in-memory JobManager only - /// ever holds a worker-pool-sized buffer of it. - /// - /// Claimed means owned under a live lease, i.e. buffered or running on some instance; everything - /// else is waiting for a pump to take it. One unfiltered scan, so the cost is proportional to the - /// backlog — callers are expected to cache the result rather than call this per poll. - /// - public async Task> ListQueuedAsync(CancellationToken ct = default) - { - var now = DateTimeOffset.UtcNow; - var rows = new List(); - - await foreach (var row in _store.QueryTableAsync(_queueTable, ct)) - { - rows.Add(new QueuedRow( - row.GetString("RunName") ?? "", - row.GetString("TaskId") ?? "", - row.GetInt32("Priority") ?? 0, - QueuedUtcOf(row), - !IsClaimable(row, now), - row.GetString("Owner") ?? "", - row.PartitionKey, - row.RowKey)); - } - - return rows; - } - - /// - /// Remove every queue row for one task, regardless of claim state. Returns how many were removed. - /// Used by the durable cancel path — the caller must have already marked the task terminal in the - /// run graph, or the orphan re-drive sees a Pending task with no row and puts one straight back. - /// - public async Task RemoveTaskAsync(string runName, string taskId, CancellationToken ct = default) - { - var removed = 0; - var partition = IndexPartition(runName); - - foreach (var e in await ReadIndexAsync(runName, ct)) - { - if (e.TaskId != taskId) continue; - - await _store.DeleteAsync(_indexTable, partition, e.IndexRowKey, ct); - await _store.DeleteAsync(_queueTable, e.Bucket, e.QueueRowKey, ct); - removed++; - } - - return removed; - } - - /// - /// Move a task's queue rows to a new priority bucket, keeping their enqueue timestamp so the task - /// keeps its place in line within the new priority. Returns how many rows moved. - /// - /// Delete-then-add, in that order: a crash in between loses the row, which the orphan re-drive - /// repairs by re-queueing the task. The other order leaves TWO claimable rows for one task, and a - /// duplicated row is executed once per copy — that is the failure mode this queue exists to prevent. - /// - public async Task ReprioritizeTaskAsync(string runName, string taskId, int newPriority, - CancellationToken ct = default) - { - var moved = 0; - var now = DateTimeOffset.UtcNow; - var partition = IndexPartition(runName); - var toMove = new List(); - - foreach (var e in await ReadIndexAsync(runName, ct)) - { - if (e.TaskId != taskId) continue; - if (e.Bucket == Bucket(newPriority)) continue; // already there - - var row = await _store.GetAsync(_queueTable, e.Bucket, e.QueueRowKey, ct); - if (row == null) continue; - - // A claimed row is already buffered on some instance and about to run — re-adding it - // unclaimed would create a second runnable copy of the task. Leave it be. - if (!IsClaimable(row, now)) continue; - - toMove.Add(row); - } - - foreach (var row in toMove) - { - await _store.DeleteAsync(_indexTable, partition, IndexRowKey(row.PartitionKey, row.RowKey), ct); - await _store.DeleteAsync(_queueTable, row.PartitionKey, row.RowKey, ct); - - var queuedUtc = QueuedUtcOf(row); - await EnqueueAsync(runName, taskId, newPriority, queuedUtc, ct); - moved++; - } - - return moved; - } - - /// - /// The task ids this run still has rows for, claimed or not. - /// - /// This is what tells a re-drive the difference between a task that is merely WAITING — Pending in - /// the run graph, sitting in this queue, not yet claimed by the pump — and one whose row is - /// genuinely gone. Under the pump, waiting is the normal state of a backlog: a 124-task run against - /// eight workers has most of its tasks Pending and absent from the JobManager for minutes at a time. - /// Treating that as orphaned re-queues the whole backlog on a timer, and because a RowKey is - /// prefixed with the enqueue timestamp, each pass adds a SECOND row for the same task rather than - /// updating the first — so the task is claimed and executed once per copy. - /// - public async Task> GetQueuedTaskIdsAsync(string runName, CancellationToken ct = default) - { - var ids = new HashSet(StringComparer.Ordinal); - - // Index only — this never touches the queue table. It is the hottest of the run-scoped reads - // (once per run per re-drive) and was the single largest source of the scan volume. - await foreach (var row in _store.QueryPartitionAsync(_indexTable, IndexPartition(runName), ct)) - { - var taskId = row.GetString("TaskId"); - if (!string.IsNullOrEmpty(taskId)) ids.Add(taskId); - } - - return ids; - } - - /// - /// Of (all belonging to ), the ones the pump - /// can still dispatch: they have a queue row that EXISTS and that the claim filter will match — now, - /// because it is free, or later, because it holds a lease that will lapse. - /// - /// This is the queue-table counterpart to , which answers purely - /// from the index. The index is what makes "does this run still have queued work" a single-partition - /// read, but it can OUTLIVE the queue rows it points at, and then it lies: it reports a task queued - /// that no pump will ever run. That divergence is not hypothetical — - /// - /// a removal deletes the index row first, so a crash in between (or a - /// that deleted queue rows before its index partition) can leave the - /// opposite; - /// a run left Pending under a build that dispatched into memory rather than this - /// queue re-enters here with index rows and no queue rows; - /// a row owned with NO LeaseUntil is excluded by - /// forever, so it sits with an index entry advertising it. - /// - /// A task in any of those states is invisible to the pump AND reported "queued" by the index, so the - /// re-drive that trusts the index never re-enqueues it and the run stalls indefinitely with it, its - /// watchdog never firing. Verifying against the queue table costs one point read per id, so the caller - /// passes a SMALL candidate set (the re-drive's aged-Pending tasks), never the whole run. - /// - public async Task> GetDispatchableTaskIdsAsync( - string runName, IReadOnlyCollection taskIds, CancellationToken ct = default) - { - var result = new HashSet(StringComparer.Ordinal); - if (taskIds.Count == 0) return result; - - var wanted = taskIds as HashSet ?? new HashSet(taskIds, StringComparer.Ordinal); - var now = DateTimeOffset.UtcNow; - - // The index carries the bucket + queue row key to address each row directly — one partition read, - // then a point read only for the ids the caller asked about. - foreach (var e in await ReadIndexAsync(runName, ct)) - { - if (!wanted.Contains(e.TaskId)) continue; - - var row = await _store.GetAsync(_queueTable, e.Bucket, e.QueueRowKey, ct); - if (row == null) continue; // index points at a queue row that is gone — a ghost - - // Owned with no lease is what the server-side claim filter cannot match (it is neither - // Owner eq '' nor LeaseUntil lt now), so the pump would never dispatch it however long it - // waits — a ghost as surely as a missing row. A free row, or one under a lease live or - // lapsed, the pump will get. - if (!string.IsNullOrEmpty(row.GetString("Owner")) && row.GetDateTimeOffset("LeaseUntil") == null) - continue; - - result.Add(e.TaskId); - } - - return result; - } - - /// - /// Bring the queue tables up to , once per storage account. The marker row - /// written at the end is checked first, so every later start is a single point read. - /// - /// v1 built the run index; v2 additionally re-keys every queue row to the deterministic - /// {run}|{task} scheme () and moves the enqueue timestamp into the - /// QueuedUtc property. One forward pass takes any older account straight to the current - /// version — there is no dual-read, and after the pass only the new key scheme is used. - /// - /// Rows are rewritten new-key-first, old-key-deleted-after, so a crash mid-pass leaves the marker - /// unset and the next start finishes the job (a row already in the new scheme is re-written harmlessly - /// and not deleted). The pump awaits before it claims — and every - /// enqueue path calls it too — so no row is ever claimed or written while the migration is only - /// half-applied. - /// - /// Awaited by InitializeAsync rather than backgrounded: the run-scoped reads and the deterministic - /// keys are only correct once it has finished. - /// - private async Task MigrateSchemaAsync(CancellationToken ct) - { - var marker = await _store.GetAsync(_indexTable, SchemaPartition, SchemaRowKey, ct); - if ((marker?.GetInt32("Version") ?? 0) >= SchemaVersion) return; - - var started = DateTime.UtcNow; - _logger.LogInformation( - "[JobQueue] Migrating queue schema to v{Version} — one full pass over {Table}", SchemaVersion, _queueTable); - - var newQueue = new Dictionary>(StringComparer.Ordinal); // by bucket - var newIndex = new Dictionary>(StringComparer.Ordinal); // by run partition - var oldQueue = new Dictionary>(StringComparer.Ordinal); // bucket -> old row keys - var oldIndex = new Dictionary>(StringComparer.Ordinal); // run partition -> old index keys - var buffered = 0; - var migrated = 0; - var rekeyed = 0; - var skipped = 0; - - static void AddRow(Dictionary> map, string key, StoreRow row) - { - if (!map.TryGetValue(key, out var list)) map[key] = list = []; - list.Add(row); - } - static void AddKey(Dictionary> map, string key, string rowKey) - { - if (!map.TryGetValue(key, out var list)) map[key] = list = []; - list.Add(rowKey); - } - - async Task FlushAsync() - { - // New rows first, so a crash before the deletes leaves BOTH and the re-run converges — never - // the index advertising a queue row that no longer exists. - foreach (var (bucket, rows) in newQueue) - await _store.UpsertBatchAsync(_queueTable, bucket, rows, ct); - foreach (var (partition, rows) in newIndex) - await _store.UpsertBatchAsync(_indexTable, partition, rows, ct); - foreach (var (bucket, keys) in oldQueue) - await _store.DeleteBatchAsync(_queueTable, bucket, keys, ct); - foreach (var (partition, keys) in oldIndex) - await _store.DeleteBatchAsync(_indexTable, partition, keys, ct); - - migrated += buffered; - newQueue.Clear(); newIndex.Clear(); oldQueue.Clear(); oldIndex.Clear(); - buffered = 0; - } - - await foreach (var row in _store.QueryTableAsync(_queueTable, ct)) - { - var runName = row.GetString("RunName"); - var taskId = row.GetString("TaskId"); - if (string.IsNullOrEmpty(runName) || string.IsNullOrEmpty(taskId)) { skipped++; continue; } - - var bucket = row.PartitionKey; - var newKey = BuildRowKey(runName, taskId); - var runPartition = IndexPartition(runName); - - // The new-scheme row: same bucket + claim state + priority, key deterministic, enqueue time as - // a property (from the row, or the legacy key, or now). - AddRow(newQueue, bucket, new StoreRow(bucket, newKey) - { - Properties = - { - ["RunName"] = runName, - ["TaskId"] = taskId, - ["Priority"] = row.GetInt32("Priority") ?? 0, - ["Owner"] = row.GetString("Owner") ?? "", - ["LeaseUntil"] = row.GetDateTimeOffset("LeaseUntil"), - ["QueuedUtc"] = new DateTimeOffset(QueuedUtcOf(row), TimeSpan.Zero), - } - }); - AddRow(newIndex, runPartition, IndexRow(runName, taskId, bucket, newKey)); - - // Delete the legacy row + index entry, UNLESS its key is already the new scheme (a re-run over - // already-migrated rows just re-writes them — deleting would drop what we just wrote). - if (row.RowKey != newKey) - { - rekeyed++; - AddKey(oldQueue, bucket, row.RowKey); - AddKey(oldIndex, runPartition, IndexRowKey(bucket, row.RowKey)); - } - buffered++; - - if (buffered >= BackfillFlushThreshold) - { - await FlushAsync(); - _logger.LogInformation("[JobQueue] Schema migration: {Migrated:N0} rows so far", migrated); - } - } - - await FlushAsync(); - - await _store.UpsertAsync(_indexTable, new StoreRow(SchemaPartition, SchemaRowKey) - { - Properties = - { - ["Version"] = SchemaVersion, - ["BuiltUtc"] = new DateTimeOffset(started, TimeSpan.Zero), - ["RowsIndexed"] = migrated, - } - }, ct); - - _logger.LogInformation( - "[JobQueue] Queue schema at v{Version}: {Migrated:N0} row(s) processed, {Rekeyed:N0} re-keyed, in {Seconds:N0}s{Skipped} — this will not run again", - SchemaVersion, migrated, rekeyed, (DateTime.UtcNow - started).TotalSeconds, - skipped > 0 ? $", {skipped:N0} malformed row(s) skipped" : ""); - } -} diff --git a/Services/Storage/OrchestratorCleanupResult.cs b/Services/Storage/OrchestratorCleanupResult.cs deleted file mode 100644 index 959bb54..0000000 --- a/Services/Storage/OrchestratorCleanupResult.cs +++ /dev/null @@ -1,15 +0,0 @@ -namespace Craft.Storage; - -/// -/// What one pass removed, so the caller can -/// log it and take the follow-up that is not the store's business (an abandoned run's queue rows). -/// -/// Run rows scanned. -/// Runs that had finished and were past retention. -/// Runs that had not finished, were not active, and had not been written to within retention. -/// Tasks/Results partitions with no Run row whose newest row was past retention. -public sealed record OrchestratorCleanupResult( - int RunsExamined, - IReadOnlyList ExpiredRuns, - IReadOnlyList AbandonedRuns, - int OrphanPartitionsRemoved); diff --git a/Services/Storage/OrchestratorRunSummary.cs b/Services/Storage/OrchestratorRunSummary.cs deleted file mode 100644 index dd21983..0000000 --- a/Services/Storage/OrchestratorRunSummary.cs +++ /dev/null @@ -1,12 +0,0 @@ -namespace Craft.Storage; - -/// -/// A run's identity and parentage, read from the run row alone — no task rows. -/// -/// Exists so startup can rebuild parent/child run links from one partition scan instead of calling -/// GetRunAsync per run, which loads every task of every run. -/// -/// Run name (the run row's RowKey). -/// Running | Completed | CompletedWithErrors | Failed. -/// Parent run, when this run was spawned by a task of another run. -public record OrchestratorRunSummary(string Name, string Status, string? ParentRunName); diff --git a/Services/Storage/OrchestratorTableStore.cs b/Services/Storage/OrchestratorTableStore.cs deleted file mode 100644 index 1e207fc..0000000 --- a/Services/Storage/OrchestratorTableStore.cs +++ /dev/null @@ -1,997 +0,0 @@ -using System.Runtime.CompilerServices; -using System.Text; -using System.Text.Json; -using Craft.Configuration; -using Craft.Orchestration; - -namespace Craft.Storage; - -/// -/// Typed CRUD wrapper over for orchestrator persistence. Manages three -/// logical tables: {prefix}Runs, {prefix}Tasks, {prefix}Results. Persists through the -/// abstraction. -/// -public class OrchestratorTableStore -{ - private readonly ILogger _logger; - private readonly ICraftTableStore _store; - private readonly string _runsTable; - private readonly string _tasksTable; - private readonly string _resultsTable; - private bool _initialized; - - private static readonly JsonSerializerOptions s_jsonOptions = new() - { - WriteIndented = false, - PropertyNamingPolicy = JsonNamingPolicy.CamelCase - }; - - public OrchestratorTableStore(ILogger logger, CraftSettings settings, ICraftTableStore store) - { - _logger = logger; - _store = store; - var prefix = settings.Orchestrator.TablePrefix; - _runsTable = $"{prefix}Runs"; - _tasksTable = $"{prefix}Tasks"; - _resultsTable = $"{prefix}Results"; - } - - /// Create the three tables if they do not exist. Called once on startup. - public async Task InitializeAsync() - { - if (_initialized) return; - - await _store.EnsureTableAsync(_runsTable); - await _store.EnsureTableAsync(_tasksTable); - await _store.EnsureTableAsync(_resultsTable); - _initialized = true; - - _logger.LogInformation("[OrchestratorStore] Tables initialized"); - } - - /// Upsert run metadata (without tasks — tasks are separate rows). - public async Task UpsertRunAsync(OrchestratorRun run) - { - await _store.UpsertAsync(_runsTable, BuildRunRow(run)); - } - - /// - /// Persist many run rows in as few round-trips as possible. - /// - /// Every run row shares the constant "Run" partition key, so N of them cost ceil(N/100) - /// transactions rather than N. Writing them one at a time inside a flush bounded by - /// StatusFlushTimeoutSeconds is what pushed real flushes past 30s and then past the 90s durable - /// barrier: every task waiting on its "Running" marker deferred, and after MaxDeferrals was - /// abandoned as Pending with nothing left to retry it. - /// - /// Returns the names of runs that did not persist, so the caller can requeue exactly those. - /// - public async Task> WriteRunStatusBatchAsync(IReadOnlyList runs, - CancellationToken ct = default) - { - if (runs.Count == 0) return []; - - var failed = new List(); - - // 100 is the Azure Table transaction ceiling. - foreach (var chunk in runs.Chunk(100)) - { - try - { - await _store.UpsertBatchAsync(_runsTable, "Run", chunk.Select(BuildRunRow).ToList(), ct); - } - catch (Exception ex) - { - // A transaction fails atomically, so one rejected row would discard the other 99. Retry - // the chunk row by row: the poison row is isolated and the rest still land. - _logger.LogWarning(ex, - "[OrchestratorStore] Run status batch of {Count} failed — retrying individually", chunk.Length); - - foreach (var run in chunk) - { - try { await _store.UpsertAsync(_runsTable, BuildRunRow(run), ct); } - catch (Exception single) - { - failed.Add(run.Name); - _logger.LogWarning(single, - "[OrchestratorStore] Run status write failed for {Run} — will retry", run.Name); - } - } - } - } - - return failed; - } - - private static StoreRow BuildRunRow(OrchestratorRun run) => new("Run", run.Name) - { - Properties = - { - ["Status"] = run.Status, - ["Priority"] = run.Priority, - ["StartedUtc"] = run.StartedUtc, - ["CompletedUtc"] = run.CompletedUtc, - ["TaskScriptName"] = run.TaskScriptName, - ["PostExecFunctionName"] = run.PostExecFunctionName, - ["PostExecParametersJson"] = run.PostExecParametersJson, - ["PostExecStatus"] = run.PostExecStatus, - ["PostExecAttemptCount"] = run.PostExecAttemptCount, - ["Reference"] = run.Reference, - ["ParentRunName"] = run.ParentRunName, - ["TaskCount"] = run.Tasks.Count, - // Persisted so a resumed sequential run keeps advancing one task at a time (0/1 — StoreRow has - // no bool reader). Absent on older rows reads as 0 = the fan-out default. - ["Sequential"] = run.Sequential ? 1 : 0 - } - }; - - /// Load a run and all its tasks. Returns null if the run does not exist. - public async Task GetRunAsync(string name) - { - var runRow = await _store.GetAsync(_runsTable, "Run", name); - if (runRow == null) return null; - - var run = new OrchestratorRun - { - Name = name, - Status = runRow.GetString("Status") ?? "Pending", - Priority = runRow.GetInt32("Priority") ?? 4, - StartedUtc = runRow.GetDateTimeOffset("StartedUtc")?.UtcDateTime ?? DateTime.UtcNow, - CompletedUtc = runRow.GetDateTimeOffset("CompletedUtc")?.UtcDateTime, - TaskScriptName = runRow.GetString("TaskScriptName"), - PostExecFunctionName = runRow.GetString("PostExecFunctionName"), - PostExecParametersJson = runRow.GetString("PostExecParametersJson"), - PostExecStatus = runRow.GetString("PostExecStatus"), - // Absent on rows written before this existed — 0 is the right reading of "never retried". - PostExecAttemptCount = runRow.GetInt32("PostExecAttemptCount") ?? 0, - // Both were in-memory only until now: a resumed run came back with a null Reference - // (so FindRunByReference could not see it) and a null ParentRunName (so its finalize - // never re-checked the parent). Absent on rows written before this existed. - Reference = runRow.GetString("Reference"), - ParentRunName = runRow.GetString("ParentRunName"), - Sequential = (runRow.GetInt32("Sequential") ?? 0) == 1 - }; - - var tasks = new List(); - await foreach (var taskRow in _store.QueryPartitionAsync(_tasksTable, name)) - { - // The remaining-count row shares this partition so it can be updated in the same transaction - // as a task completion. It is bookkeeping, not work — materializing it would give every run a - // phantom task that never completes, and no run would ever finalize. - if (taskRow.RowKey == CounterRowKey) continue; - - var parametersJson = taskRow.GetString("ParametersJson"); - Dictionary parameters; - try - { - parameters = !string.IsNullOrEmpty(parametersJson) - ? JsonSerializer.Deserialize>(parametersJson, s_jsonOptions) ?? [] - : []; - } - catch - { - parameters = []; - } - - tasks.Add(new OrchestratorTaskItem - { - Id = taskRow.RowKey, - Status = taskRow.GetString("Status") ?? "Pending", - Parameters = parameters, - AttemptCount = taskRow.GetInt32("AttemptCount") ?? 0, - LastError = taskRow.GetString("LastError"), - // Absent on rows written before per-task priority existed — null means "inherit the run's". - Priority = taskRow.GetInt32("Priority"), - Sequence = taskRow.GetInt32("Sequence") ?? 0, - CompletedUtc = taskRow.GetDateTimeOffset("CompletedUtc")?.UtcDateTime - }); - } - - run.Tasks = tasks; - return run; - } - - /// - /// Read one task's Parameters from the Tasks table (a single point read + deserialize), matching the - /// deserialization uses. For the pending-Parameters shedding path: the live - /// graph keeps the task object but drops its Parameters payload while it waits, and this rehydrates them - /// at dispatch. Null if the row or its ParametersJson is missing. - /// - public async Task?> GetTaskParametersAsync( - string runName, string taskId, CancellationToken ct = default) - { - var row = await _store.GetAsync(_tasksTable, runName, taskId, ct); - var parametersJson = row?.GetString("ParametersJson"); - if (string.IsNullOrEmpty(parametersJson)) return null; - try - { - return JsonSerializer.Deserialize>(parametersJson, s_jsonOptions) ?? []; - } - catch - { - return []; - } - } - - /// List all known run names. - public async Task> ListRunsAsync() - { - var names = new List(); - await foreach (var row in _store.QueryPartitionAsync(_runsTable, "Run")) - names.Add(row.RowKey); - return names; - } - - /// - /// List every run's identity and parentage without loading its tasks. - /// - /// One scan of the single "Run" partition. would answer the same question - /// but does a task-partition query per run, so using it to rebuild parent/child links at startup - /// would load every task of every run — the opposite of what recovery should cost. - /// - public async Task> ListRunSummariesAsync() - { - var summaries = new List(); - await foreach (var row in _store.QueryPartitionAsync(_runsTable, "Run")) - { - summaries.Add(new OrchestratorRunSummary( - row.RowKey, - row.GetString("Status") ?? "Pending", - row.GetString("ParentRunName"))); - } - return summaries; - } - - // ── Remaining-task counter ──────────────────────────────────────────────── - // - // How many of a run's tasks are not yet terminal, kept in storage rather than derived from the - // in-memory run graph, so finalization stops depending on one process holding the whole graph. - // Scanning instead is not an option: Azure Table cannot count server-side, so "is this run done" - // would be a 7,000-row read per check. - // - // The row lives in the TASKS table, in the run's own partition, and that placement is deliberate: a - // task's terminal write and the counter decrement can then share ONE conditional transaction where - // exactly-once matters most — the cancel-a-run path (CancelPendingTaskAsync) uses exactly that, so a - // cancel racing a task's real completion cannot decrement the counter twice. - // - // The hot fan-out path decrements SEPARATELY (DecrementRemainingAsync), by design: the batched status - // writer coalesces terminal writes, and decrementing once per group after it lands keeps the write - // batched. Idempotency there rests on the writer never re-sending a group that landed, backstopped by - // ReconcileRemainingAsync (a full-partition recount) whenever a decrement is lost. - // - // The reserved row key cannot collide with a task id — task ids are caller-supplied names like - // "CIPPStandard_IntuneTemplate__", never a control character. - private const string CounterRowKey = "!!run-counter"; - private const int CounterAttempts = 8; - - /// The three statuses that mean a task will not run again, and so has been counted. - private static bool IsTerminal(string? status) => - status is "Completed" or "Failed" or "Cancelled"; - - /// Seed the counter when a run is created. Idempotent, so re-seeding a resumed run is safe. - public Task InitRemainingAsync(string runName, int total, CancellationToken ct = default) => - _store.UpsertAsync(_tasksTable, new StoreRow(runName, CounterRowKey) - { - Properties = { ["Remaining"] = total, ["Total"] = total } - }, ct); - - /// How many tasks the STORE believes are outstanding, or null if the run has no counter row. - public async Task GetRemainingAsync(string runName, CancellationToken ct = default) - => (await _store.GetAsync(_tasksTable, runName, CounterRowKey, ct))?.GetInt32("Remaining"); - - /// - /// The run's counter row as (Remaining, Total), or null if the run has no counter. This is what the - /// status APIs use for a run's true size and durable progress — the in-memory JobManager only ever - /// sees the slice of a run that has been claimed onto this instance. - /// - public async Task<(int Remaining, int Total)?> GetCounterAsync(string runName, CancellationToken ct = default) - { - var row = await _store.GetAsync(_tasksTable, runName, CounterRowKey, ct); - if (row == null) return null; - return (row.GetInt32("Remaining") ?? 0, row.GetInt32("Total") ?? 0); - } - - /// - /// Subtract from a run's outstanding count, for the batch path. - /// - /// Exactly-once here rests on the caller: the coalescing status writer holds one pending write per - /// task and re-queues only the runs whose batch did NOT land, so a group that applied is never - /// re-sent. Call this only after a group has been written successfully, counting the terminal rows - /// in that group. Retries on a lost race so a concurrent decrement cannot swallow this one. - /// - public async Task DecrementRemainingAsync(string runName, int by, CancellationToken ct = default) - { - if (by <= 0) return await GetRemainingAsync(runName, ct); - - for (var attempt = 0; attempt < CounterAttempts; attempt++) - { - var counter = await _store.GetAsync(_tasksTable, runName, CounterRowKey, ct); - if (counter == null) return null; - - counter["Remaining"] = Math.Max(0, (counter.GetInt32("Remaining") ?? 0) - by); - - if (await _store.TryReplaceBatchAsync(_tasksTable, runName, [counter], ct)) - return (int)counter["Remaining"]!; - } - - _logger.LogWarning("[OrchestratorStore] Could not decrement remaining for {Run} by {By} after {Attempts} attempts", - runName, by, CounterAttempts); - return null; - } - - /// - /// Recount Remaining from the task rows the counter summarizes, and repair the counter row - /// when they disagree. - /// - /// A decrement that exhausts its retries is never re-applied — the terminal task rows landed but - /// the counter kept its old value, and from then on it permanently overstates the outstanding work - /// and finalize defers forever. The scan is the whole-partition read the counter exists to avoid, - /// which is why this runs only when a caller has evidence of drift (a lost decrement, a finalize - /// deferred repeatedly), never on the hot path. - /// - /// The reconciled outstanding count, or null if the run has no counter row or a concurrent - /// writer moved the counter mid-recount — the caller's next pass re-reads either way. - public async Task ReconcileRemainingAsync(string runName, CancellationToken ct = default) - { - var counter = await _store.GetAsync(_tasksTable, runName, CounterRowKey, ct); - if (counter == null) return null; - - var outstanding = 0; - await foreach (var row in _store.QueryPartitionAsync(_tasksTable, runName, ct)) - { - if (row.RowKey == CounterRowKey) continue; - if (!IsTerminal(row.GetString("Status"))) outstanding++; - } - - var stored = counter.GetInt32("Remaining") ?? 0; - if (stored == outstanding) return outstanding; - - // ETag-guarded: a decrement landing between the read above and this write rejects the - // replace, so a recount can never overwrite fresher progress with a stale count. - counter["Remaining"] = outstanding; - if (!await _store.TryReplaceBatchAsync(_tasksTable, runName, [counter], ct)) - return null; - - _logger.LogWarning( - "[OrchestratorStore] Reconciled remaining for {Run}: counter said {Stored}, task rows say {Actual}", - runName, stored, outstanding); - return outstanding; - } - - /// What a status-guarded cancel actually did, so the caller can keep its view honest. - public sealed record CancelWriteResult(bool Cancelled, string? CurrentStatus); - - /// - /// Write a task as Cancelled and decrement the run counter, but ONLY while storage still shows the - /// task Pending. This is the cancel-a-run primitive, and the guard is the point: between the - /// caller's read and this write, dispatch can move a task to Running — an unguarded terminal write - /// would clobber that, and the task's real completion would then decrement the counter a SECOND - /// time, letting the run finalize while work is still outstanding. - /// - /// A run without a counter row (pre-counter) gets the same status guard with a task-ETag-only write. - /// - /// - /// Whether the cancel landed, plus the status storage showed when it did not — so the caller can - /// correct its in-memory copy rather than believing a cancel that never happened. - /// - public async Task CancelPendingTaskAsync(string runName, OrchestratorTaskItem task, - CancellationToken ct = default) - { - for (var attempt = 0; attempt < CounterAttempts; attempt++) - { - var existing = await _store.GetAsync(_tasksTable, runName, task.Id, ct); - var currentStatus = existing?.GetString("Status"); - if (existing == null || currentStatus != "Pending") - return new CancelWriteResult(false, currentStatus); - - var guarded = new StoreRow(runName, task.Id) - { - ETag = existing.ETag, - Properties = BuildTaskRow(runName, task).Properties, - }; - - var counter = await _store.GetAsync(_tasksTable, runName, CounterRowKey, ct); - if (counter == null) - { - if (await _store.TryReplaceBatchAsync(_tasksTable, runName, [guarded], ct)) - return new CancelWriteResult(true, null); - continue; - } - - counter["Remaining"] = Math.Max(0, (counter.GetInt32("Remaining") ?? 0) - 1); - if (await _store.TryReplaceBatchAsync(_tasksTable, runName, [guarded, counter], ct)) - return new CancelWriteResult(true, null); - } - - _logger.LogWarning("[OrchestratorStore] Could not cancel {Task} in {Run} after {Attempts} attempts", - task.Id, runName, CounterAttempts); - return new CancelWriteResult(false, null); - } - - /// Upsert a single task row. - public Task UpsertTaskAsync(string runName, OrchestratorTaskItem task) => - _store.UpsertAsync(_tasksTable, BuildTaskRow(runName, task)); - - /// Batch upsert all tasks for a run (used at run creation). - public Task UpsertTaskBatchAsync(string runName, List tasks) => - _store.UpsertBatchAsync(_tasksTable, runName, tasks.Select(t => BuildTaskRow(runName, t)).ToList()); - - /// - /// Write a set of coalesced task-status transitions. Rows are grouped by run (partition) and handed - /// to the store, which applies each group as atomically as the backend allows and chunks to any - /// per-request limits. Used by the batched status writer; the large-result path - /// () is untouched. - /// - /// - /// The run names whose writes did NOT persist. Callers must retry these rather than dropping them — - /// they are terminal task states, and losing one means a finished task looks Pending forever. - /// An empty list means everything landed. - /// - public async Task> WriteTaskStatusBatchAsync(IReadOnlyList writes, - int maxConcurrency = 8, CancellationToken ct = default) - { - // Grouped by run because a batch has to share a partition key. Run these with bounded - // concurrency instead of one after another: the workload that broke this was ~600 runs of a - // single task each, so "batching" degenerated into hundreds of sequential round-trips inside - // one flush — with the single drain loop, and therefore every task waiting on its durable - // marker, blocked for the whole duration. - var groups = writes.GroupBy(w => w.RunName).ToList(); - if (groups.Count == 0) return []; - - var failed = new System.Collections.Concurrent.ConcurrentBag(); - using var gate = new SemaphoreSlim(Math.Max(1, maxConcurrency)); - - var tasks = groups.Select(async group => - { - await gate.WaitAsync(ct); - try - { - await _store.UpsertBatchAsync(_tasksTable, group.Key, group.Select(BuildTaskRow).ToList(), ct); - - // The group landed, so the tasks it just moved to a terminal state are now durably - // finished. Decrement once for them here rather than per task: this is the only place - // that knows a terminal write actually applied, and the writer never re-sends a group - // that did, which is what keeps the count honest. - var terminal = group.Count(w => IsTerminal(w.Status)); - if (terminal > 0 && await DecrementRemainingAsync(group.Key, terminal, ct) == null) - { - // Retry exhaustion here loses the decrement for good — the terminal rows above - // landed, so the writer will never re-send this group. Recount now rather than - // letting the counter overstate the run's outstanding work forever. - await ReconcileRemainingAsync(group.Key, ct); - } - } - catch (Exception ex) - { - // One run's failure must not discard the other 599. Record it and carry on; the caller - // requeues just this run. - failed.Add(group.Key); - _logger.LogWarning(ex, "[OrchestratorStore] Task status write failed for run {Run} ({Count} tasks) — will retry", - group.Key, group.Count()); - } - finally - { - gate.Release(); - } - }); - - await Task.WhenAll(tasks); - return failed.ToList(); - } - - private static StoreRow BuildTaskRow(string runName, OrchestratorTaskItem task) => new(runName, task.Id) - { - Properties = - { - ["Status"] = task.Status, - ["ParametersJson"] = JsonSerializer.Serialize(task.Parameters, s_jsonOptions), - ["AttemptCount"] = task.AttemptCount, - ["LastError"] = task.LastError, - ["Priority"] = task.Priority, - ["Sequence"] = task.Sequence, - ["CompletedUtc"] = task.CompletedUtc.HasValue - ? new DateTimeOffset(task.CompletedUtc.Value, TimeSpan.Zero) - : (DateTimeOffset?)null - } - }; - - private static StoreRow BuildTaskRow(TaskStatusWrite w) => new(w.RunName, w.TaskId) - { - Properties = - { - ["Status"] = w.Status, - ["ParametersJson"] = w.ParametersJson, - ["AttemptCount"] = w.AttemptCount, - ["LastError"] = w.LastError, - ["Priority"] = w.Priority, - ["Sequence"] = w.Sequence, - ["CompletedUtc"] = w.CompletedUtc.HasValue - ? new DateTimeOffset(w.CompletedUtc.Value, TimeSpan.Zero) - : (DateTimeOffset?)null - } - }; - - // ─── Result storage ─── - // Results can be large (50–150 MB for big runs). We chunk a result across multiple properties and, - // if needed, multiple rows in the same partition. These bounds are sized for Azure Table Storage - // (64 KiB/property, 1 MiB/entity); on a backend without those limits the chunking is simply - // unnecessary but still correct, and it keeps per-row payloads small (good for e.g. SQL packet size). - private const int MaxPropertyChars = 30_000; - private const int MaxEntityChars = 450_000; - - /// - /// Write a set of coalesced SMALL task results (single-property rows) to the Results table, grouped - /// by run (partition) and chunked to the backend's transaction limits by the store. The batched - /// counterpart to for results that fit one property; larger results - /// still go through StoreResultAsync's multi-row chunking path. - /// - /// - /// The run names whose results did NOT persist. The caller must retry these AND withhold those runs' - /// terminal task markers this flush, or a task could be counted done while its result is lost. - /// An empty list means everything landed. - /// - public async Task> WriteResultBatchAsync(IReadOnlyList results, - int maxConcurrency = 8, CancellationToken ct = default) - { - var groups = results.GroupBy(r => r.RunName).ToList(); - if (groups.Count == 0) return []; - - var failed = new System.Collections.Concurrent.ConcurrentBag(); - using var gate = new SemaphoreSlim(Math.Max(1, maxConcurrency)); - - var tasks = groups.Select(async group => - { - await gate.WaitAsync(ct); - try - { - var rows = group.Select(r => new StoreRow(r.RunName, r.TaskId) - { - Properties = { ["ResultJson"] = r.ResultJson } - }).ToList(); - await _store.UpsertBatchAsync(_resultsTable, group.Key, rows, ct); - } - catch (Exception ex) - { - // One run's failure must not discard the others. Record it; the caller requeues just this - // run's results and holds its terminal markers until they land together. - failed.Add(group.Key); - _logger.LogWarning(ex, "[OrchestratorStore] Result write failed for run {Run} ({Count} results) — will retry", - group.Key, group.Count()); - } - finally - { - gate.Release(); - } - }); - - await Task.WhenAll(tasks); - return failed.ToList(); - } - - /// Store a single task result, chunking large JSON across properties/rows as needed. - public async Task StoreResultAsync(string runName, string taskId, string resultJson) - { - // Fast path: fits in a single property - if (resultJson.Length <= MaxPropertyChars) - { - var row = new StoreRow(runName, taskId) { Properties = { ["ResultJson"] = resultJson } }; - await _store.UpsertAsync(_resultsTable, row); - return; - } - - var chunks = ChunkString(resultJson, MaxPropertyChars); - - // Try to fit all chunks into a single row - if (EstimateTotalChars(chunks) <= MaxEntityChars) - { - var row = new StoreRow(runName, taskId); - for (int i = 0; i < chunks.Count; i++) - row[$"ResultJson_{i}"] = chunks[i]; - row["ResultChunkCount"] = chunks.Count; - - await _store.UpsertAsync(_resultsTable, row); - return; - } - - // Row too large — split across multiple rows - var rowIndex = 0; - var chunkIndex = 0; - - while (chunkIndex < chunks.Count) - { - var rowKey = rowIndex == 0 ? taskId : $"{taskId}-part{rowIndex}"; - var row = new StoreRow(runName, rowKey); - - if (rowIndex > 0) - { - row["OriginalEntityId"] = taskId; - row["PartIndex"] = rowIndex; - } - - var currentChars = runName.Length + rowKey.Length + 100; // overhead estimate - - while (chunkIndex < chunks.Count) - { - var chunkChars = chunks[chunkIndex].Length; - if (currentChars + chunkChars + 20 > MaxEntityChars) - break; - - row[$"ResultJson_{chunkIndex}"] = chunks[chunkIndex]; - currentChars += chunkChars + 20; - chunkIndex++; - } - - row["ResultChunkCount"] = chunks.Count; - await _store.UpsertAsync(_resultsTable, row); - rowIndex++; - } - } - - /// - /// Get all result JSON strings for a run, reassembling any chunked/multi-row results. - /// - /// Buffers every result by signature — prefer or - /// for run-sized payloads. - /// - public async Task GetResultsAsync(string runName, CancellationToken ct = default) - { - var results = new List(); - await foreach (var result in StreamResultsAsync(runName, ct)) - results.Add(result); - return results.ToArray(); - } - - /// - /// Stream each run result, reassembled, as it becomes available from the backing store. - /// - /// Nothing is buffered except spill groups still waiting for their remaining rows, so a run whose - /// results each fit in one row holds ONE row at a time regardless of how many there are. - /// - public async IAsyncEnumerable StreamResultsAsync(string runName, - [EnumeratorCancellation] CancellationToken ct = default) - { - await foreach (var chunks in StreamResultChunkGroupsAsync(runName, ct)) - yield return chunks.Count == 1 ? chunks[0] : string.Concat(chunks); - } - - /// - /// Stream all result JSON strings for a run directly to a file, reassembling any chunked/multi-row - /// results on the fly. Writes JSON Lines (NDJSON): one result per line, no enclosing array. - /// - /// Chunks are written individually, so the largest allocation this makes is one chunk - /// () — the reassembled result is never built as a string. - /// - /// JSON Lines rather than a JSON array because the consumer is PowerShell. A JSON array forces the - /// reader to hold the whole document to find where each element ends; one result per line lets - /// Invoke-CraftPostExecution walk the file with File.ReadLines and hold ONE result at a time. That - /// is the difference between a 50-150MB Large Object Heap allocation per post-execution and none. - /// It also isolates failure: a malformed result costs that result, not the entire aggregate. - /// - /// Returns the number of results written. - /// - public async Task StreamResultsToJsonLinesAsync(string runName, string filePath, - CancellationToken ct = default) - { - var count = 0; - - await using (var writer = new StreamWriter(filePath, append: false, Encoding.UTF8, bufferSize: 65536)) - { - await foreach (var chunks in StreamResultChunkGroupsAsync(runName, ct)) - { - foreach (var chunk in chunks) - await WriteSingleLineAsync(writer, chunk); - await writer.WriteAsync('\n'); - count++; - } - } - - _logger.LogInformation("[OrchestratorStore] Streamed {Count} results to {Path} for run {Name}", - count, filePath, runName); - - return count; - } - - /// - /// Write a chunk with any raw CR/LF removed, so one result stays on one line. - /// - /// Results are expected to be compact JSON on a single line — that is what Invoke-CraftTask's - /// `ConvertTo-Json -Compress` produces, and JSON escapes newlines inside strings as \n rather than - /// emitting them raw. A raw newline can therefore only appear as inter-token whitespace, which - /// carries no meaning, or in a result that was not valid JSON to begin with (a task script that - /// wrote several objects to the output stream — the runner joins those with "\n"). Dropping the - /// character is right in the first case and no worse than today's behaviour in the second, where - /// the malformed result currently takes the whole aggregate's parse down with it. - /// - /// The scan is the common-case fast path: no newline means the chunk is written untouched, with - /// no copy and no per-character work beyond the search itself. - /// - private static async Task WriteSingleLineAsync(StreamWriter writer, string chunk) - { - var start = 0; - int idx; - - while ((idx = chunk.AsSpan(start).IndexOfAny('\r', '\n')) >= 0) - { - var abs = start + idx; - if (abs > start) - await writer.WriteAsync(chunk.AsMemory(start, abs - start)); - start = abs + 1; - } - - if (start == 0) - await writer.WriteAsync(chunk); - else if (start < chunk.Length) - await writer.WriteAsync(chunk.AsMemory(start)); - } - - /// - /// The shared core: yields each logical result as its ordered chunk list, as soon as that result is - /// complete, and drops every row it has finished with. - /// - /// This used to be LoadResultGroupsAsync, which materialized EVERY result row for the run into a - /// dictionary before a single byte was written — so the callers named "stream" held the entire - /// payload (as UTF-16, ~2x the stored size) before they started. For a 738-task run whose aggregate - /// is 50-150MB that was a few hundred MB against a 2398MB heap cap, concurrently per post-execution. - /// - /// Rows are grouped by (OriginalEntityId ?? RowKey) and completed by chunk count rather than by - /// arrival order, so this makes no assumption about the order - /// returns rows in — the interface promises none. - /// - private async IAsyncEnumerable> StreamResultChunkGroupsAsync(string runName, - [EnumeratorCancellation] CancellationToken ct = default) - { - // Allocated only if this run actually has a result too large for a single row. - Dictionary? pending = null; - - await foreach (var row in _store.QueryPartitionAsync(_resultsTable, runName, ct)) - { - var totalChunks = row.GetInt32("ResultChunkCount") ?? 0; - - // Fast path: the whole result is one property on this row. Emit and release it. - if (totalChunks == 0) - { - var json = row.GetString("ResultJson"); - if (!string.IsNullOrEmpty(json)) yield return new[] { json }; - continue; - } - - var originalId = row.GetString("OriginalEntityId"); - var key = !string.IsNullOrEmpty(originalId) ? originalId : row.RowKey; - - pending ??= new Dictionary(StringComparer.OrdinalIgnoreCase); - if (!pending.TryGetValue(key, out var group)) - pending[key] = group = new PendingResult(totalChunks); - - group.Absorb(row); - - // Chunked but single-row results complete on their first (only) row. - if (group.IsComplete) - { - pending.Remove(key); - if (group.HasContent) yield return group.Chunks; - } - } - - // A spill row never arrived (partial write, or cleanup raced us). Emit what we have rather than - // silently dropping the result, and say so. - if (pending is { Count: > 0 }) - { - foreach (var (key, group) in pending) - { - _logger.LogWarning( - "[OrchestratorStore] Result {Key} in run {Run} is incomplete: {Have}/{Total} chunks present", - key, runName, group.PresentCount, group.TotalChunks); - if (group.HasContent) yield return group.Chunks; - } - } - } - - /// - /// A result being reassembled from chunks spread over one or more rows. Holds only this result's - /// chunks — never the s they came from. - /// - private sealed class PendingResult(int totalChunks) - { - private readonly string[] _chunks = new string[totalChunks]; - - public int TotalChunks => _chunks.Length; - public int PresentCount { get; private set; } - public bool IsComplete => PresentCount == _chunks.Length; - public bool HasContent => _chunks.Any(c => !string.IsNullOrEmpty(c)); - - /// Take any chunks this row carries that we do not already have. - public void Absorb(StoreRow row) - { - for (var i = 0; i < _chunks.Length; i++) - { - if (_chunks[i] != null) continue; - var chunk = row.GetString($"ResultJson_{i}"); - if (chunk == null) continue; - _chunks[i] = chunk; - PresentCount++; - } - } - - /// The chunks in index order. Missing chunks (incomplete result) are skipped. - public IReadOnlyList Chunks => - PresentCount == _chunks.Length ? _chunks : _chunks.Where(c => c != null).ToArray(); - } - - /// Delete all entities across the 3 tables for a completed run. - public async Task CleanupRunAsync(string runName) - { - try - { - await _store.DeleteAsync(_runsTable, "Run", runName); - await _store.DeletePartitionAsync(_tasksTable, runName); - await _store.DeletePartitionAsync(_resultsTable, runName); - - _logger.LogInformation("[OrchestratorStore] Cleaned up run: {Name}", runName); - } - catch (Exception ex) - { - _logger.LogError(ex, "[OrchestratorStore] Failed to cleanup run: {Name}", runName); - } - } - - /// The statuses a RUN ends in. Distinct from , which is about tasks. - private static bool IsTerminalRun(string? status) => - status is "Completed" or "CompletedWithErrors" or "Failed" or "Cancelled"; - - /// Keys and Timestamp only — what the orphan scan needs, and nothing a Results chunk carries. - private static readonly string[] s_keysAndTimestamp = ["PartitionKey", "RowKey", "Timestamp"]; - - /// - /// Retention sweep over the three tables. Everything removed is decided per run: - /// - /// A run in a terminal status (Completed, CompletedWithErrors, Failed, Cancelled) whose - /// CompletedUtc — or StartedUtc, for a row written before completion was stamped — is older than - /// loses its Run row and its Tasks and Results partitions. - /// A run in any other status is exempt while it is in (this - /// process is driving it). Otherwise it is abandoned once nothing about it has been written for - /// : task completion is written in one transaction with the - /// '!!run-counter' row, so that row's Timestamp is the heartbeat, and the Run row's own Timestamp - /// and StartedUtc count too. That covers runs recovery could not resume (task script gone), runs - /// queued but never dispatched, and — on a host that shares the tables — runs another process - /// stopped driving. - /// A Tasks or Results partition with no Run row at all is removed once its newest row is - /// older than . Those come from racing a - /// late status write, and from a Run row deleted while its partitions were still being written; - /// nothing else ever looked at them. - /// - /// This used to consider only Completed/CompletedWithErrors/Failed runs that carried a CompletedUtc, - /// and ran only from the startup recovery pass — so Cancelled runs, abandoned runs and orphaned - /// partitions lived forever, and on a host that was not restarted so did everything else. - /// - public async Task CleanupOldRunsAsync(TimeSpan retention, - IReadOnlySet? activeRuns = null, CancellationToken ct = default) - { - var cutoff = DateTimeOffset.UtcNow - retention; - var known = new HashSet(StringComparer.Ordinal); - var expired = new List(); - var abandoned = new List(); - - // Collect first, then delete — avoids mutating the "Run" partition while enumerating it. - var runs = new List(); - await foreach (var row in _store.QueryPartitionAsync(_runsTable, "Run", ct)) - runs.Add(row); - - foreach (var row in runs) - { - known.Add(row.RowKey); - - if (IsTerminalRun(row.GetString("Status"))) - { - var ended = row.GetDateTimeOffset("CompletedUtc") - ?? row.GetDateTimeOffset("StartedUtc") - ?? row.Timestamp; - if (ended < cutoff) expired.Add(row.RowKey); - continue; - } - - if (activeRuns != null && activeRuns.Contains(row.RowKey)) continue; - - var counter = await _store.GetAsync(_tasksTable, row.RowKey, CounterRowKey, ct); - var lastActivity = Newest(counter?.Timestamp, row.Timestamp, row.GetDateTimeOffset("StartedUtc")); - if (lastActivity < cutoff) abandoned.Add(row.RowKey); - } - - foreach (var name in expired) - await CleanupRunAsync(name); - foreach (var name in abandoned) - { - _logger.LogInformation( - "[OrchestratorStore] Run {Name} is not active and has not been written to for {Hours:F0}h — treating it as abandoned", - name, retention.TotalHours); - await CleanupRunAsync(name); - } - - var orphans = 0; - foreach (var table in new[] { _tasksTable, _resultsTable }) - orphans += await CleanupOrphanPartitionsAsync(table, known, cutoff, ct); - - if (expired.Count + abandoned.Count + orphans > 0) - _logger.LogInformation( - "[OrchestratorStore] Retention sweep removed {Expired} finished run(s), {Abandoned} abandoned run(s) and {Orphans} orphaned partition(s) older than {Hours:F0}h ({Examined} runs examined)", - expired.Count, abandoned.Count, orphans, retention.TotalHours, runs.Count); - - return new OrchestratorCleanupResult(runs.Count, expired, abandoned, orphans); - } - - private async Task CleanupOrphanPartitionsAsync(string table, HashSet knownRuns, - DateTimeOffset cutoff, CancellationToken ct) - { - // Newest row per partition that has no Run row. Only keys and Timestamp travel: a Results - // partition IS the run's payload, and reading that back every sweep would be the cost this - // sweep exists to avoid. A backend that ignores the projection still answers correctly, just - // expensively. - var newest = new Dictionary(StringComparer.Ordinal); - var unstamped = new HashSet(StringComparer.Ordinal); - await foreach (var row in _store.QueryTableAsync(table, null, s_keysAndTimestamp, ct)) - { - if (knownRuns.Contains(row.PartitionKey)) continue; - if (row.Timestamp is not { } stamped) - { - // No way to tell how old it is — never guess in the direction of deleting. - unstamped.Add(row.PartitionKey); - continue; - } - if (!newest.TryGetValue(row.PartitionKey, out var current) || stamped > current) - newest[row.PartitionKey] = stamped; - } - - var removed = 0; - foreach (var (partition, stamped) in newest) - { - if (stamped >= cutoff || unstamped.Contains(partition)) continue; - try - { - await _store.DeletePartitionAsync(table, partition, ct); - removed++; - _logger.LogInformation("[OrchestratorStore] Removed orphaned partition {Table}/{Partition}", table, partition); - } - catch (Exception ex) - { - _logger.LogWarning(ex, "[OrchestratorStore] Failed to remove orphaned partition {Table}/{Partition}", table, partition); - } - } - return removed; - } - - private static DateTimeOffset? Newest(params DateTimeOffset?[] candidates) - { - DateTimeOffset? newest = null; - foreach (var candidate in candidates) - if (candidate.HasValue && (!newest.HasValue || candidate.Value > newest.Value)) newest = candidate; - return newest; - } - - /// Split a string into chunks of at most maxChars characters, avoiding surrogate splits. - internal static List ChunkString(string value, int maxChars) - { - var chunks = new List(); - var start = 0; - - while (start < value.Length) - { - var remaining = value.Length - start; - var take = Math.Min(remaining, maxChars); - - if (take < remaining && char.IsHighSurrogate(value[start + take - 1])) - take--; - - chunks.Add(value.Substring(start, take)); - start += take; - } - - return chunks; - } - - private static int EstimateTotalChars(List chunks) - { - var total = 200; // overhead for keys + metadata properties - for (int i = 0; i < chunks.Count; i++) - total += chunks[i].Length + 20; // chunk + property name - return total; - } -} diff --git a/Services/Storage/ResultStore.cs b/Services/Storage/ResultStore.cs new file mode 100644 index 0000000..d2c2f38 --- /dev/null +++ b/Services/Storage/ResultStore.cs @@ -0,0 +1,315 @@ +using System.Runtime.CompilerServices; +using System.Text; +using Craft.Configuration; + +namespace Craft.Storage; + +/// +/// Task results for a run's aggregation, one partition per run (the run key). A result is written before +/// its task is counted done, so the aggregation never runs short of one that finished. +/// +public sealed class ResultStore +{ + private readonly ILogger _logger; + private readonly ICraftTableStore _store; + private readonly string _resultsTable; + + public ResultStore(ILogger logger, CraftSettings settings, ICraftTableStore store) + { + _logger = logger; + _store = store; + _resultsTable = $"{settings.Orchestrator.TablePrefix}TaskResults"; + } + + public Task InitializeAsync(CancellationToken ct = default) => _store.EnsureTableAsync(_resultsTable, ct); + + // ─── Result storage ─── + // Results can be large (50–150 MB for big runs). We chunk a result across multiple properties and, + // if needed, multiple rows in the same partition. These bounds are sized for Azure Table Storage + // (64 KiB/property, 1 MiB/entity); on a backend without those limits the chunking is simply + // unnecessary but still correct, and it keeps per-row payloads small (good for e.g. SQL packet size). + private const int MaxPropertyChars = 30_000; + private const int MaxEntityChars = 450_000; + + /// Store a single task result, chunking large JSON across properties/rows as needed. + public async Task StoreResultAsync(string runName, string taskId, string resultJson) + { + // Fast path: fits in a single property + if (resultJson.Length <= MaxPropertyChars) + { + var row = new StoreRow(runName, taskId) { Properties = { ["ResultJson"] = resultJson } }; + await _store.UpsertAsync(_resultsTable, row); + return; + } + + var chunks = ChunkString(resultJson, MaxPropertyChars); + + // Try to fit all chunks into a single row + if (EstimateTotalChars(chunks) <= MaxEntityChars) + { + var row = new StoreRow(runName, taskId); + for (int i = 0; i < chunks.Count; i++) + row[$"ResultJson_{i}"] = chunks[i]; + row["ResultChunkCount"] = chunks.Count; + + await _store.UpsertAsync(_resultsTable, row); + return; + } + + // Row too large — split across multiple rows + var rowIndex = 0; + var chunkIndex = 0; + + while (chunkIndex < chunks.Count) + { + var rowKey = rowIndex == 0 ? taskId : $"{taskId}-part{rowIndex}"; + var row = new StoreRow(runName, rowKey); + + if (rowIndex > 0) + { + row["OriginalEntityId"] = taskId; + row["PartIndex"] = rowIndex; + } + + var currentChars = runName.Length + rowKey.Length + 100; // overhead estimate + + while (chunkIndex < chunks.Count) + { + var chunkChars = chunks[chunkIndex].Length; + if (currentChars + chunkChars + 20 > MaxEntityChars) + break; + + row[$"ResultJson_{chunkIndex}"] = chunks[chunkIndex]; + currentChars += chunkChars + 20; + chunkIndex++; + } + + row["ResultChunkCount"] = chunks.Count; + await _store.UpsertAsync(_resultsTable, row); + rowIndex++; + } + } + + /// + /// Get all result JSON strings for a run, reassembling any chunked/multi-row results. + /// + /// Buffers every result by signature — prefer or + /// for run-sized payloads. + /// + public async Task GetResultsAsync(string runName, CancellationToken ct = default) + { + var results = new List(); + await foreach (var result in StreamResultsAsync(runName, ct)) + results.Add(result); + return results.ToArray(); + } + + /// + /// Stream each run result, reassembled, as it becomes available from the backing store. + /// + /// Nothing is buffered except spill groups still waiting for their remaining rows, so a run whose + /// results each fit in one row holds ONE row at a time regardless of how many there are. + /// + public async IAsyncEnumerable StreamResultsAsync(string runName, + [EnumeratorCancellation] CancellationToken ct = default) + { + await foreach (var chunks in StreamResultChunkGroupsAsync(runName, ct)) + yield return chunks.Count == 1 ? chunks[0] : string.Concat(chunks); + } + + /// + /// Stream all result JSON strings for a run directly to a file, reassembling any chunked/multi-row + /// results on the fly. Writes JSON Lines (NDJSON): one result per line, no enclosing array. + /// + /// Chunks are written individually, so the largest allocation this makes is one chunk + /// () — the reassembled result is never built as a string. + /// + /// JSON Lines rather than a JSON array because the consumer is PowerShell. A JSON array forces the + /// reader to hold the whole document to find where each element ends; one result per line lets + /// Invoke-CraftPostExecution walk the file with File.ReadLines and hold ONE result at a time. That + /// is the difference between a 50-150MB Large Object Heap allocation per post-execution and none. + /// It also isolates failure: a malformed result costs that result, not the entire aggregate. + /// + /// Returns the number of results written. + /// + public async Task StreamResultsToJsonLinesAsync(string runName, string filePath, + CancellationToken ct = default) + { + var count = 0; + + await using (var writer = new StreamWriter(filePath, append: false, Encoding.UTF8, bufferSize: 65536)) + { + await foreach (var chunks in StreamResultChunkGroupsAsync(runName, ct)) + { + foreach (var chunk in chunks) + await WriteSingleLineAsync(writer, chunk); + await writer.WriteAsync('\n'); + count++; + } + } + + _logger.LogInformation("[OrchestratorStore] Streamed {Count} results to {Path} for run {Name}", + count, filePath, runName); + + return count; + } + + /// + /// Write a chunk with any raw CR/LF removed, so one result stays on one line. + /// + /// Results are expected to be compact JSON on a single line — that is what Invoke-CraftTask's + /// `ConvertTo-Json -Compress` produces, and JSON escapes newlines inside strings as \n rather than + /// emitting them raw. A raw newline can therefore only appear as inter-token whitespace, which + /// carries no meaning, or in a result that was not valid JSON to begin with (a task script that + /// wrote several objects to the output stream — the runner joins those with "\n"). Dropping the + /// character is right in the first case and no worse than today's behaviour in the second, where + /// the malformed result currently takes the whole aggregate's parse down with it. + /// + /// The scan is the common-case fast path: no newline means the chunk is written untouched, with + /// no copy and no per-character work beyond the search itself. + /// + private static async Task WriteSingleLineAsync(StreamWriter writer, string chunk) + { + var start = 0; + int idx; + + while ((idx = chunk.AsSpan(start).IndexOfAny('\r', '\n')) >= 0) + { + var abs = start + idx; + if (abs > start) + await writer.WriteAsync(chunk.AsMemory(start, abs - start)); + start = abs + 1; + } + + if (start == 0) + await writer.WriteAsync(chunk); + else if (start < chunk.Length) + await writer.WriteAsync(chunk.AsMemory(start)); + } + + /// + /// The shared core: yields each logical result as its ordered chunk list, as soon as that result is + /// complete, and drops every row it has finished with. + /// + /// This used to be LoadResultGroupsAsync, which materialized EVERY result row for the run into a + /// dictionary before a single byte was written — so the callers named "stream" held the entire + /// payload (as UTF-16, ~2x the stored size) before they started. For a 738-task run whose aggregate + /// is 50-150MB that was a few hundred MB against a 2398MB heap cap, concurrently per post-execution. + /// + /// Rows are grouped by (OriginalEntityId ?? RowKey) and completed by chunk count rather than by + /// arrival order, so this makes no assumption about the order + /// returns rows in — the interface promises none. + /// + private async IAsyncEnumerable> StreamResultChunkGroupsAsync(string runName, + [EnumeratorCancellation] CancellationToken ct = default) + { + // Allocated only if this run actually has a result too large for a single row. + Dictionary? pending = null; + + await foreach (var row in _store.QueryPartitionAsync(_resultsTable, runName, ct)) + { + var totalChunks = row.GetInt32("ResultChunkCount") ?? 0; + + // Fast path: the whole result is one property on this row. Emit and release it. + if (totalChunks == 0) + { + var json = row.GetString("ResultJson"); + if (!string.IsNullOrEmpty(json)) yield return new[] { json }; + continue; + } + + var originalId = row.GetString("OriginalEntityId"); + var key = !string.IsNullOrEmpty(originalId) ? originalId : row.RowKey; + + pending ??= new Dictionary(StringComparer.OrdinalIgnoreCase); + if (!pending.TryGetValue(key, out var group)) + pending[key] = group = new PendingResult(totalChunks); + + group.Absorb(row); + + // Chunked but single-row results complete on their first (only) row. + if (group.IsComplete) + { + pending.Remove(key); + if (group.HasContent) yield return group.Chunks; + } + } + + // A spill row never arrived (partial write, or cleanup raced us). Emit what we have rather than + // silently dropping the result, and say so. + if (pending is { Count: > 0 }) + { + foreach (var (key, group) in pending) + { + _logger.LogWarning( + "[OrchestratorStore] Result {Key} in run {Run} is incomplete: {Have}/{Total} chunks present", + key, runName, group.PresentCount, group.TotalChunks); + if (group.HasContent) yield return group.Chunks; + } + } + } + + /// + /// A result being reassembled from chunks spread over one or more rows. Holds only this result's + /// chunks — never the s they came from. + /// + private sealed class PendingResult(int totalChunks) + { + private readonly string[] _chunks = new string[totalChunks]; + + public int TotalChunks => _chunks.Length; + public int PresentCount { get; private set; } + public bool IsComplete => PresentCount == _chunks.Length; + public bool HasContent => _chunks.Any(c => !string.IsNullOrEmpty(c)); + + /// Take any chunks this row carries that we do not already have. + public void Absorb(StoreRow row) + { + for (var i = 0; i < _chunks.Length; i++) + { + if (_chunks[i] != null) continue; + var chunk = row.GetString($"ResultJson_{i}"); + if (chunk == null) continue; + _chunks[i] = chunk; + PresentCount++; + } + } + + /// The chunks in index order. Missing chunks (incomplete result) are skipped. + public IReadOnlyList Chunks => + PresentCount == _chunks.Length ? _chunks : _chunks.Where(c => c != null).ToArray(); + } + + /// Drop a run's results once its aggregation has read them. + public Task DeleteRunAsync(string runKey, CancellationToken ct = default) => + _store.DeletePartitionAsync(_resultsTable, runKey, ct); + + /// Split a string into chunks of at most maxChars characters, avoiding surrogate splits. + internal static List ChunkString(string value, int maxChars) + { + var chunks = new List(); + var start = 0; + + while (start < value.Length) + { + var remaining = value.Length - start; + var take = Math.Min(remaining, maxChars); + + if (take < remaining && char.IsHighSurrogate(value[start + take - 1])) + take--; + + chunks.Add(value.Substring(start, take)); + start += take; + } + + return chunks; + } + + private static int EstimateTotalChars(List chunks) + { + var total = 200; // overhead for keys + metadata properties + for (int i = 0; i < chunks.Count; i++) + total += chunks[i].Length + 20; // chunk + property name + return total; + } +} diff --git a/Services/Storage/ResultWrite.cs b/Services/Storage/ResultWrite.cs deleted file mode 100644 index dbc8e6a..0000000 --- a/Services/Storage/ResultWrite.cs +++ /dev/null @@ -1,10 +0,0 @@ -namespace Craft.Storage; - -/// -/// A small task result queued for the coalescing status writer. Only results that fit a single Azure -/// Table property travel this way; larger results keep the chunked, directly-awaited -/// path. Written to the Results table before the -/// task's terminal status marker in the same flush, so a result is always durable before its task is -/// counted done. -/// -public record ResultWrite(string RunName, string TaskId, string ResultJson); diff --git a/Services/Storage/TaskStatusWrite.cs b/Services/Storage/TaskStatusWrite.cs deleted file mode 100644 index 6c22012..0000000 --- a/Services/Storage/TaskStatusWrite.cs +++ /dev/null @@ -1,12 +0,0 @@ -namespace Craft.Storage; - -/// -/// An immutable snapshot of a task's status for the coalescing batched writer. -/// -/// Every column the task row carries must appear here. The store upserts with -/// TableUpdateMode.Replace, so any column missing from this snapshot is ERASED by the next -/// status transition — which is how a per-task Priority override would silently vanish the first -/// time the task moved to Running. -/// -public record TaskStatusWrite(string RunName, string TaskId, string Status, string? ParametersJson, - int AttemptCount, string? LastError, DateTime? CompletedUtc, int? Priority, int Sequence = 0); diff --git a/Services/Storage/WorkStore.cs b/Services/Storage/WorkStore.cs new file mode 100644 index 0000000..fe36def --- /dev/null +++ b/Services/Storage/WorkStore.cs @@ -0,0 +1,1203 @@ +using System.Collections.Concurrent; +using System.Globalization; +using System.Text.Json; +using Craft.Configuration; + +namespace Craft.Storage; + +/// +/// Durable orchestration state. Each run is one partition of the Work table, and every task's whole state +/// is one row in it whose RowKey prefix is its state: +/// +/// $run header: identity, options and the Total/Done counts +/// T|{seq} payload (immutable) +/// P|{seq} pending R|{seq} running (Owner, LeaseUntil) D|{seq} done +/// C|{child} a child run the parent waits for +/// +/// So every state change is one partition transaction, the counts can never drift from the rows, and +/// nothing has to be reconciled: a task is exactly where its row says. A worker that dies leaves an R row +/// whose lease lapses, and the next claim takes it back. +/// +/// Three small tables sit beside it: Ready (one row per run with work, ordered by band then start time, +/// read by the scheduler), Names (latest run per name; every unfinished run, partition A; and the +/// instance lock) and Finished (completion order, for retention). +/// +/// The active-run row is written before anything else of a run and removed after everything else, so it is +/// the authoritative list of runs that exist; Ready and Finished are indexes derived from the runs and can be +/// rebuilt from it (). A stale index row costs a read, never a wrong answer. +/// +public sealed class WorkStore +{ + public const string HeaderKey = "$run"; + + /// The PostExecution parameters, kept off the header: the header is rewritten in every finish + /// transaction, where rows cannot be split, and the parameters can outgrow a property. + private const string PostExecKey = "$post"; + public const int AggregateSeq = 99_999_999; + + /// Tasks per claim or completion transaction: two ops each plus the header. + public const int MaxPerTransaction = 49; + + private readonly int MaxAttempts; + private const int ConflictRetries = 16; + private const int MaxRunningScan = 500; + + private readonly ICraftTableStore _store; + private readonly ILogger _logger; + private readonly PartitionRateLimiter _rate; + private readonly string _work, _ready, _names, _finished, _results; + private readonly string[] _legacyTables; + private volatile bool _initialized; + + public WorkStore(ILogger logger, CraftSettings settings, ICraftTableStore store, + PartitionRateLimiter? rate = null) + { + _logger = logger; + _store = store; + _rate = rate ?? new PartitionRateLimiter(); + MaxAttempts = Math.Max(1, settings.Orchestrator.MaxRetries); + var p = settings.Orchestrator.TablePrefix; + _work = $"{p}Work"; + _ready = $"{p}Ready"; + _names = $"{p}Names"; + _finished = $"{p}Finished"; + _results = $"{p}TaskResults"; + _legacyTables = [$"{p}Queue", $"{p}QueueIndex", $"{p}Tasks", $"{p}Runs", $"{p}Results"]; + } + + public string ResultsTable => _results; + + /// Called after every finish transaction that reached the barrier or completed a run, whoever made it. + public Func? AfterFinish { get; set; } + + public async Task InitializeAsync(CancellationToken ct = default) + { + if (_initialized) return; + foreach (var t in new[] { _work, _ready, _names, _finished, _results }) + await _store.EnsureTableAsync(t, ct); + _initialized = true; + } + + /// The earlier orchestration tables, which this store replaces outright. Their in-flight work is + /// dropped; the timers that created it create it again. + public IReadOnlyList LegacyTables => _legacyTables; + + // ── keys ── + + public static string RunKeyFor(string name, DateTime startedUtc) => + $"{name}~{startedUtc.Ticks.ToString("x", CultureInfo.InvariantCulture)}"; + + private static string Seq(int seq) => seq.ToString("D8", CultureInfo.InvariantCulture); + private static string Key(char state, int seq) => $"{state}|{Seq(seq)}"; + private static int SeqOf(string rowKey) => int.Parse(rowKey.AsSpan(2), CultureInfo.InvariantCulture); + private static string ReadyPartition(int band) => "P" + Math.Clamp(band, 0, 99).ToString("D2", CultureInfo.InvariantCulture); + private static string ReadyKey(RunHeader h) => $"{h.StartedUtc.Ticks.ToString("D19", CultureInfo.InvariantCulture)}|{h.RunKey}"; + + /// A run's rows in one state, in seq order; 0 means all. A bounded read asks + /// the service for only that many rows a page. + private async IAsyncEnumerable Range(string runKey, char state, int max = 0, + [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) + { + var count = 0; + await foreach (var row in _store.QueryRowKeyRangeAsync(_work, runKey, $"{state}|", $"{state}}}", + maxPerPage: max > 0 ? Math.Min(max, 1000) : null, ct: ct)) + { + yield return row; + if (max > 0 && ++count >= max) yield break; + } + } + + // ── create ── + + public sealed record NewTask(string TaskId, Dictionary Parameters); + + /// + /// Persist a run. Payload and pending rows go first and the header last, so a crash part-way leaves rows + /// that no Ready entry points at, never a visible run with tasks missing. + /// + public async Task CreateRunAsync(RunHeader header, IReadOnlyList tasks, CancellationToken ct = default) + { + await InitializeAsync(ct); + var payload = new List(tasks.Count); + var pending = new List(tasks.Count); + for (var i = 0; i < tasks.Count; i++) + { + payload.Add(new StoreRow(header.RunKey, Key('T', i)) + { + Properties = { ["TaskId"] = tasks[i].TaskId, ["ParametersJson"] = JsonSerializer.Serialize(tasks[i].Parameters, RunHeader.Json) } + }); + pending.Add(PendingRow(header.RunKey, i, tasks[i].TaskId, 0)); + } + + // The active-run row first: from here on a crash leaves a run the startup repair can see, finish or remove. + await _store.UpsertAsync(_names, new StoreRow(ActivePartition, header.RunKey) + { + Properties = { ["Name"] = header.Name, ["StartedUtc"] = new DateTimeOffset(DateTime.SpecifyKind(header.StartedUtc, DateTimeKind.Utc)) } + }, ct); + await _rate.TakeAsync(header.RunKey, tasks.Count * 2 + 1, ct); + await _store.UpsertBatchAsync(_work, header.RunKey, payload, ct); + await _store.UpsertBatchAsync(_work, header.RunKey, pending, ct); + if (header.PostExecParametersJson is { } post) + await _store.UpsertAsync(_work, new StoreRow(header.RunKey, PostExecKey) { Properties = { ["Json"] = post } }, ct); + header.Total = tasks.Count; + await _store.UpsertAsync(_work, header.ToRow(), ct); + + await IndexAsync("record the latest run of its name", header.RunKey, + () => _store.UpsertAsync(_names, new StoreRow("N", header.Name) { Properties = { ["RunKey"] = header.RunKey } }, ct), ct); + await IndexAsync("list it as ready", header.RunKey, () => PublishReadyAsync(header, ct), ct); + Changed(header.RunKey); + return (await GetRunAsync(header.RunKey, ct))!; + } + + private static StoreRow PendingRow(string runKey, int seq, string taskId, int attempt) => new(runKey, Key('P', seq)) + { + Properties = { ["TaskId"] = taskId, ["Attempt"] = attempt } + }; + + public Task PublishReadyAsync(RunHeader h, CancellationToken ct = default) => + _store.UpsertAsync(_ready, new StoreRow(ReadyPartition(h.Priority), ReadyKey(h)) + { + Properties = + { + ["RunKey"] = h.RunKey, ["Name"] = h.Name, ["Total"] = h.Total, ["Done"] = h.Done, ["Failed"] = h.Failed, + ["Cancelled"] = h.Cancelled, ["Reference"] = h.Reference, ["Sequential"] = h.Sequential ? 1 : 0, + ["MaxConcurrency"] = h.MaxConcurrency, + } + }, ct); + + // ── read ── + + public async Task GetRunAsync(string runKey, CancellationToken ct = default) + { + var row = await _store.GetAsync(_work, runKey, HeaderKey, ct); + return row == null ? null : RunHeader.FromRow(row); + } + + private const string ActivePartition = "A"; + + /// + /// Every unfinished run with this name, oldest first. Run keys are {name}~{hex ticks}, so one name's + /// outings share a key prefix and sort by start; the Name check drops a different name that merely + /// starts with this one and a tilde. + /// + public async Task> GetActiveRunsAsync(string name, CancellationToken ct = default) + { + var runs = new List(); + var stale = new List(); + await foreach (var row in _store.QueryRowKeyRangeAsync(_names, ActivePartition, $"{name}~", $"{name}~g", ct: ct)) + { + if (row.GetString("Name") != name) continue; + if (await GetRunAsync(row.RowKey, ct) is { IsFinished: false } run) runs.Add(run); + else stale.Add(row.RowKey); + } + foreach (var key in stale) await _store.DeleteAsync(_names, ActivePartition, key, ct); + return runs; + } + + /// + /// The name runs collide on: the run name without a trailing -{guid}, the queue id a caller appends so + /// its queue page can find each outing (Cache-{queueId} and Cache are one family). + /// + public static string CollisionFamily(string name) => + name.Length > 37 && name[^37] == '-' && Guid.TryParseExact(name.AsSpan(name.Length - 36), "D", out _) + ? name[..^37] + : name; + + /// Whether any run of this collision family is unfinished: one range read for the plain name and one + /// for {family}-.... + public async Task IsFamilyActiveAsync(string family, CancellationToken ct = default) + { + if ((await GetActiveRunsAsync(family, ct)).Count > 0) return true; + await foreach (var row in _store.QueryRowKeyRangeAsync(_names, ActivePartition, $"{family}-", $"{family}.", ct: ct)) + { + if (row.GetString("Name") is { } name && CollisionFamily(name) == family + && await GetRunAsync(row.RowKey, ct) is { IsFinished: false }) + return true; + } + return false; + } + + /// A run by its key, or else the newest unfinished run with that name. + public async Task ResolveRunAsync(string keyOrName, CancellationToken ct = default) => + (keyOrName.Contains('~') ? await GetRunAsync(keyOrName, ct) : null) + ?? (await GetActiveRunsAsync(keyOrName, ct)).LastOrDefault(); + + /// The latest run with this name, active or finished. + public async Task GetRunByNameAsync(string name, CancellationToken ct = default) + { + var row = await _store.GetAsync(_names, "N", name, ct); + return row?.GetString("RunKey") is { } key ? await GetRunAsync(key, ct) : null; + } + + public async Task GetPostExecParametersAsync(string runKey, CancellationToken ct = default) => + (await _store.GetAsync(_work, runKey, PostExecKey, ct))?.GetString("Json"); + + public async Task?> GetPayloadAsync(string runKey, int seq, CancellationToken ct = default) + { + var row = await _store.GetAsync(_work, runKey, Key('T', seq), ct); + var json = row?.GetString("ParametersJson"); + if (json == null) return null; + try { return JsonSerializer.Deserialize>(json, RunHeader.Json) ?? []; } + catch (JsonException) { return []; } + } + + /// A run with work, as the scheduler sees it; its mode rides along so the pump can apply a + /// concurrency limit without reading the run. + public sealed record ReadyEntry(int Band, string RunKey, string Name, int Total, int Done, DateTime StartedUtc, + string? Reference = null, int Failed = 0, int Cancelled = 0, bool Sequential = false, int MaxConcurrency = 0); + + /// Runs with work, best band first and oldest first within it. + public async IAsyncEnumerable ReadReadyAsync(int pageSize = 32, + [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) + { + await foreach (var row in _store.QueryTableAsync(_ready, null, pageSize, ct)) + { + if (row.GetString("RunKey") is not { } key) continue; + var band = int.TryParse(row.PartitionKey.AsSpan(1), NumberStyles.None, CultureInfo.InvariantCulture, out var b) ? b : 99; + var ticks = long.TryParse(row.RowKey.AsSpan(0, Math.Min(19, row.RowKey.Length)), NumberStyles.None, CultureInfo.InvariantCulture, out var t) ? t : 0; + yield return new ReadyEntry(band, key, row.GetString("Name") ?? key, row.GetInt32("Total") ?? 0, + row.GetInt32("Done") ?? 0, new DateTime(ticks, DateTimeKind.Utc), row.GetString("Reference"), + row.GetInt32("Failed") ?? 0, row.GetInt32("Cancelled") ?? 0, row.GetInt32("Sequential") == 1, + row.GetInt32("MaxConcurrency") ?? 0); + } + } + + public sealed record TaskRow(int Seq, string TaskId, char State, string? Status, int Attempt, string? Owner, + DateTimeOffset? LeaseUntil, string? LastError); + + private static TaskRow ToTask(StoreRow r) => new(SeqOf(r.RowKey), r.GetString("TaskId") ?? "", r.RowKey[0], + r.GetString("Status"), r.GetInt32("Attempt") ?? 0, r.GetString("Owner"), r.GetDateTimeOffset("LeaseUntil"), + r.GetString("LastError")); + + /// A run's task rows (P, R and D, or one state), in seq order. bounds the + /// rows read per state; 0 reads them all, which on a large run is a long read, so status views pass a bound. + public async Task> GetTasksAsync(string runKey, char? state = null, int max = 0, CancellationToken ct = default) + { + var rows = new List(); + foreach (var s in state is { } one ? [one] : new[] { 'P', 'R', 'D' }) + await foreach (var r in Range(runKey, s, max, ct)) rows.Add(ToTask(r)); + return rows; + } + + /// One pending task by its position, or null when it is not pending. A point read. + public async Task GetPendingAsync(string runKey, int seq, CancellationToken ct = default) => + await _store.GetAsync(_work, runKey, Key('P', seq), ct) is { } row ? ToTask(row) : null; + + // ── claim ── + + public sealed record ClaimedTask(string RunKey, int Seq, string TaskId, int Attempt); + + /// What a claim saw of the run, for the scheduler: whether its pending tasks ran out, and when its + /// earliest live claim lapses (if it is not renewed by then, there is work to take back). + public sealed class ClaimProbe + { + public bool PendingExhausted { get; internal set; } + public DateTimeOffset? EarliestLeaseUntil { get; internal set; } + } + + /// Raised in-process with the run key whenever a run's work may have changed (created, a task + /// finished or was handed back, a child added, moved band), so a scheduler that had written the run off + /// looks at it again without depending on the Ready index write having landed. + public event Action? RunChanged; + + private void Changed(string runKey) + { + try { RunChanged?.Invoke(runKey); } + catch (Exception ex) { _logger.LogWarning(ex, "[WorkStore] A run-changed listener failed for {Run}", runKey); } + } + + /// + /// Move up to pending tasks to running under , plus, with + /// , running tasks whose lease lapsed. A task claimed for the + /// th time and lapsing again is failed instead of claimed. One transaction; a lost + /// race returns empty and the caller moves on. + /// + public async Task> ClaimAsync(string runKey, int max, string owner, TimeSpan lease, + bool reclaimExpired, ClaimProbe? probe = null, bool othersAreDead = false, CancellationToken ct = default) + { + max = Math.Min(max, MaxPerTransaction); + if (max <= 0) return []; + + var now = DateTimeOffset.UtcNow; + var expired = new List(); + if (reclaimExpired || probe != null) + { + // Bounded so a run left with thousands of lapsed claims costs a fixed read per claim; an earliest + // lease from a partial scan only makes the scheduler look again sooner. + await foreach (var r in Range(runKey, 'R', MaxRunningScan, ct)) + { + var dead = r.GetDateTimeOffset("LeaseUntil") is not { } until || until <= now + || (othersAreDead && r.GetString("Owner") != owner); + if (dead) + { + if (reclaimExpired && expired.Count < max) expired.Add(r); + else if (probe != null) probe.EarliestLeaseUntil = now; + } + else if (probe != null && r.GetDateTimeOffset("LeaseUntil") is { } live + && (probe.EarliestLeaseUntil is not { } seen || live < seen)) + { + probe.EarliestLeaseUntil = live; + } + if (probe == null && expired.Count >= max) break; + } + } + + var pending = new List(); + if (expired.Count < max) + await foreach (var r in Range(runKey, 'P', max - expired.Count, ct: ct)) pending.Add(r); + if (probe != null) probe.PendingExhausted = pending.Count < max - expired.Count; + if (pending.Count + expired.Count == 0) return []; + + var leaseUntil = now.Add(lease); + var ops = new List(); + var claimed = new List(); + var poisoned = new List<(StoreRow Row, int Attempt)>(); + + foreach (var r in pending) + { + var seq = SeqOf(r.RowKey); + var attempt = (r.GetInt32("Attempt") ?? 0) + 1; + ops.Add(StoreOp.Delete(r)); + ops.Add(StoreOp.Insert(RunningRow(runKey, seq, r.GetString("TaskId")!, attempt, owner, leaseUntil))); + claimed.Add(new ClaimedTask(runKey, seq, r.GetString("TaskId")!, attempt)); + } + foreach (var r in expired) + { + var attempt = r.GetInt32("Attempt") ?? 0; + if (attempt >= MaxAttempts) { poisoned.Add((r, attempt)); continue; } + var seq = SeqOf(r.RowKey); + ops.Add(StoreOp.Replace(new StoreRow(runKey, r.RowKey) + { + ETag = r.ETag, + Properties = RunningRow(runKey, seq, r.GetString("TaskId")!, attempt + 1, owner, leaseUntil).Properties + })); + claimed.Add(new ClaimedTask(runKey, seq, r.GetString("TaskId")!, attempt + 1)); + } + + if (poisoned.Count > 0) + { + var finish = poisoned.Select(p => new Finish(SeqOf(p.Row.RowKey), "Failed", + $"Interrupted {p.Attempt} times without completing")).ToList(); + await FinishAsync(runKey, finish, null, ct); + } + if (ops.Count == 0) return []; + + await _rate.TakeAsync(runKey, ops.Count, ct); + return await _store.TrySubmitAsync(_work, runKey, ops, ct) ? claimed : []; + } + + /// + /// Claim the next step of a sequential run for , taking or keeping the run's driver + /// lease in the same transaction. Empty while another owner's driver lease is live, so a run never has two + /// drivers; a lapsed driver's step is reclaimed with it. + /// + /// True for the driver claiming its own next step; false (the pump) defers to any live driver. + /// The caller holds the instance lock, so a driver lease held by any other process + /// belongs to a process that has stopped, and its step is taken over at once. + public async Task ClaimSequentialAsync(string runKey, string owner, TimeSpan lease, bool continuing = false, + bool othersAreDead = false, CancellationToken ct = default) + { + var headerRow = await _store.GetAsync(_work, runKey, HeaderKey, ct); + if (headerRow == null) return null; + var header = RunHeader.FromRow(headerRow); + var now = DateTimeOffset.UtcNow; + var driverLive = header.DriverOwner != null && header.DriverLease > now && !(othersAreDead && header.DriverOwner != owner); + if (driverLive && !(continuing && header.DriverOwner == owner)) return null; + + StoreRow? row = null; + await foreach (var r in Range(runKey, 'R', ct: ct)) { row = r; break; } + var reclaiming = row != null; + if (row == null) await foreach (var r in Range(runKey, 'P', 1, ct)) { row = r; break; } + if (row == null) return null; + + var seq = SeqOf(row.RowKey); + var attempt = (row.GetInt32("Attempt") ?? 0) + 1; + if (reclaiming && attempt > MaxAttempts) + { + await FinishAsync(runKey, [new Finish(seq, "Failed", $"Interrupted {attempt - 1} times without completing")], 'R', ct); + if (header.StopOnFailure) + { + await CancelPendingAsync(runKey, StoppedReason(row.GetString("TaskId")), ct: ct); + return null; + } + return await ClaimSequentialAsync(runKey, owner, lease, continuing, othersAreDead, ct); + } + + var until = now.Add(lease); + header.DriverOwner = owner; + header.DriverLease = until; + var running = RunningRow(runKey, seq, row.GetString("TaskId")!, attempt, owner, until); + var ops = new List { StoreOp.Replace(header.ToRow(headerRow.ETag)) }; + if (reclaiming) ops.Add(StoreOp.Replace(new StoreRow(runKey, row.RowKey) { ETag = row.ETag, Properties = running.Properties })); + else { ops.Add(StoreOp.Delete(row)); ops.Add(StoreOp.Insert(running)); } + + await _rate.TakeAsync(runKey, ops.Count, ct); + return await _store.TrySubmitAsync(_work, runKey, ops, ct) + ? new ClaimedTask(runKey, seq, row.GetString("TaskId")!, attempt) + : null; + } + + /// Why a stop-on-failure run cancelled its remaining steps. + public static string StoppedReason(string? failedTaskId) => $"Not run: step {failedTaskId} failed and the run stops on failure"; + + /// Give up a sequential run's driver lease so the next step can be claimed by anyone. + public async Task ReleaseDriverAsync(string runKey, string owner, CancellationToken ct = default) + { + for (var attempt = 0; attempt < ConflictRetries; attempt++) + { + var headerRow = await _store.GetAsync(_work, runKey, HeaderKey, ct); + if (headerRow == null) return; + var header = RunHeader.FromRow(headerRow); + if (header.DriverOwner != owner) return; + header.DriverOwner = null; + header.DriverLease = null; + if (await _store.TrySubmitAsync(_work, runKey, [StoreOp.Replace(header.ToRow(headerRow.ETag))], ct)) return; + } + } + + private static StoreRow RunningRow(string runKey, int seq, string taskId, int attempt, string owner, DateTimeOffset leaseUntil) => + new(runKey, Key('R', seq)) + { + Properties = { ["TaskId"] = taskId, ["Attempt"] = attempt, ["Owner"] = owner, ["LeaseUntil"] = leaseUntil } + }; + + /// + /// Hand a running task back to pending without counting the attempt as spent: a shutdown that interrupted + /// it, or an aggregation that failed and has attempts left. Only if still holds it. + /// + public async Task ReleaseAsync(string runKey, int seq, string owner, bool refundAttempt, CancellationToken ct = default) + { + var row = await _store.GetAsync(_work, runKey, Key('R', seq), ct); + if (row == null || row.GetString("Owner") != owner) return false; + var attempt = (row.GetInt32("Attempt") ?? 1) - (refundAttempt ? 1 : 0); + await _rate.TakeAsync(runKey, 2, ct); + var released = await _store.TrySubmitAsync(_work, runKey, + [StoreOp.Delete(row), StoreOp.Insert(PendingRow(runKey, seq, row.GetString("TaskId")!, attempt))], ct); + if (released) Changed(runKey); + return released; + } + + /// Push the lease out on claims this owner still holds. Returns the claims it no longer holds. + public async Task> RenewAsync(IReadOnlyList claims, string owner, TimeSpan lease, + CancellationToken ct = default) + { + var lost = new List(); + var leaseUntil = DateTimeOffset.UtcNow.Add(lease); + foreach (var byRun in claims.GroupBy(c => c.RunKey)) + { + var ops = new List(); + foreach (var c in byRun) + { + var row = await _store.GetAsync(_work, c.RunKey, Key('R', c.Seq), ct); + if (row == null) continue; + if (row.GetString("Owner") != owner) { lost.Add(c); continue; } + row["LeaseUntil"] = leaseUntil; + ops.Add(StoreOp.Replace(row)); + } + foreach (var chunk in ops.Chunk(100)) + { + await _rate.TakeAsync(byRun.Key, chunk.Length, ct); + if (!await _store.TrySubmitAsync(_work, byRun.Key, chunk, ct)) + _logger.LogWarning("[WorkStore] Lease renewal for {Run} lost a race; the next renewal retries", byRun.Key); + } + } + return lost; + } + + // ── finish ── + + /// A task reaching a terminal status. set means "only if I still hold it". + public sealed record Finish(int Seq, string Status, string? Error = null, string? Owner = null, string? ChildKey = null); + + /// What a finish did to the run: the header after it, and whether this was the transaction that + /// completed the run's tasks (the barrier) or its aggregation. + public sealed record FinishOutcome(RunHeader Header, bool ReachedBarrier, bool Completed, int Applied); + + /// + /// Move tasks (R or, for a cancel, P) and child placeholders to done and update the counts, in one + /// transaction per chunk. The chunk that brings Done to Total is the barrier: it inserts the aggregation + /// task when the run has one, or completes the run when it does not. Finishing the aggregation task + /// completes the run. + /// + public async Task FinishAsync(string runKey, IReadOnlyList finishes, char? fromState = 'R', + CancellationToken ct = default) + { + FinishOutcome? outcome = null; + foreach (var chunk in finishes.Chunk(MaxPerTransaction)) + outcome = await FinishChunkAsync(runKey, chunk, fromState, ct) ?? outcome; + return outcome; + } + + /// What did: the finish, the step it claimed next (null when there is + /// none, the run was asked to cancel, or another driver holds it) and that step's payload when it could be read. + public sealed record StepResult(FinishOutcome? Outcome, ClaimedTask? Next, Dictionary? Payload); + + private sealed class StepClaim(string owner, TimeSpan lease) + { + public string Owner { get; } = owner; + public TimeSpan Lease { get; } = lease; + public ClaimedTask? Claimed { get; set; } + } + + /// + /// Finish a sequential run's current step and claim its next one in the same transaction, renewing the + /// driver lease: the next pending step, or the aggregation task when this step completes the run's tasks. + /// One round trip of concurrent reads, one transaction, then the Ready update alongside the next payload read. + /// + public async Task FinishStepAsync(string runKey, Finish finish, string owner, TimeSpan lease, + CancellationToken ct = default) + { + var claim = new StepClaim(owner, lease); + Task?>? payload = null; + var outcome = await FinishChunkAsync(runKey, [finish], 'R', ct, claim: claim, + afterSubmit: () => payload = claim.Claimed is { Seq: not AggregateSeq } next ? TryGetPayloadAsync(runKey, next.Seq, ct) : null); + return new StepResult(outcome, claim.Claimed, payload == null ? null : await payload); + } + + private async Task?> TryGetPayloadAsync(string runKey, int seq, CancellationToken ct) + { + try { return await GetPayloadAsync(runKey, seq, ct); } + catch (Exception ex) when (ex is not OperationCanceledException) { return null; } + } + + private async Task FirstAsync(char state, string runKey, CancellationToken ct) + { + await foreach (var r in Range(runKey, state, 1, ct)) return r; + return null; + } + + /// Rows the caller has just read (by seq), used on the first attempt instead of reading each + /// one again; the transaction is still guarded by their ETags, and a retry reads afresh. + private async Task FinishChunkAsync(string runKey, IReadOnlyList chunk, char? fromState, + CancellationToken ct, Dictionary? known = null, StepClaim? claim = null, Action? afterSubmit = null) + { + for (var attempt = 0; attempt < ConflictRetries; attempt++) + { + if (attempt > 0) known = null; + StoreRow? headerRow, nextRow = null; + if (claim != null) + { + var headerRead = _store.GetAsync(_work, runKey, HeaderKey, ct); + var stepRead = _store.GetAsync(_work, runKey, Key('R', chunk[0].Seq), ct); + var nextRead = FirstAsync('P', runKey, ct); + await Task.WhenAll(headerRead, stepRead, nextRead); + (headerRow, nextRow) = (headerRead.Result, nextRead.Result); + known = stepRead.Result is { } stepRow ? new() { [chunk[0].Seq] = stepRow } : []; + } + else headerRow = await _store.GetAsync(_work, runKey, HeaderKey, ct); + if (headerRow == null) return null; + var header = RunHeader.FromRow(headerRow); + var ops = new List(); + var applied = 0; + var barrier = false; + var completed = false; + + // An entity may appear once in a transaction: a task finished twice in one batch (a cancel and a + // completion landing together, or a retried finish) is applied once. + var seen = new HashSet(StringComparer.Ordinal); + foreach (var f in chunk) + { + if (!seen.Add(f.ChildKey is { } ck ? $"C|{ck}" : Seq(f.Seq))) continue; + if (f.ChildKey is { } child) + { + var placeholder = await _store.GetAsync(_work, runKey, $"C|{child}", ct); + if (placeholder == null) continue; + ops.Add(StoreOp.Delete(placeholder)); + header.Done++; + if (f.Status != "Completed") header.Failed++; + applied++; + continue; + } + + StoreRow? row = known != null && known.TryGetValue(f.Seq, out var seenRow) ? seenRow : null; + foreach (var state in row != null ? [] : fromState is { } s ? [s] : new[] { 'R', 'P' }) + { + row = await _store.GetAsync(_work, runKey, Key(state, f.Seq), ct); + if (row != null) break; + } + if (row == null) continue; + if (f.Owner != null && row.RowKey[0] == 'R' && row.GetString("Owner") != f.Owner) continue; + + ops.Add(StoreOp.Delete(row)); + if (f.Seq == AggregateSeq) + { + header.PostExecStatus = f.Status == "Completed" ? "Completed" : "Failed"; + completed = true; + } + else + { + ops.Add(StoreOp.Insert(new StoreRow(runKey, Key('D', f.Seq)) + { + Properties = + { + ["TaskId"] = row.GetString("TaskId"), + ["Status"] = f.Status, + ["LastError"] = f.Error, + ["Attempt"] = row.GetInt32("Attempt") ?? 0, + ["CompletedUtc"] = DateTimeOffset.UtcNow, + } + })); + header.Done++; + if (f.Status == "Failed") header.Failed++; + else if (f.Status == "Cancelled") header.Cancelled++; + } + applied++; + } + + if (applied == 0) return new FinishOutcome(header, false, false, 0); + + var now = DateTimeOffset.UtcNow; + var claimable = claim != null && !header.CancelRequested + && !(header.DriverOwner is { } driver && driver != claim.Owner && header.DriverLease > now); + var until = now.Add(claim?.Lease ?? TimeSpan.Zero); + ClaimedTask? next = null; + + if (!completed && header.Phase == RunPhase.Tasks && header.Done >= header.Total) + { + barrier = true; + if (header.HasPostExec) + { + header.Phase = RunPhase.Aggregate; + header.PostExecStatus = "Pending"; + if (claimable) + { + ops.Add(StoreOp.Insert(RunningRow(runKey, AggregateSeq, "PostExecution", 1, claim!.Owner, until))); + next = new ClaimedTask(runKey, AggregateSeq, "PostExecution", 1); + } + else ops.Add(StoreOp.Insert(PendingRow(runKey, AggregateSeq, "PostExecution", 0))); + } + else completed = true; + } + else if (claimable && !completed && nextRow != null) + { + var seq = SeqOf(nextRow.RowKey); + var taskId = nextRow.GetString("TaskId")!; + var nextAttempt = (nextRow.GetInt32("Attempt") ?? 0) + 1; + ops.Add(StoreOp.Delete(nextRow)); + ops.Add(StoreOp.Insert(RunningRow(runKey, seq, taskId, nextAttempt, claim!.Owner, until))); + next = new ClaimedTask(runKey, seq, taskId, nextAttempt); + } + if (next != null) + { + header.DriverOwner = claim!.Owner; + header.DriverLease = until; + } + if (completed) + { + header.Phase = RunPhase.Done; + header.Status = header.Failed > 0 || header.Cancelled > 0 ? "CompletedWithErrors" : "Completed"; + header.CompletedUtc = DateTime.UtcNow; + } + ops.Add(StoreOp.Replace(header.ToRow(headerRow.ETag))); + + await _rate.TakeAsync(runKey, ops.Count, ct); + if (await _store.TrySubmitAsync(_work, runKey, ops, ct)) + { + if (claim != null) + { + claim.Claimed = next; + afterSubmit?.Invoke(); + } + // The step path wrote the header under its ETag, so what it wrote is the state after it. + var after = claim != null ? header : (await GetRunAsync(runKey, ct)) ?? header; + if (completed) await RetireAsync(after, ct); + else await IndexAsync("update its Ready counts", runKey, () => PublishReadyAsync(after, ct), ct); + Changed(runKey); + var outcome = new FinishOutcome(after, barrier, completed, applied); + if ((barrier || completed) && AfterFinish is { } hook) + { + try { await hook(outcome); } + catch (Exception ex) { _logger.LogWarning(ex, "[WorkStore] After-finish handling for {Run} failed", runKey); } + } + return outcome; + } + } + + throw new InvalidOperationException($"Finishing {chunk.Count} task(s) of {runKey} kept losing races to other writers"); + } + + /// Take a finished run off the Ready list and record it for retention. + /// Retire a finished run: the Finished row first (so retention will always find it), then off the + /// Ready list, then off the active list last. Each step is idempotent, so the repair can redo any of them. + private async Task RetireAsync(RunHeader h, CancellationToken ct) + { + _rate.Forget(h.RunKey); + await IndexAsync("record it for retention", h.RunKey, () => RecordFinishedAsync(h, ct), ct); + await IndexAsync("take it off the Ready list", h.RunKey, () => _store.DeleteAsync(_ready, ReadyPartition(h.Priority), ReadyKey(h), ct), ct); + await IndexAsync("take it off the active list", h.RunKey, () => _store.DeleteAsync(_names, ActivePartition, h.RunKey, ct), ct); + } + + private Task RecordFinishedAsync(RunHeader h, CancellationToken ct) + { + var done = (h.CompletedUtc ?? DateTime.UtcNow).Ticks.ToString("D19", CultureInfo.InvariantCulture); + return _store.UpsertAsync(_finished, new StoreRow("F", $"{done}|{h.RunKey}") { Properties = { ["RunKey"] = h.RunKey } }, ct); + } + + private static readonly TimeSpan[] IndexRetryDelays = + [TimeSpan.FromMilliseconds(200), TimeSpan.FromMilliseconds(500), TimeSpan.FromSeconds(1), TimeSpan.FromSeconds(2), TimeSpan.FromSeconds(4)]; + + /// Index writes that failed for good; a scheduler repairs the indexes when this moves. + public int IndexFailures => Volatile.Read(ref _indexFailures); + private int _indexFailures; + + /// Delays between index-write retries; tests shorten them. + internal TimeSpan[] IndexRetries { get; set; } = IndexRetryDelays; + + /// + /// An index write that follows a committed change to a run. Retried through a short storage blip; if it + /// still fails the change stands, the in-process event keeps this instance + /// scheduling correctly, and (run at startup, or on demand) puts the + /// index right. Logged as an error naming the run, so a run missing from a listing has an explanation. + /// + private async Task IndexAsync(string what, string runKey, Func write, CancellationToken ct) + { + for (var attempt = 0; ; attempt++) + { + try + { + await write(); + return; + } + catch (Exception ex) when (ex is not OperationCanceledException || !ct.IsCancellationRequested) + { + if (attempt >= IndexRetries.Length) + { + Interlocked.Increment(ref _indexFailures); + _logger.LogError(ex, "[WorkStore] Could not {What} for run {Run}; the index repair (startup, or RepairIndexes) fixes it", + what, runKey); + return; + } + await Task.Delay(IndexRetries[attempt], ct); + } + } + } + + /// Remove a stale Ready entry (its run is gone or finished). + public Task DropReadyAsync(ReadyEntry e, CancellationToken ct = default) => + _store.DeleteAsync(_ready, ReadyPartition(e.Band), $"{e.StartedUtc.Ticks.ToString("D19", CultureInfo.InvariantCulture)}|{e.RunKey}", ct); + + // ── children ── + + /// + /// Make wait for a child run. Only while the parent is still running its tasks: + /// a run queued from an aggregation is not a child. + /// + public async Task AddChildAsync(string parentKey, string childKey, CancellationToken ct = default) + { + for (var attempt = 0; attempt < ConflictRetries; attempt++) + { + var headerRow = await _store.GetAsync(_work, parentKey, HeaderKey, ct); + if (headerRow == null) return false; + var header = RunHeader.FromRow(headerRow); + if (header.Phase != RunPhase.Tasks) return false; + header.Total++; + var ops = new List + { + StoreOp.Insert(new StoreRow(parentKey, $"C|{childKey}") { Properties = { ["Child"] = childKey } }), + StoreOp.Replace(header.ToRow(headerRow.ETag)), + }; + if (await _store.TrySubmitAsync(_work, parentKey, ops, ct)) + { + Changed(parentKey); + return true; + } + } + return false; + } + + // ── cancel ── + + /// Cancel every pending task of a run. Running tasks finish; the barrier then fires as usual. + private readonly ConcurrentDictionary _cancelling = new(StringComparer.Ordinal); + + /// Whether a full cancel of the run is in progress in this process (the scheduler leaves it alone). + public bool IsCancelling(string runKey) => _cancelling.ContainsKey(runKey); + + /// Pages of up to 49 to cancel before returning; the scheduler does one per visit to finish + /// a cancel that was interrupted. A full cancel (no bound) marks the run as being cancelled while it works. + public async Task<(int Cancelled, FinishOutcome? Outcome)> CancelPendingAsync(string runKey, + string reason = "Cancelled by user", int maxPages = int.MaxValue, CancellationToken ct = default) + { + if (maxPages != int.MaxValue) return await CancelPagesAsync(runKey, reason, maxPages, ct); + _cancelling.AddOrUpdate(runKey, 1, (_, n) => n + 1); + try + { + return await CancelPagesAsync(runKey, reason, maxPages, ct); + } + finally + { + if (_cancelling.AddOrUpdate(runKey, 0, (_, n) => n - 1) <= 0) _cancelling.TryRemove(runKey, out _); + } + } + + private async Task<(int Cancelled, FinishOutcome? Outcome)> CancelPagesAsync(string runKey, string reason, int maxPages, + CancellationToken ct) + { + var cancelled = 0; + var idle = 0; + FinishOutcome? outcome = null; + for (var pages = 0; pages < maxPages; pages++) + { + var page = new List(); + var rows = new Dictionary(); + await foreach (var r in Range(runKey, 'P', MaxPerTransaction, ct: ct)) + { + var seq = SeqOf(r.RowKey); + if (seq == AggregateSeq) continue; + page.Add(new Finish(seq, "Cancelled", reason)); + rows[seq] = r; + } + if (page.Count == 0) return (cancelled, outcome); + var result = await FinishChunkAsync(runKey, page, 'P', ct, rows); + if (result == null) return (cancelled, outcome); + // Nothing applied means another writer cancelled (or claimed) this page first, not that none is left: + // only an empty pending range ends the loop. Bounded, so rows that somehow cannot be cancelled do not spin. + if (result.Applied == 0) + { + if (++idle >= 3) return (cancelled, outcome); + continue; + } + idle = 0; + cancelled += result.Applied; + outcome = result; + } + return (cancelled, outcome); + } + + // ── the instance lock ── + + private const string LockPartition = "$instance", LockKey = "lock"; + + /// Who holds the instance lock, and until when (null when nobody does). + public sealed record InstanceLock(string Owner, DateTimeOffset LeaseUntil, DateTimeOffset AcquiredUtc); + + public async Task GetInstanceLockAsync(CancellationToken ct = default) + { + await InitializeAsync(ct); + var row = await _store.GetAsync(_names, LockPartition, LockKey, ct); + return row?.GetString("Owner") is { } owner + ? new InstanceLock(owner, row.GetDateTimeOffset("LeaseUntil") ?? DateTimeOffset.MinValue, + row.GetDateTimeOffset("AcquiredUtc") ?? DateTimeOffset.MinValue) + : null; + } + + /// + /// Take or keep the instance lock: the one row that says which process works the queue. Succeeds when + /// nobody holds it, its lease has run out, or already holds it (a renewal). Guarded + /// by the row's ETag, so two processes racing for it cannot both win. Returns the holder afterwards. + /// + public async Task<(bool Held, InstanceLock? Holder)> TryHoldInstanceLockAsync(string owner, TimeSpan lease, + CancellationToken ct = default) + { + await InitializeAsync(ct); + var now = DateTimeOffset.UtcNow; + var row = await _store.GetAsync(_names, LockPartition, LockKey, ct); + var holder = row?.GetString("Owner"); + var until = row?.GetDateTimeOffset("LeaseUntil") ?? DateTimeOffset.MinValue; + if (row != null && holder != owner && until > now) + return (false, new InstanceLock(holder!, until, row.GetDateTimeOffset("AcquiredUtc") ?? DateTimeOffset.MinValue)); + + var acquired = holder == owner ? row!.GetDateTimeOffset("AcquiredUtc") ?? now : now; + var next = new StoreRow(LockPartition, LockKey) + { + ETag = row?.ETag, + Properties = { ["Owner"] = owner, ["LeaseUntil"] = now.Add(lease), ["AcquiredUtc"] = acquired }, + }; + var ok = await _store.TrySubmitAsync(_names, LockPartition, [row == null ? StoreOp.Insert(next) : StoreOp.Replace(next)], ct); + return ok ? (true, new InstanceLock(owner, now.Add(lease), acquired)) : (false, await GetInstanceLockAsync(ct)); + } + + /// Give the instance lock up, if still holds it, so a successor starts at once. + /// False when still held it and the delete lost a race. + public async Task ReleaseInstanceLockAsync(string owner, CancellationToken ct = default) + { + var row = await _store.GetAsync(_names, LockPartition, LockKey, ct); + return row?.GetString("Owner") != owner || await _store.TrySubmitAsync(_names, LockPartition, [StoreOp.Delete(row)], ct); + } + + // ── index repair ── + + /// What a repair pass found and did. + public sealed record RepairResult(int Active, int Relisted, int Retired, int Removed, int Young); + + /// + /// Rebuild the indexes from the active-run list, which is written before a run's other rows and removed + /// after them. For each run on it: a run that finished is retired (Finished row, off Ready, off the list); + /// a run still going gets its Ready entry rewritten; a run whose header never landed (its creation was + /// interrupted) is removed once it is older than . Every step is + /// idempotent, so this is safe while the queue is being worked. One point read per active run. + /// + public async Task RepairIndexesAsync(TimeSpan abandonAfter, CancellationToken ct = default) + { + await InitializeAsync(ct); + var rows = new List(); + await foreach (var row in _store.QueryPartitionAsync(_names, ActivePartition, ct)) rows.Add(row); + + int relisted = 0, retired = 0, removed = 0, young = 0; + var cutoff = DateTimeOffset.UtcNow - abandonAfter; + foreach (var row in rows) + { + var runKey = row.RowKey; + var header = await GetRunAsync(runKey, ct); + if (header == null) + { + if ((row.GetDateTimeOffset("StartedUtc") ?? DateTimeOffset.MinValue) > cutoff) { young++; continue; } + _logger.LogWarning("[WorkStore] Removing run {Run}: its creation never finished", runKey); + await _store.DeletePartitionAsync(_work, runKey, ct); + await _store.DeletePartitionAsync(_results, runKey, ct); + await _store.DeleteAsync(_names, ActivePartition, runKey, ct); + removed++; + } + else if (header.IsFinished) + { + await RecordFinishedAsync(header, ct); + await _store.DeleteAsync(_ready, ReadyPartition(header.Priority), ReadyKey(header), ct); + await _store.DeleteAsync(_names, ActivePartition, runKey, ct); + retired++; + } + else + { + await PublishReadyAsync(header, ct); + Changed(runKey); + relisted++; + } + } + return new RepairResult(rows.Count, relisted, retired, removed, young); + } + + // ── inspection ── + + /// Whether the run has its entry on the Ready list (the scheduler only sees runs that do). + public async Task IsListedAsync(RunHeader h, CancellationToken ct = default) => + await _store.GetAsync(_ready, ReadyPartition(h.Priority), ReadyKey(h), ct) != null; + + /// The run's child placeholders: runs it is waiting for (at most ). + public async Task> GetChildWaitsAsync(string runKey, int max = 1000, CancellationToken ct = default) + { + var children = new List(); + await foreach (var r in Range(runKey, 'C', max, ct)) children.Add(r.RowKey[2..]); + return children; + } + + // ── retention ── + + /// Delete runs that finished before the retention cutoff: their Work and Results partitions and + /// index rows. Reads the Finished index oldest first and stops at the cutoff. + public async Task SweepFinishedAsync(TimeSpan retention, CancellationToken ct = default) + { + var cutoff = (DateTime.UtcNow - retention).Ticks.ToString("D19", CultureInfo.InvariantCulture); + var expired = new List(); + await foreach (var row in _store.QueryRowKeyRangeAsync(_finished, "F", "", cutoff, ct: ct)) + expired.Add(row); + + foreach (var row in expired) + { + var runKey = row.GetString("RunKey") ?? row.RowKey[20..]; + await DeleteRunAsync(runKey, ct); + await _store.DeleteAsync(_finished, "F", row.RowKey, ct); + } + return expired.Count; + } + + /// Delete a run's partitions and its name entry if it still points at this run. + public async Task DeleteRunAsync(string runKey, CancellationToken ct = default) + { + var header = await GetRunAsync(runKey, ct); + await _store.DeletePartitionAsync(_work, runKey, ct); + await _store.DeletePartitionAsync(_results, runKey, ct); + await _store.DeleteAsync(_names, ActivePartition, runKey, ct); + if (header == null) return; + var name = await _store.GetAsync(_names, "N", header.Name, ct); + if (name?.GetString("RunKey") == runKey) await _store.DeleteAsync(_names, "N", header.Name, ct); + await _store.DeleteAsync(_ready, ReadyPartition(header.Priority), ReadyKey(header), ct); + } + + /// Delete the tables of the previous orchestration design. Their in-flight work is dropped; the timers + /// that created it create it again. + public async Task DropLegacyTablesAsync(CancellationToken ct = default) + { + foreach (var t in _legacyTables) + { + try { await _store.DeleteTableAsync(t, ct); } + catch (Exception ex) { _logger.LogWarning(ex, "[WorkStore] Could not drop legacy table {Table}", t); } + } + } + + /// Flag a run as cancelled, so its running tasks and a sequential driver stop at their next step. + public Task RequestCancelAsync(string runKey, CancellationToken ct = default) => + UpdateHeaderAsync(runKey, h => { h.CancelRequested = true; return true; }, ct); + + /// Move a run to another priority band (its Ready entry moves with it). + public async Task SetPriorityAsync(string runKey, int priority, CancellationToken ct = default) + { + RunHeader? before = null; + var ok = await UpdateHeaderAsync(runKey, h => + { + before ??= new RunHeader { RunKey = h.RunKey, Name = h.Name, Priority = h.Priority, StartedUtc = h.StartedUtc }; + if (h.IsFinished) return false; + h.Priority = priority; + return true; + }, ct); + if (!ok || before == null) return false; + await _store.DeleteAsync(_ready, ReadyPartition(before.Priority), ReadyKey(before), ct); + if (await GetRunAsync(runKey, ct) is { IsFinished: false } after) await PublishReadyAsync(after, ct); + Changed(runKey); + return true; + } + + private async Task UpdateHeaderAsync(string runKey, Func change, CancellationToken ct) + { + for (var attempt = 0; attempt < ConflictRetries; attempt++) + { + var row = await _store.GetAsync(_work, runKey, HeaderKey, ct); + if (row == null) return false; + var header = RunHeader.FromRow(row); + if (!change(header)) return false; + if (await _store.TrySubmitAsync(_work, runKey, [StoreOp.Replace(header.ToRow(row.ETag))], ct)) return true; + } + return false; + } +} + +public enum RunPhase { Tasks, Aggregate, Done } + +/// A run's header row: identity, options and counts. +public sealed class RunHeader +{ + internal static readonly JsonSerializerOptions Json = new() { PropertyNamingPolicy = JsonNamingPolicy.CamelCase }; + + public string RunKey { get; init; } = ""; + public string Name { get; init; } = ""; + public string Status { get; set; } = "Running"; + public RunPhase Phase { get; set; } = RunPhase.Tasks; + public int Priority { get; set; } = 4; + public DateTime StartedUtc { get; init; } + public DateTime? CompletedUtc { get; set; } + public string? TaskScriptName { get; init; } + public string? PostExecFunctionName { get; init; } + /// Set on creation only; read back with . + public string? PostExecParametersJson { get; init; } + public string? PostExecStatus { get; set; } + public string? ParentRunKey { get; init; } + + /// The placeholder this run fills in its parent (C|{key}), completed when this run finishes. + public string? ParentChildKey { get; init; } + public string? Reference { get; init; } + + /// Sequential runs: the worker driving the run, and until when. One driver at a time. + public string? DriverOwner { get; set; } + public DateTimeOffset? DriverLease { get; set; } + public bool Sequential { get; init; } + + /// At most this many of the run's tasks run at once; 0 is no limit. Not used with . + public int MaxConcurrency { get; init; } + + /// Sequential runs: the first failed step cancels the steps after it instead of carrying on. + public bool StopOnFailure { get; init; } + public bool CancelRequested { get; set; } + public int Total { get; set; } + public int Done { get; set; } + public int Failed { get; set; } + public int Cancelled { get; set; } + + public bool HasPostExec => !string.IsNullOrEmpty(PostExecFunctionName); + public bool IsFinished => Phase == RunPhase.Done; + + public StoreRow ToRow(string? etag = null) => new(RunKey, WorkStore.HeaderKey) + { + ETag = etag, + Properties = + { + ["Name"] = Name, + ["Status"] = Status, + ["Phase"] = Phase.ToString(), + ["Priority"] = Priority, + ["StartedUtc"] = new DateTimeOffset(DateTime.SpecifyKind(StartedUtc, DateTimeKind.Utc)), + ["CompletedUtc"] = CompletedUtc is { } c ? new DateTimeOffset(DateTime.SpecifyKind(c, DateTimeKind.Utc)) : (DateTimeOffset?)null, + ["TaskScriptName"] = TaskScriptName, + ["PostExecFunctionName"] = PostExecFunctionName, + ["PostExecStatus"] = PostExecStatus, + ["ParentRunKey"] = ParentRunKey, + ["ParentChildKey"] = ParentChildKey, + ["Reference"] = Reference, + ["DriverOwner"] = DriverOwner, + ["DriverLease"] = DriverLease, + ["Sequential"] = Sequential ? 1 : 0, + ["MaxConcurrency"] = MaxConcurrency, + ["StopOnFailure"] = StopOnFailure ? 1 : 0, + ["CancelRequested"] = CancelRequested ? 1 : 0, + ["Total"] = Total, + ["Done"] = Done, + ["Failed"] = Failed, + ["Cancelled"] = Cancelled, + } + }; + + public static RunHeader FromRow(StoreRow r) => new() + { + RunKey = r.PartitionKey, + Name = r.GetString("Name") ?? r.PartitionKey, + Status = r.GetString("Status") ?? "Running", + Phase = Enum.TryParse(r.GetString("Phase"), out var phase) ? phase : RunPhase.Tasks, + Priority = r.GetInt32("Priority") ?? 4, + StartedUtc = r.GetDateTimeOffset("StartedUtc")?.UtcDateTime ?? DateTime.UnixEpoch, + CompletedUtc = r.GetDateTimeOffset("CompletedUtc")?.UtcDateTime, + TaskScriptName = r.GetString("TaskScriptName"), + PostExecFunctionName = r.GetString("PostExecFunctionName"), + PostExecStatus = r.GetString("PostExecStatus"), + ParentRunKey = r.GetString("ParentRunKey"), + ParentChildKey = r.GetString("ParentChildKey"), + Reference = r.GetString("Reference"), + DriverOwner = r.GetString("DriverOwner"), + DriverLease = r.GetDateTimeOffset("DriverLease"), + Sequential = r.GetInt32("Sequential") == 1, + MaxConcurrency = r.GetInt32("MaxConcurrency") ?? 0, + StopOnFailure = r.GetInt32("StopOnFailure") == 1, + CancelRequested = r.GetInt32("CancelRequested") == 1, + Total = r.GetInt32("Total") ?? 0, + Done = r.GetInt32("Done") ?? 0, + Failed = r.GetInt32("Failed") ?? 0, + Cancelled = r.GetInt32("Cancelled") ?? 0, + }; +} + +/// +/// Keeps each partition under Azure's ~2,000 entities/s target: a token bucket per partition at +/// , refilled continuously. Callers await their cost before touching the partition. +/// +public sealed class PartitionRateLimiter(int perSecond = 1_900) +{ + public int PerSecond { get; } = perSecond; + private readonly ConcurrentDictionary _buckets = new(StringComparer.Ordinal); + + public void Forget(string partition) => _buckets.TryRemove(partition, out _); + + public async Task TakeAsync(string partition, int cost, CancellationToken ct = default) + { + var bucket = _buckets.GetOrAdd(partition, _ => new Bucket(PerSecond)); + while (true) + { + var wait = bucket.TryTake(Math.Min(cost, PerSecond), PerSecond); + if (wait <= TimeSpan.Zero) return; + await Task.Delay(wait, ct); + } + } + + private sealed class Bucket(double tokens) + { + private double _tokens = tokens; + private long _stamp = Environment.TickCount64; + + public TimeSpan TryTake(int cost, int perSecond) + { + lock (this) + { + var now = Environment.TickCount64; + _tokens = Math.Min(perSecond, _tokens + (now - _stamp) * perSecond / 1000.0); + _stamp = now; + if (_tokens >= cost) { _tokens -= cost; return TimeSpan.Zero; } + return TimeSpan.FromMilliseconds(Math.Ceiling((cost - _tokens) * 1000.0 / perSecond)); + } + } + } +} diff --git a/appsettings.example.jsonc b/appsettings.example.jsonc index 76198c7..521e0e8 100644 --- a/appsettings.example.jsonc +++ b/appsettings.example.jsonc @@ -173,6 +173,9 @@ // docs/dispatch-analysis.md). Set false only to A/B or if a module misbehaves on a long-lived thread. // "ReuseRunspaceThread": true, + // Minutes between timed memory trims (compacting full GC). 0 = disabled. Default 5. + // "MemoryTrimIntervalMinutes": 5, + // Host-tier pool sizing. When set, the first matching entry overrides // HttpPoolSize/BgPoolSize above based on the runtime environment. // SkuEnv: name of the env var to read for the host tier identifier diff --git a/docs/configuration.md b/docs/configuration.md index ee195f7..853d93c 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -179,6 +179,10 @@ Controls the PowerShell runspace pools that execute all scripts. // only to A/B or if a module misbehaves on a long-lived pipeline thread. "ReuseRunspaceThread": true, + // Minutes between timed memory trims (compacting full GC that hands freed heap back to the OS). + // Runs alongside the every-100-invocations trim, sharing its 2-minute cooldown. 0 = disabled. + "MemoryTrimIntervalMinutes": 5, + // Maximum execution time (seconds) for HTTP request handlers. // When exceeded, the PowerShell pipeline is stopped and the worker is reclaimed. // 0 = no timeout (default). Recommended: 120-300 for HTTP endpoints. @@ -370,27 +374,16 @@ By default it starts narrow and ramps slowly, to keep idle memory low; tune it f ### Orchestrator -Fan-out/fan-in task execution with crash recovery. +Fan-out/fan-in task execution. Every run's state lives in storage — one partition of `{Prefix}Work` per +run, where each task is one row and every state change is one partition transaction — so a restart has +nothing to recover: claims held by a stopped process lapse and are taken again. ```jsonc "Orchestrator": { - // Prefix for Azure Tables: {Prefix}Runs, {Prefix}Tasks, {Prefix}Results + // Prefix for Azure Tables: {Prefix}Work, {Prefix}Ready, {Prefix}Names, {Prefix}Finished, {Prefix}TaskResults. + // The previous design's {Prefix}Queue/QueueIndex/Tasks/Runs/Results tables are dropped at startup. "TablePrefix": "Orchestrator", - // Batch + coalesce per-task/run STATUS writes off the fan-out critical path, in ≤100-entity byte-budgeted - // Azure Table transactions. Default true. This is the throughput fix for large fan-outs — the per-task - // table write was the ceiling (see docs/orch-analysis.md). Results are NEVER batched (their chunking / - // multi-row large-payload path is untouched). Set false to fall back to per-task writes. - "BatchStatusWrites": true, - // Write the pre-invoke "Running" marker under a durable barrier (persisted BEFORE the task runs, batched - // with concurrently-starting tasks) so AttemptCount/MaxRetries still bounds poison tasks. Default true. - // False = eventual: the marker rides the periodic flush and the task doesn't wait — max throughput (100% - // pool utilization) at the cost of the strict poison-before-invoke guarantee. Terminal + run states stay - // durable in both modes (flushed before a run finalizes and on shutdown). - "DurableRunningBarrier": true, - // Status-writer flush interval / barrier latency ceiling (ms). Default 25. - "StatusFlushIntervalMs": 25, - // PS function that executes individual tasks. Receives TaskJson parameter. // Default: "Invoke-CraftTask" (provided in CraftRuntime/) "GenericTaskFunction": "Invoke-CraftTask", @@ -407,14 +400,13 @@ Fan-out/fan-in task execution with crash recovery. // Default: "Invoke-CraftPostExecution" (provided in CraftRuntime/) "PostExecFunction": "Invoke-CraftPostExecution", - // Max task interruptions (host crash/restart) before marking Failed. + // Max task interruptions (host crash/restart) before marking Failed; also PostExecution attempts. "MaxRetries": 3, - // Retention sweep over the three tables. A run that finished — or that nothing is driving and that - // last wrote to storage — longer ago than RetentionHours is removed together with its Tasks/Results - // partitions, as is any Tasks/Results partition whose Run row is already gone. Runs once at startup - // (after crash recovery) and then every CleanupIntervalHours; 0 keeps only the startup pass. Craft - // needs the rows only while a run is live — the retention is for operators reading recent history. + // Retention sweep. A run that finished longer ago than RetentionHours is removed with its Work and + // Results partitions. Runs once at startup and then every CleanupIntervalHours; 0 keeps only the + // startup pass. Craft needs the rows only while a run is live — the retention is for operators + // reading recent history. "RetentionHours": 48, "CleanupIntervalHours": 4 } diff --git a/perf-harness/api-harness/API/Modules/PerfApi/PerfApi.psd1 b/perf-harness/api-harness/API/Modules/PerfApi/PerfApi.psd1 index 28b8e31..3d33224 100644 --- a/perf-harness/api-harness/API/Modules/PerfApi/PerfApi.psd1 +++ b/perf-harness/api-harness/API/Modules/PerfApi/PerfApi.psd1 @@ -5,7 +5,7 @@ Author = 'CRAFT perf-harness' Description = 'Synthetic HTTP endpoints for load-testing CRAFT in http-only mode. Not for production.' PowerShellVersion = '7.2' - FunctionsToExport = @('Invoke-PerfPing', 'Invoke-PerfEcho', 'Invoke-PerfCpu', 'Invoke-PerfSleep', 'Invoke-PerfJson', 'Invoke-PerfFile', 'Invoke-PerfBgEnqueue', 'Push-PerfBg', 'Push-PerfBgLeaf', 'Invoke-PerfManyRuns', 'Push-PerfHold', 'Invoke-PerfThreads', 'Invoke-PerfTableOp', 'Push-PerfCheck', 'Invoke-PerfCheckCounts', 'Push-PerfSeq', 'Invoke-PerfSeqResult', 'Push-PerfSeqWorker', 'Invoke-PerfSeqWorkerEnqueue', 'Invoke-PerfSeqWorkerResult', 'Invoke-PerfThreadBreakdown', 'Invoke-PerfSeedRuns', 'Invoke-ListPerf', 'Invoke-PerfWhoami', 'Invoke-PerfTimerTick', 'Invoke-PerfTimerCount', 'Invoke-PerfPublish', 'Invoke-PerfAllocation', 'Invoke-PerfRuns') + FunctionsToExport = @('Invoke-PerfPing', 'Invoke-PerfEcho', 'Invoke-PerfCpu', 'Invoke-PerfSleep', 'Invoke-PerfJson', 'Invoke-PerfFile', 'Invoke-PerfBgEnqueue', 'Push-PerfBg', 'Push-PerfBgLeaf', 'Invoke-PerfManyRuns', 'Push-PerfHold', 'Invoke-PerfThreads', 'Invoke-PerfTableOp', 'Push-PerfCheck', 'Invoke-PerfCheckCounts', 'Push-PerfSeq', 'Invoke-PerfSeqResult', 'Push-PerfSeqWorker', 'Invoke-PerfSeqWorkerEnqueue', 'Invoke-PerfSeqWorkerResult', 'Invoke-PerfThreadBreakdown', 'Invoke-PerfSeedRuns', 'Invoke-ListPerf', 'Invoke-PerfWhoami', 'Invoke-PerfTimerTick', 'Invoke-PerfTimerCount', 'Invoke-PerfPublish', 'Invoke-PerfAllocation', 'Invoke-PerfRuns', 'Push-PerfE2E', 'Push-PerfE2EPost', 'Invoke-PerfE2EStart', 'Invoke-PerfE2EState', 'Invoke-PerfE2ERuns', 'Invoke-PerfE2EBridge', 'Invoke-PerfE2ELegacy') CmdletsToExport = @() VariablesToExport = @() AliasesToExport = @() diff --git a/perf-harness/api-harness/API/Modules/PerfApi/PerfApi.psm1 b/perf-harness/api-harness/API/Modules/PerfApi/PerfApi.psm1 index 8b3661f..cb82228 100644 --- a/perf-harness/api-harness/API/Modules/PerfApi/PerfApi.psm1 +++ b/perf-harness/api-harness/API/Modules/PerfApi/PerfApi.psm1 @@ -562,3 +562,298 @@ function Invoke-PerfPublish { [Craft.Services.RealtimeBridge]::Publish($userId, $jobId, $mode, $data, "/perf/$jobId", "View job") return @{ StatusCode = 200; Body = @{ ok = $true; endpoint = 'PerfPublish'; jobId = $jobId; mode = $mode; userId = $userId } } } + +# -- Orchestration e2e probes (scripts/run-e2e-orchestration.ps1) ----------------------------------------- +# Each check gets its own namespace (ns). Tasks and PostExecutions record what they saw under that ns, in the +# shared cache by default or in the E2EProbe table (sink=table) when the record has to survive a restart. +# Records are hashtables: kind T (task: run label, idx, start/end ticks, worker, stamped priority/run), +# kind P (PostExecution: what Results and Parameters held) and kind Q (what a child enqueue returned). + +function Get-PerfE2ECache([string]$Ns) { [Craft.Services.PowerShellRunnerService]::GetSharedCache("E2E:$Ns") } + +function Get-PerfE2ETable { + $Tc = [Azure.Data.Tables.TableClient]::new($env:AzureWebJobsStorage, 'E2EProbe') + $Flags = [Craft.Services.PowerShellRunnerService]::GetSharedCache('E2EProbeInit') + if (-not $Flags['created']) { $Tc.CreateIfNotExists() | Out-Null; $Flags['created'] = $true } + $Tc +} + +function Write-PerfE2ERecord([string]$Ns, [string]$Sink, [string]$Key, [hashtable]$Fields) { + if ($Sink -eq 'table') { + $E = [Azure.Data.Tables.TableEntity]::new($Ns, $Key) + foreach ($K in $Fields.Keys) { $E[$K] = $Fields[$K] } + (Get-PerfE2ETable).UpsertEntity[Azure.Data.Tables.TableEntity]($E, [Azure.Data.Tables.TableUpdateMode]::Replace, [System.Threading.CancellationToken]::None) | Out-Null + } else { + (Get-PerfE2ECache $Ns)[$Key] = $Fields + } +} + +# Deterministic >64 KB payload; the harness recomputes it to compare length and hash. +function Get-PerfE2EBig([int]$Kb) { ('0123456789abcdef' * ($Kb * 64)) + 'END' } + +function Get-PerfE2EBatch([string]$Ns, [string]$Sink, [string]$Label, $Spec) { + $Count = [int]$Spec.tasks + @(for ($I = 0; $I -lt $Count; $I++) { + $T = @{ FunctionName = 'PerfE2E'; ns = $Ns; sink = $Sink; run = $Label; idx = $I; TenantFilter = "t$I" } + if ($Spec.holdms) { $T.holdms = [int]$Spec.holdms } + if ($Spec.task) { foreach ($K in $Spec.task.Keys) { $T[$K] = $Spec.task[$K] } } + if ($Spec.overrides -and $Spec.overrides["$I"]) { foreach ($K in $Spec.overrides["$I"].Keys) { $T[$K] = $Spec.overrides["$I"][$K] } } + $T + }) +} + +# Queue a child run from inside a task. via=start goes through Start-CraftOrchestrator; via=bridge calls +# QueueOrchestrationFromFile directly, so a 0-task batch or a collision-off skip reaches the bridge, which +# registers the child with its parent before the run is ever created. +function Start-PerfE2EChild($Item, $Ctx) { + $C = $Item.child + $Label = if ($C.label) { [string]$C.label } else { [string]$C.name } + $Batch = Get-PerfE2EBatch -Ns $Item.ns -Sink $Item.sink -Label $Label -Spec $C + if ([string]$C.via -eq 'bridge') { + $Path = Join-Path ([IO.Path]::GetTempPath()) "e2e-child-$([guid]::NewGuid().ToString('N')).jsonl" + [IO.File]::WriteAllLines($Path, [string[]]@(foreach ($B in $Batch) { ConvertTo-Json -InputObject $B -Compress -Depth 10 })) + $Parent = if ($Ctx.RunKey) { [string]$Ctx.RunKey } else { [string]$Ctx.RunName } + $Prio = if ($null -ne $C.priority) { [int]$C.priority } elseif ($null -ne $Ctx.Priority) { [int]$Ctx.Priority } else { 4 } + [Craft.Services.OrchestratorBridge]::QueueOrchestrationFromFile([string]$C.name, $Path, $Prio, $null, $null, $null, + $Parent, $false, ($C.allowCollision -ne $false)) + $Result = 'bridge' + } else { + $In = @{ OrchestratorName = [string]$C.name; Batch = $Batch } + foreach ($K in 'priority', 'sequential', 'allowCollision') { if ($C.ContainsKey($K)) { $In[$K] = $C[$K] } } + $Result = [string](Start-CraftOrchestrator -InputObject $In -WarningAction SilentlyContinue) + } + Write-PerfE2ERecord $Item.ns $Item.sink "Q|$Label|$([guid]::NewGuid().ToString('N').Substring(0, 8))" @{ + kind = 'Q'; run = $Label; parent = [string]$Item.run; result = $Result; ticks = [DateTime]::UtcNow.Ticks } +} + +# The task. Records start (and, unless it dies, end), optionally queues a child, holds, emits output, or throws. +function Push-PerfE2E { + param($Item) + $Start = [DateTime]::UtcNow.Ticks + $Ctx = Get-Variable -Name 'CraftOperationContext' -Scope Global -ValueOnly -ErrorAction SilentlyContinue + $Key = "T|$($Item.run)|$($Item.idx)|$([guid]::NewGuid().ToString('N').Substring(0, 8))" + $Rec = @{ + kind = 'T'; run = [string]$Item.run; idx = [int]$Item.idx; start = $Start; end = [long]0 + worker = [string]$Ctx.WorkerId; runName = [string]$Ctx.RunName; runKey = [string]$Ctx.RunKey + prio = $(if ($null -ne $Ctx.Priority) { [int]$Ctx.Priority } else { -1 }) + } + Write-PerfE2ERecord $Item.ns $Item.sink $Key $Rec + if ($Item.child) { Start-PerfE2EChild -Item $Item -Ctx $Ctx } + if ($Item.holdms -and [int]$Item.holdms -gt 0) { Start-Sleep -Milliseconds ([int]$Item.holdms) } + $Done = $Rec.Clone() + $Done['end'] = [DateTime]::UtcNow.Ticks + if ($Item.fail) { + $Done['failed'] = $true + Write-PerfE2ERecord $Item.ns $Item.sink $Key $Done + throw "e2e: task $($Item.run)/$($Item.idx) failed on purpose" + } + Write-PerfE2ERecord $Item.ns $Item.sink $Key $Done + $Out = if ($Item.outKb) { 'o' * ([int]$Item.outKb * 1024) } else { [string]$Item.out } + return @{ run = [string]$Item.run; idx = [int]$Item.idx; out = $Out } +} + +# The PostExecution. Records what arrived: the raw line count, each entry's run/idx and output length, and +# the Parameters (big payload length + hash, marker, nested object). followOn queues a run from here. +function Push-PerfE2EPost { + param($Item) + $Ticks = [DateTime]::UtcNow.Ticks + $P = $Item.Parameters + $Entries = [System.Collections.Generic.List[object]]::new() + foreach ($R in @($Item.Results)) { + if ($R -is [System.Collections.IList]) { foreach ($X in $R) { $Entries.Add($X) } } else { $Entries.Add($R) } + } + $Idxs = @(foreach ($E in $Entries) { if ($E -is [System.Collections.IDictionary]) { "$($E['run'])/$($E['idx'])" } else { "?$E" } }) + $Lens = @(foreach ($E in $Entries) { if ($E -is [System.Collections.IDictionary]) { ([string]$E['out']).Length } }) + $Big = [string]$P.big + $Sha = if ($Big) { [Convert]::ToHexString([Security.Cryptography.SHA256]::HashData([Text.Encoding]::UTF8.GetBytes($Big))) } else { '' } + $First = @($Item.Results)[0] + Write-PerfE2ERecord ([string]$P.ns) ([string]$P.sink) "P|$($P.run)|$([guid]::NewGuid().ToString('N').Substring(0, 8))" @{ + kind = 'P'; run = [string]$P.run; ticks = $Ticks; lines = @($Item.Results).Count; entries = $Entries.Count + idxs = ($Idxs -join ','); outLens = ($Lens -join ','); bigLen = $Big.Length; bigSha = $Sha + marker = [string]$P.marker; nested = $(if ($P.nested) { ConvertTo-Json -InputObject $P.nested -Compress -Depth 5 } else { '' }) + firstType = $(if ($null -ne $First) { $First.GetType().Name } else { '' }) + } + if ($P.followOn) { + $F = $P.followOn + $Batch = @(for ($I = 0; $I -lt [int]$F.tasks; $I++) { + @{ FunctionName = 'PerfE2E'; ns = [string]$P.ns; sink = [string]$P.sink; run = [string]$F.label; idx = $I; holdms = [int]$F.holdms; TenantFilter = "t$I" } }) + Start-CraftOrchestrator -InputObject @{ OrchestratorName = [string]$F.name; Batch = $Batch } | Out-Null + } +} + +# POST { ns, sink, runs: [ { name, label, tasks, task{}, overrides{idx:{}}, post{marker,bigKb,nested,followOn}, +# Priority, Sequential, AllowCollision, MaxConcurrency, StopOnFailure } ] }. Runs are queued in order in one +# invocation. Returns each Start-CraftOrchestrator result, its warnings, and the server tick it was called at. +function Invoke-PerfE2EStart { + param($Request, $TriggerMetadata) + $Spec = $Request.Body | ConvertTo-Json -Depth 30 -Compress | ConvertFrom-Json -AsHashtable + $Ns = [string]$Spec.ns + $Sink = [string]$Spec.sink + $Out = [System.Collections.Generic.List[object]]::new() + foreach ($R in @($Spec.runs)) { + $Label = if ($R.label) { [string]$R.label } else { [string]$R.name } + $In = @{ OrchestratorName = [string]$R.name; Batch = (Get-PerfE2EBatch -Ns $Ns -Sink $Sink -Label $Label -Spec $R) } + foreach ($K in 'Priority', 'Sequential', 'AllowCollision', 'MaxConcurrency', 'StopOnFailure', 'Reference') { + if ($R.ContainsKey($K)) { $In[$K] = $R[$K] } + } + if ($R.post) { + $Params = @{ ns = $Ns; sink = $Sink; run = $Label; marker = [string]$R.post.marker } + if ($R.post.nested) { $Params.nested = $R.post.nested } + if ($R.post.followOn) { $Params.followOn = $R.post.followOn } + if ($R.post.bigKb) { $Params.big = Get-PerfE2EBig ([int]$R.post.bigKb) } + $In.PostExecution = @{ FunctionName = 'PerfE2EPost'; Parameters = $Params } + } + $Ticks = [DateTime]::UtcNow.Ticks + $Warn = $null + $Res = Start-CraftOrchestrator -InputObject $In -WarningVariable Warn -WarningAction SilentlyContinue + $Out.Add(@{ name = [string]$R.name; label = $Label; result = [string]$Res; warning = (@($Warn) -join ' | '); enqueueTicks = $Ticks }) + } + return @{ StatusCode = 200; Body = @{ ok = $true; ns = $Ns; runs = @($Out) } } +} + +# GET ?ns=X[&source=table][&summary=1][&wipe=1] -> every record written under the namespace. +function Invoke-PerfE2EState { + param($Request, $TriggerMetadata) + $Ns = [string]$Request.Query.ns + $Rows = [System.Collections.Generic.List[object]]::new() + if ([string]$Request.Query.source -eq 'table') { + foreach ($E in (Get-PerfE2ETable).Query[Azure.Data.Tables.TableEntity]("PartitionKey eq '$Ns'")) { + $H = @{} + foreach ($K in $E.Keys) { if ($K -notin 'odata.etag', 'PartitionKey', 'RowKey', 'Timestamp') { $H[$K] = $E[$K] } } + $Rows.Add($H) + } + } else { + $C = Get-PerfE2ECache $Ns + [System.Threading.Monitor]::Enter($C.SyncRoot) + try { foreach ($V in $C.Values) { $Rows.Add($V) } } finally { [System.Threading.Monitor]::Exit($C.SyncRoot) } + if ($Request.Query['wipe']) { $C.Clear() } + } + if ($Request.Query['summary']) { + $By = @{} + foreach ($R in $Rows) { + $S = $By[$R.run] + if (-not $S) { $S = @{ run = $R.run; tasks = 0; ended = 0; posts = 0; minStart = [long]0; maxEnd = [long]0 }; $By[$R.run] = $S } + if ($R.kind -eq 'P') { $S.posts = $S.posts + 1; continue } + if ($R.kind -ne 'T') { continue } + $S.tasks = $S.tasks + 1 + if ($S.minStart -eq 0 -or $R.start -lt $S.minStart) { $S.minStart = $R.start } + if ($R.end -gt 0) { $S.ended = $S.ended + 1; if ($R.end -gt $S.maxEnd) { $S.maxEnd = $R.end } } + } + return @{ StatusCode = 200; Body = @{ ok = $true; ns = $Ns; count = $Rows.Count; runs = @($By.Values) } } + } + return @{ StatusCode = 200; Body = @{ ok = $true; ns = $Ns; count = $Rows.Count; rows = @($Rows) } } +} + +# GET ?name=X | ?prefix=X [&detail=1] -> run headers read straight from {TablePrefix}Work (the durable state); +# detail adds each task's D row (status, attempt, error) and the count of P/R/C rows still open. +function Invoke-PerfE2ERuns { + param($Request, $TriggerMetadata) + $Name = [string]$Request.Query.name + $Prefix = if ($Name) { "$Name~" } else { [string]$Request.Query.prefix } + $Upper = $Prefix.Substring(0, $Prefix.Length - 1) + [char]([int]$Prefix[-1] + 1) + $Tp = if ($env:App__Orchestrator__TablePrefix) { $env:App__Orchestrator__TablePrefix } else { 'E2EOrch' } + $Tc = [Azure.Data.Tables.TableClient]::new($env:AzureWebJobsStorage, "${Tp}Work") + $Runs = [System.Collections.Generic.List[object]]::new() + try { + foreach ($E in $Tc.Query[Azure.Data.Tables.TableEntity]("PartitionKey ge '$Prefix' and PartitionKey lt '$Upper' and RowKey eq '`$run'")) { + if ($Name -and [string]$E['Name'] -ne $Name) { continue } + $H = @{ + runKey = $E.PartitionKey; name = [string]$E['Name']; status = [string]$E['Status']; phase = [string]$E['Phase'] + priority = $E['Priority']; total = $E['Total']; done = $E['Done']; failed = $E['Failed']; cancelled = $E['Cancelled'] + postExecStatus = [string]$E['PostExecStatus']; sequential = $E['Sequential']; parentRunKey = [string]$E['ParentRunKey'] + startedTicks = $(if ($E['StartedUtc']) { ([DateTimeOffset]$E['StartedUtc']).UtcTicks } else { 0 }) + completedTicks = $(if ($E['CompletedUtc']) { ([DateTimeOffset]$E['CompletedUtc']).UtcTicks } else { 0 }) + } + if ([string]$Request.Query.detail -eq '1') { + $Tasks = [System.Collections.Generic.List[object]]::new() + $Open = @{ P = 0; R = 0; C = 0 } + foreach ($T in $Tc.Query[Azure.Data.Tables.TableEntity]("PartitionKey eq '$($E.PartitionKey)' and RowKey ge 'C|' and RowKey lt 'S'")) { + $State = $T.RowKey.Substring(0, 1) + if ($State -eq 'D') { + $Tasks.Add(@{ seq = [int]$T.RowKey.Substring(2); taskId = [string]$T['TaskId']; status = [string]$T['Status'] + attempt = $T['Attempt']; error = [string]$T['LastError'] }) + } elseif ($Open.ContainsKey($State)) { $Open[$State] = $Open[$State] + 1 } + } + $H.tasks = @($Tasks | Sort-Object { $_.seq }) + $H.open = $Open + } + $Runs.Add($H) + } + } catch { + return @{ StatusCode = 500; Body = @{ ok = $false; error = "$_" } } + } + return @{ StatusCode = 200; Body = @{ ok = $true; count = $Runs.Count; runs = @($Runs | Sort-Object { $_.startedTicks }) } } +} + +# GET ?op=caps|active|cancel|summaries|summary|jobs [&name=X][&status=S] -> the orchestration bridges. +function Invoke-PerfE2EBridge { + param($Request, $TriggerMetadata) + $Name = [string]$Request.Query.name + try { + switch ([string]$Request.Query.op) { + 'caps' { + $Longest = [Craft.Services.OrchestratorBridge].GetMethods() | Where-Object Name -eq 'QueueOrchestrationFromFile' | + Sort-Object { $_.GetParameters().Count } | Select-Object -Last 1 + $Names = @($Longest.GetParameters() | ForEach-Object Name) + $Def = (Get-Command Start-CraftOrchestrator).Definition + $Body = @{ queueFromFileParams = $Names.Count; paramNames = $Names + startHasMaxConcurrency = ($Def -match 'MaxConcurrency'); startHasStopOnFailure = ($Def -match 'StopOnFailure') } + } + 'active' { $Body = @{ active = [Craft.Services.OrchestratorBridge]::IsRunActive($Name) } } + 'queue' { $Body = @{ entries = @([Craft.Services.QueueStatusBridge]::GetRunStatus($null, $Name) | ConvertFrom-Json) } } + 'workers' { + $Body = @{ busy = @([Craft.Services.WorkerMetricsBridge]::GetSnapshot().BgPool.Workers | Where-Object IsBusy | + ForEach-Object { @{ id = $_.WorkerId; fn = $_.CurrentFunction } }) } + } + 'cancel' { $Body = @{ cancelled = [Craft.Services.WorkerMetricsBridge]::CancelRun($Name) } } + 'summaries' { + $Body = @{ runs = @([Craft.Services.WorkerMetricsBridge]::GetRunSummaries() | Where-Object { -not $Name -or $_.Name -eq $Name } | ForEach-Object { + @{ name = $_.Name; priority = $_.Priority; total = $_.Total; queued = $_.Queued; running = $_.Running + completed = $_.Completed; failed = $_.Failed; completedUtc = $_.CompletedUtc } }) } + } + 'summary' { + $S = [Craft.Services.WorkerMetricsBridge]::GetSummary() + $M = [Craft.Services.WorkerMetricsBridge]::GetSnapshot().Memory + $Body = @{ jobsQueued = $S.JobsQueued; jobsQueuedLocal = $S.JobsQueuedLocal; jobsQueuedDurable = $S.JobsQueuedDurable + jobsActive = $S.JobsActive; bgBusy = $S.BgBusy; bgPoolSize = $S.BgPoolSize; limiterMax = $S.LimiterMax + heapMB = $M.HeapMB; rssMB = $M.RssMB; committedMB = $M.CommittedMB + workingSetMB = [math]::Round([System.Diagnostics.Process]::GetCurrentProcess().WorkingSet64 / 1MB, 1) } + } + 'jobs' { + $St = if ($Request.Query.status) { [string]$Request.Query.status } else { $null } + $Rn = if ($Name) { $Name } else { $null } + $Body = @{ jobs = @([Craft.Services.WorkerMetricsBridge]::GetJobDetails($Rn, $St, 500) | ForEach-Object { + @{ id = $_.Id; runName = $_.RunName; priority = $_.Priority; status = $_.Status } }) } + } + default { return @{ StatusCode = 400; Body = @{ ok = $false; error = 'unknown op' } } } + } + } catch { + return @{ StatusCode = 500; Body = @{ ok = $false; error = "$_" } } + } + $Body.ok = $true + return @{ StatusCode = 200; Body = $Body } +} + +# GET ?op=seed|list -> create (one row each) or list the previous orchestration design's tables, which the +# engine must drop at startup. +function Invoke-PerfE2ELegacy { + param($Request, $TriggerMetadata) + $Tp = if ($env:App__Orchestrator__TablePrefix) { $env:App__Orchestrator__TablePrefix } else { 'E2EOrch' } + $Legacy = @('Queue', 'QueueIndex', 'Tasks', 'Runs', 'Results' | ForEach-Object { "$Tp$_" }) + try { + $Svc = [Azure.Data.Tables.TableServiceClient]::new($env:AzureWebJobsStorage) + if ([string]$Request.Query.op -eq 'seed') { + foreach ($T in $Legacy) { + $Tc = $Svc.GetTableClient($T) + $Tc.CreateIfNotExists() | Out-Null + $E = [Azure.Data.Tables.TableEntity]::new('seed', 'row1') + $E['Note'] = 'legacy row seeded by the e2e' + $Tc.UpsertEntity[Azure.Data.Tables.TableEntity]($E, [Azure.Data.Tables.TableUpdateMode]::Replace, [System.Threading.CancellationToken]::None) | Out-Null + } + } + $Present = @($Svc.Query() | ForEach-Object Name | Where-Object { $_ -in $Legacy }) + return @{ StatusCode = 200; Body = @{ ok = $true; legacy = $Legacy; present = $Present } } + } catch { + return @{ StatusCode = 500; Body = @{ ok = $false; error = "$_" } } + } +} diff --git a/perf-harness/docker-compose.bench.yml b/perf-harness/docker-compose.bench.yml new file mode 100644 index 0000000..1864c53 --- /dev/null +++ b/perf-harness/docker-compose.bench.yml @@ -0,0 +1,7 @@ +# Override for scripts/run-orch-bench.ps1: point the SUT at another storage account and an isolated table prefix, +# so one engine's run never reads another's tables. Layered on docker-compose.e2e-azure.yml. +services: + sut: + environment: + - AzureWebJobsStorage=${BENCH_STORAGE} + - App__Orchestrator__TablePrefix=${BENCH_PREFIX} diff --git a/perf-harness/docker-compose.bg.yml b/perf-harness/docker-compose.bg.yml index 4e429f6..cee201f 100644 --- a/perf-harness/docker-compose.bg.yml +++ b/perf-harness/docker-compose.bg.yml @@ -55,15 +55,10 @@ services: - BackgroundBurstToCeiling=${BG_BURST:-false} - BackgroundOverSubscribe=${BG_OVERSUB:-0} # Batched status writer (#3). run-orch.ps1 -NoBatch sets false to A/B the per-task-write "before". - - App__Orchestrator__BatchStatusWrites=${BATCH_WRITES:-true} - - App__Orchestrator__DurableRunningBarrier=${DURABLE_BARRIER:-true} - - App__Orchestrator__StatusFlushIntervalMs=${FLUSH_MS:-25} # Per-run status/re-drive tick cadence. run-manyruns.ps1 lowers it to compress the re-drive backoff. - App__Orchestrator__StatusTimerIntervalSeconds=${STATUS_INTERVAL:-60} # Re-drive backoff (② ). run-manyruns.ps1 flips it to A/B the backoff's effect at a fixed interval. - - App__Orchestrator__RedriveBackoff=${REDRIVE_BACKOFF:-true} # Pending-Parameters shedding (retained-memory fix). run-manyruns.ps1 flips it to A/B the memory effect. - - App__Orchestrator__ShedPendingParameters=${SHED_PARAMS:-true} # Default Warning keeps the other bg-harness runs quiet; run-manyruns.ps1 sets LOG_LEVEL=Information # to reproduce production's Info-level per-run status logging (the log flood is part of the cost under test). - CRAFT_LOG_LEVEL=${LOG_LEVEL:-Warning} diff --git a/perf-harness/docker-compose.e2e-azure.yml b/perf-harness/docker-compose.e2e-azure.yml index f1b6770..abdd232 100644 --- a/perf-harness/docker-compose.e2e-azure.yml +++ b/perf-harness/docker-compose.e2e-azure.yml @@ -40,6 +40,12 @@ services: # Realtime SSE is opt-in (off by default) — turn it on so the realtime bridge test can run. - CRAFT_REALTIME_ENABLED=true - App__Orchestrator__TablePrefix=E2EOrch + # Orchestration checks: a 60s claim lease (the minimum) so the restart check reclaims interrupted tasks in + # minutes, and a fixed BG concurrency of the full pool (no ramp-up, no HTTP-pressure throttle from the + # harness's own polling) so concurrency and ordering assertions are deterministic. + - JobQueueLeaseSeconds=60 + - BackgroundBaseConcurrency=4 + - BackgroundHttpPressureThreshold=0 - App__RateLimit__Enabled=false # Scheduler: fast tick so the timer test doesn't wait long - App__Scheduler__ConfigFile=e2e-timers.json diff --git a/perf-harness/scripts/run-e2e-orchestration.ps1 b/perf-harness/scripts/run-e2e-orchestration.ps1 new file mode 100644 index 0000000..be0772e --- /dev/null +++ b/perf-harness/scripts/run-e2e-orchestration.ps1 @@ -0,0 +1,632 @@ +<# +.SYNOPSIS + Orchestration-engine section of the e2e regression. Dot-sourced by run-e2e.ps1 (needs $base, Info, + Add-Result and Add-Skip from it); not run on its own. + +.DESCRIPTION + Every check queues its own uniquely named runs into the SUT through PerfApi (/API/PerfE2EStart), lets the + PerfE2E task / PerfE2EPost PostExecution probes record what they saw (start/end ticks, worker, stamped + priority, received results and parameters), and asserts on those records plus the durable run state read + straight from the {TablePrefix}Work table (/API/PerfE2ERuns). All waits are bounded. + + Checks that need MaxConcurrency / StopOnFailure are capability-gated and reported SKIP on an image without + them. The restart check runs last: it restarts the SUT container. + + $OrchChecks (from run-e2e.ps1) limits the run to the named groups: + post fail child prio collide attrib cancel seq order status maxconc stopfail perf restart +#> + +# Perf gates. Calibrated from 3 full runs on the reference dev box (combined role, BgPoolSize=4, cpus=2, +# Azurite) on craft:orch-v2-a4d72cf: 2x the median. Short tasks are bounded by the pump (about BgPoolSize claims +# per 1s poll, so ~4 tasks/s here), which is what the fan-out numbers measure. +$OrchGates = @{ + # Provisional after the pump wake-up and bulk-cancel fixes (2026-10-06, one run: 15.6s / 9s / 1371ms / 238MB / + # 84ms; cancel-5000 measured after the fix). About 2.5x the measured value; re-calibrate from 3 runs. + Fanout1000Sec = 40 # 1,000 no-op tasks + PostExecution, enqueue -> PostExecution ran + ManyRuns300Sec = 25 # 300 single-task runs queued at once, enqueue -> all 300 Done + Ttfs5000Ms = 3500 # 5,000-task run on a warm pump, enqueue -> first task started + RssMB = 470 # SUT RSS after the perf runs (median 233MB) + IdleClaimMs = 1500 # run queued into an idle engine -> first start (the pump wakes on a new run) + Cancel5000Sec = 30 # CancelRun on a 5,000-task run -> run finished with every pending task cancelled + SeqStepMs = 20 # sequential no-op steps, first start -> last end per step (9.6ms; 32.6ms before finish+claim shared a txn) +} + +$OrchSkipped = @{} +# pwsh -File hands "a,b" over as one string. +$OrchChecks = @($OrchChecks | ForEach-Object { "$_" -split ',' } | Where-Object { $_ }) +$OrchContainer = 'craft-e2e-az-sut' + +function New-OrchId { [guid]::NewGuid().ToString('N').Substring(0, 8) } + +function Invoke-OrchGet([string]$Path, [int]$TimeoutSec = 30) { Invoke-RestMethod "$base$Path" -TimeoutSec $TimeoutSec } + +function Start-OrchRuns([string]$Ns, $Runs, [string]$Sink = 'cache', [int]$TimeoutSec = 120) { + $Body = @{ ns = $Ns; sink = $Sink; runs = @($Runs) } | ConvertTo-Json -Depth 30 -Compress + $R = Invoke-RestMethod "$base/API/PerfE2EStart" -Method Post -ContentType 'application/json' -Body $Body -TimeoutSec $TimeoutSec + return , @($R.runs) +} + +function Get-OrchRecords([string]$Ns, [string]$Source = 'cache') { + return , @((Invoke-OrchGet "/API/PerfE2EState?ns=$Ns&source=$Source" 60).rows) +} + +# Per-run-label counts (tasks recorded, ended, posts, min start / max end ticks) without shipping every record. +function Get-OrchSummary([string]$Ns, [string]$Source = 'cache') { + $Map = @{} + foreach ($R in @((Invoke-OrchGet "/API/PerfE2EState?ns=$Ns&source=$Source&summary=1" 60).runs)) { $Map[$R.run] = $R } + return $Map +} + +function Get-OrchTasks($Rows, [string]$Run) { return , @($Rows.Where({ $_.kind -eq 'T' -and $_.run -eq $Run })) } +function Get-OrchPosts($Rows, [string]$Run) { return , @($Rows.Where({ $_.kind -eq 'P' -and $_.run -eq $Run })) } +function Get-OrchRuns([string]$Name, [switch]$Detail) { + return , @((Invoke-OrchGet "/API/PerfE2ERuns?name=$Name$(if ($Detail) { '&detail=1' })").runs) +} +function Get-OrchRunsByPrefix([string]$Prefix) { return , @((Invoke-OrchGet "/API/PerfE2ERuns?prefix=$Prefix" 60).runs) } +function Invoke-OrchBridge([string]$Op, [string]$Name = '') { Invoke-OrchGet "/API/PerfE2EBridge?op=$Op&name=$Name" } + +# Poll $Condition until it returns something truthy or the timeout lapses. Locals are prefixed so they cannot +# shadow the caller's variables the condition reads (scriptblocks resolve variables dynamically). +function Wait-Orch([scriptblock]$Condition, [int]$TimeoutSec, [int]$IntervalMs = 500) { + $WaitSw = [Diagnostics.Stopwatch]::StartNew() + while ($true) { + $WaitVal = try { & $Condition } catch { $null } + if ($WaitVal) { return [pscustomobject]@{ ok = $true; sec = [math]::Round($WaitSw.Elapsed.TotalSeconds, 1); value = $WaitVal } } + if ($WaitSw.Elapsed.TotalSeconds -ge $TimeoutSec) { break } + Start-Sleep -Milliseconds $IntervalMs + } + return [pscustomobject]@{ ok = $false; sec = [math]::Round($WaitSw.Elapsed.TotalSeconds, 1); value = $null } +} + +function Wait-OrchPosts([string]$Ns, [string[]]$Labels, [int]$TimeoutSec, [string]$Source = 'cache') { + Wait-Orch { + $S = Get-OrchSummary $Ns $Source + if (@($Labels.Where({ -not $S[$_] -or $S[$_].posts -lt 1 })).Count -eq 0) { $true } + } $TimeoutSec +} + +function Wait-OrchRunsDone([string]$Name, [int]$Count, [int]$TimeoutSec, [int]$IntervalMs = 1000) { + Wait-Orch { + $R = Get-OrchRuns $Name + if ($R.Count -ge $Count -and @($R.Where({ $_.phase -ne 'Done' })).Count -eq 0) { , $R } + } $TimeoutSec $IntervalMs +} + +function ConvertTo-OrchMs([long]$Ticks) { [math]::Round($Ticks / 10000) } + +# Highest number of intervals open at once. Ends sort before starts at the same tick. +function Get-OrchMaxOverlap($Tasks) { + $Events = [System.Collections.Generic.List[object]]::new() + foreach ($T in $Tasks) { + if ([long]$T.end -le 0) { continue } + $Events.Add([pscustomobject]@{ t = [long]$T.start; d = 1 }) + $Events.Add([pscustomobject]@{ t = [long]$T.end; d = -1 }) + } + $Max = 0; $Cur = 0 + foreach ($E in ($Events | Sort-Object t, d)) { $Cur = $Cur + $E.d; if ($Cur -gt $Max) { $Max = $Cur } } + return $Max +} + +function Get-OrchMin($Values) { ($Values | Measure-Object -Minimum).Minimum } +function Get-OrchMax($Values) { ($Values | Measure-Object -Maximum).Maximum } + +function Invoke-OrchCheck([string]$Group, [scriptblock]$Body) { + if ($OrchChecks -and $Group -notin $OrchChecks) { return } + if ($OrchSkipped[$Group]) { + foreach ($N in $OrchSkipped[$Group].names) { Add-Skip "orch-$Group" $N $OrchSkipped[$Group].reason } + return + } + Info "orchestration: $Group ..." + try { & $Body } + catch { Add-Result "orch-$Group" 'check-error' $false '-' "exception: $($_.Exception.Message) @ line $($_.InvocationInfo.ScriptLineNumber)" } +} + +function Get-OrchBig([int]$Kb) { ('0123456789abcdef' * ($Kb * 64)) + 'END' } + +# Capability gate for the features built after a4d72cf. +$OrchCaps = try { Invoke-OrchBridge 'caps' } catch { $null } +$OrchHasNew = $OrchCaps -and ([int]$OrchCaps.queueFromFileParams -ge 11) -and $OrchCaps.startHasMaxConcurrency -and $OrchCaps.startHasStopOnFailure +if (-not $OrchHasNew) { + $Why = "image lacks the feature (QueueOrchestrationFromFile has $($OrchCaps.queueFromFileParams) params; Start-CraftOrchestrator MaxConcurrency=$($OrchCaps.startHasMaxConcurrency) StopOnFailure=$($OrchCaps.startHasStopOnFailure))" + $OrchSkipped['maxconc'] = @{ reason = $Why; names = @('limit-2', 'unlimited-reaches-pool', 'limit-1-lets-p1-in', 'ignored-when-sequential') } + $OrchSkipped['stopfail'] = @{ reason = $Why; names = @('stops-at-failure', 'rest-cancelled', 'postexec-runs', 'default-runs-all') } +} +$OrchPool = [int](Invoke-OrchGet '/API/PerfAllocation').pool.bgTotal + +# -- 1. PostExecution contract ---------------------------------------------------------------------------- +Invoke-OrchCheck 'post' { + $Id = New-OrchId; $Ns = "post-$Id"; $Name = "E2EPost-$Id" + $null = Start-OrchRuns $Ns @(@{ + name = $Name; label = 'p'; tasks = 6; task = @{ holdms = 300; out = 'v' }; overrides = @{ '2' = @{ outKb = 80 } } + post = @{ marker = "mk-$Id"; bigKb = 150; nested = @{ a = 1; b = @('x', 'y'); c = @{ d = 'e' } } } + }) + $W = Wait-OrchPosts $Ns @('p') 90 + Start-Sleep -Seconds 3 # window in which a duplicate PostExecution would show up + $Rows = Get-OrchRecords $Ns + $T = Get-OrchTasks $Rows 'p'; $P = Get-OrchPosts $Rows 'p'; $H = (Get-OrchRuns $Name)[0] + if (-not $W.ok -or $P.Count -eq 0) { + Add-Result 'orch-post' 'postexec-ran' $false "$($W.sec)s" "no PostExecution within 90s; tasks=$($T.Count) status=$($H.status) phase=$($H.phase) post=$($H.postExecStatus)" + return + } + $Post = $P[0] + $Got = @($Post.idxs -split ',' | Sort-Object) -join ',' + $Exp = @(0..5 | ForEach-Object { "p/$_" } | Sort-Object) -join ',' + Add-Result 'orch-post' 'one-line-per-task' ($Post.lines -eq 6 -and $Got -eq $Exp) "$($W.sec)s" "lines=$($Post.lines) entries=$($Post.entries) idxs=$($Post.idxs) lineType=$($Post.firstType)" + $Lens = @($Post.outLens -split ',' | ForEach-Object { [int]$_ } | Sort-Object) -join ',' + Add-Result 'orch-post' 'task-output-over-64k' ($Lens -eq '1,1,1,1,1,81920') '80KB' "output lengths=$($Post.outLens) (task 2 returns 80 KB)" + $Big = Get-OrchBig 150 + $Sha = [Convert]::ToHexString([Security.Cryptography.SHA256]::HashData([Text.Encoding]::UTF8.GetBytes($Big))) + Add-Result 'orch-post' 'params-over-64k' ($Post.bigLen -eq $Big.Length -and $Post.bigSha -eq $Sha) "$([math]::Round($Big.Length / 1KB))KB" "received len=$($Post.bigLen) expected=$($Big.Length) sha-match=$($Post.bigSha -eq $Sha)" + $Nested = if ($Post.nested) { $Post.nested | ConvertFrom-Json } else { $null } + $NestOk = $Nested -and $Nested.a -eq 1 -and (@($Nested.b) -join ',') -eq 'x,y' -and $Nested.c.d -eq 'e' + Add-Result 'orch-post' 'params-intact' ($Post.marker -eq "mk-$Id" -and $NestOk) '-' "marker=$($Post.marker) nested=$($Post.nested)" + $MaxEnd = Get-OrchMax $T.end + $Gap = ConvertTo-OrchMs ($Post.ticks - $MaxEnd) + $OnceOk = $P.Count -eq 1 -and $Post.ticks -ge $MaxEnd -and $T.Count -eq 6 -and $H.status -eq 'Completed' -and $H.postExecStatus -eq 'Completed' + Add-Result 'orch-post' 'once-after-all-tasks' $OnceOk "+${Gap}ms" "posts=$($P.Count) tasks=$($T.Count) post-minus-last-task-end=${Gap}ms run=$($H.status)/$($H.postExecStatus)" +} + +# -- 2. A failing task ------------------------------------------------------------------------------------- +Invoke-OrchCheck 'fail' { + $Id = New-OrchId; $Ns = "fail-$Id"; $Name = "E2EFail-$Id" + $null = Start-OrchRuns $Ns @(@{ name = $Name; label = 'f'; tasks = 6; task = @{ holdms = 200 }; overrides = @{ '3' = @{ fail = $true } }; post = @{ marker = 'f' } }) + $W = Wait-OrchPosts $Ns @('f') 90 + Start-Sleep -Seconds 2 + $Rows = Get-OrchRecords $Ns + $T = Get-OrchTasks $Rows 'f'; $P = Get-OrchPosts $Rows 'f'; $H = (Get-OrchRuns $Name -Detail)[0] + $Failed = @($H.tasks.Where({ $_.status -eq 'Failed' })) + $RunOk = $H.phase -eq 'Done' -and $H.status -eq 'CompletedWithErrors' -and $H.failed -eq 1 -and $H.done -eq 6 -and $H.total -eq 6 -and + $Failed.Count -eq 1 -and $Failed[0].seq -eq 3 -and $Failed[0].error -match 'failed on purpose' + Add-Result 'orch-fail' 'run-completes-with-errors' $RunOk "$($W.sec)s" "status=$($H.status) phase=$($H.phase) done=$($H.done)/$($H.total) failed=$($H.failed) failedSeq=$($Failed.seq) err='$($Failed.error)'" + $Post = $P | Select-Object -First 1 + $PostOk = $P.Count -eq 1 -and $Post.lines -eq 5 -and ($Post.idxs -split ',') -notcontains 'f/3' -and $H.postExecStatus -eq 'Completed' + Add-Result 'orch-fail' 'postexec-still-runs' $PostOk '-' "posts=$($P.Count) lines=$($Post.lines) idxs=$($Post.idxs) postExec=$($H.postExecStatus)" + $PerIdx = @($T | Group-Object idx | ForEach-Object { "$($_.Name)x$($_.Count)" }) -join ',' + $Attempts = @($H.tasks | ForEach-Object { $_.attempt } | Sort-Object -Unique) -join ',' + $OnceOk = $T.Count -eq 6 -and @($T | Group-Object idx).Count -eq 6 -and $Attempts -eq '1' + Add-Result 'orch-fail' 'every-task-ran-once' $OnceOk '-' "starts per idx=$PerIdx attempts=$Attempts" +} + +# -- 3. Child-run gating ------------------------------------------------------------------------------------ +Invoke-OrchCheck 'child' { + $Id = New-OrchId; $Ns = "child-$Id" + # Blocker for the collision-off child: a run of the child's name that is still going when the child is queued. + $null = Start-OrchRuns $Ns @(@{ name = "E2EBlk-$Id"; label = 'blk'; tasks = 1; task = @{ holdms = 15000 } }) + $null = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['blk'].tasks -ge 1) { $true } } 30 + $null = Start-OrchRuns $Ns @( + @{ name = "E2EChP-$Id"; label = 'chP'; tasks = 2; task = @{ holdms = 200 }; post = @{ marker = 'chP' } + overrides = @{ '0' = @{ child = @{ name = "E2EChK-$Id"; label = 'chK'; tasks = 3; holdms = 3000 } } } } + @{ name = "E2EZ-$Id"; label = 'z'; tasks = 2; post = @{ marker = 'z' } + overrides = @{ '0' = @{ child = @{ name = "E2EZs-$Id"; label = 'zs'; tasks = 0 } } + '1' = @{ child = @{ name = "E2EZb-$Id"; label = 'zb'; tasks = 0; via = 'bridge' } } } } + @{ name = "E2ECs-$Id"; label = 'cs'; tasks = 1; post = @{ marker = 'cs' } + overrides = @{ '0' = @{ child = @{ name = "E2EBlk-$Id"; label = 'blk2'; tasks = 1; via = 'bridge'; allowCollision = $false } } } } + @{ name = "E2ESelf-$Id"; label = 'self'; tasks = 1; post = @{ marker = 'self' } + overrides = @{ '0' = @{ child = @{ name = "E2ESelf-$Id"; label = 'self2'; tasks = 1; holdms = 8000 } } } } + @{ name = "E2EFol-$Id"; label = 'fol'; tasks = 1 + post = @{ marker = 'fol'; followOn = @{ name = "E2EFolK-$Id"; label = 'folK'; tasks = 1; holdms = 8000 } } } + ) + $W = Wait-OrchPosts $Ns @('chP', 'z', 'cs', 'self', 'fol') 90 + $null = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['self2'].ended -ge 1 -and $S['folK'].ended -ge 1 -and $S['blk'].ended -ge 1) { $true } } 40 + $Rows = Get-OrchRecords $Ns + $PostOf = @{}; foreach ($L in 'chP', 'z', 'cs', 'self', 'fol') { $PostOf[$L] = (Get-OrchPosts $Rows $L) | Select-Object -First 1 } + + # a) a child holds its parent's PostExecution until the child's last task finished + $Kid = Get-OrchTasks $Rows 'chK'; $KidEnd = Get-OrchMax $Kid.end + $Par = (Get-OrchRuns "E2EChP-$Id")[0]; $KidRun = (Get-OrchRuns "E2EChK-$Id")[0] + $Gap = if ($PostOf['chP']) { ConvertTo-OrchMs ($PostOf['chP'].ticks - $KidEnd) } else { 'n/a' } + $Ok = $PostOf['chP'] -and $Kid.Count -eq 3 -and @($Kid.Where({ $_.end -le 0 })).Count -eq 0 -and $PostOf['chP'].ticks -gt $KidEnd -and + $Par.total -eq 3 -and $KidRun.parentRunKey -eq $Par.runKey + Add-Result 'orch-child' 'child-gates-parent' $Ok "+${Gap}ms" "parent post - child last end=${Gap}ms childTasks=$($Kid.Count) parentTotal=$($Par.total) (2 tasks+1 child) childParentKey-match=$($KidRun.parentRunKey -eq $Par.runKey)" + + # b) a child that is never created releases its parent: via Start-CraftOrchestrator (0 tasks -> NoTasks, never + # reaches the bridge) and via the bridge (registered with the parent, then abandoned when the batch is empty) + $ZRun = (Get-OrchRuns "E2EZ-$Id")[0] + $Zs = @($Rows.Where({ $_.kind -eq 'Q' -and $_.run -eq 'zs' })) | Select-Object -First 1 + $NoKids = (Get-OrchRuns "E2EZs-$Id").Count + (Get-OrchRuns "E2EZb-$Id").Count + $Ok = $PostOf['z'] -and $ZRun.phase -eq 'Done' -and $ZRun.status -eq 'Completed' -and $ZRun.total -eq 3 -and $ZRun.done -eq 3 -and + $Zs.result -like '*-NoTasks' -and $NoKids -eq 0 + Add-Result 'orch-child' 'zero-task-child-releases' $Ok '-' "post=$([bool]$PostOf['z']) status=$($ZRun.status) total/done=$($ZRun.total)/$($ZRun.done) (2 tasks + 1 bridge-registered child) startResult=$($Zs.result) childRunsCreated=$NoKids" + + # c) a collision-off child skipped because its name is active releases the parent without waiting for the blocker + $BlkEnd = Get-OrchMax (Get-OrchTasks $Rows 'blk').end + $BlkRuns = Get-OrchRuns "E2EBlk-$Id" + $Ok = $PostOf['cs'] -and $PostOf['cs'].ticks -lt $BlkEnd -and $BlkRuns.Count -eq 1 -and (Get-OrchTasks $Rows 'blk2').Count -eq 0 + $Lead = if ($PostOf['cs']) { ConvertTo-OrchMs ($BlkEnd - $PostOf['cs'].ticks) } else { 'n/a' } + Add-Result 'orch-child' 'skipped-child-releases' $Ok "-${Lead}ms" "parent post ran ${Lead}ms before the blocker ended; runs named E2EBlk=$($BlkRuns.Count) skippedChildTasks=$((Get-OrchTasks $Rows 'blk2').Count)" + + # d) a run re-queueing its own name is not its child + $Self2 = (Get-OrchTasks $Rows 'self2') | Select-Object -First 1 + $SelfRuns = Get-OrchRuns "E2ESelf-$Id" + $Ok = $PostOf['self'] -and $Self2 -and $PostOf['self'].ticks -lt $Self2.end -and $SelfRuns.Count -eq 2 -and @($SelfRuns.Where({ $_.parentRunKey })).Count -eq 0 + $Lead = if ($PostOf['self'] -and $Self2) { ConvertTo-OrchMs ($Self2.end - $PostOf['self'].ticks) } else { 'n/a' } + Add-Result 'orch-child' 'self-requeue-not-child' $Ok "-${Lead}ms" "parent post ${Lead}ms before the re-queued run ended; runs=$($SelfRuns.Count) withParent=$(@($SelfRuns.Where({ $_.parentRunKey })).Count)" + + # e) a run queued from a PostExecution is not a child (the parent is aggregating, not running tasks) + $Fol = (Get-OrchRuns "E2EFol-$Id")[0]; $FolK = (Get-OrchTasks $Rows 'folK') | Select-Object -First 1; $FolKRun = (Get-OrchRuns "E2EFolK-$Id")[0] + $Ok = $Fol.phase -eq 'Done' -and $FolK -and $Fol.completedTicks -lt $FolK.end -and -not $FolKRun.parentRunKey + $Lead = if ($FolK) { ConvertTo-OrchMs ($FolK.end - $Fol.completedTicks) } else { 'n/a' } + Add-Result 'orch-child' 'postexec-run-not-child' $Ok "-${Lead}ms" "parent finished ${Lead}ms before the follow-on ended; parent=$($Fol.status)/$($Fol.postExecStatus) followOnParent='$($FolKRun.parentRunKey)'" + if (-not $W.ok) { Add-Result 'orch-child' 'all-parents-finished' $false "$($W.sec)s" "not every parent PostExecution ran within 90s: $((@('chP','z','cs','self','fol').Where({ -not $PostOf[$_] })) -join ',') missing" } +} + +# -- 4. What a child inherits ------------------------------------------------------------------------------- +Invoke-OrchCheck 'prio' { + $Id = New-OrchId; $Ns = "prio-$Id" + $null = Start-OrchRuns $Ns @( + @{ name = "E2EPri-$Id"; label = 'pri'; Priority = 7; tasks = 2; post = @{ marker = 'pri' } + overrides = @{ '0' = @{ child = @{ name = "E2EPriK1-$Id"; label = 'k1'; tasks = 1 } } + '1' = @{ child = @{ name = "E2EPriK2-$Id"; label = 'k2'; tasks = 1; priority = 2 } } } } + @{ name = "E2ESq-$Id"; label = 'sq'; Sequential = $true; tasks = 2; task = @{ holdms = 200 }; post = @{ marker = 'sq' } + overrides = @{ '0' = @{ child = @{ name = "E2ESqK-$Id"; label = 'sqk'; tasks = 3; holdms = 2000 } } } } + @{ name = "E2ENc-$Id"; label = 'nc'; AllowCollision = $false; tasks = 2; post = @{ marker = 'nc' } + overrides = @{ '0' = @{ child = @{ name = "E2ENcK-$Id"; label = 'nck'; tasks = 1; holdms = 2000 } } + '1' = @{ child = @{ name = "E2ENcK-$Id"; label = 'nck'; tasks = 1; holdms = 2000 } } } } + ) + $W = Wait-OrchPosts $Ns @('pri', 'sq', 'nc') 90 + $Rows = Get-OrchRecords $Ns + $ParPrio = @((Get-OrchTasks $Rows 'pri').prio | Sort-Object -Unique) -join ',' + $K1 = (Get-OrchTasks $Rows 'k1') | Select-Object -First 1; $K1Run = (Get-OrchRuns "E2EPriK1-$Id")[0] + Add-Result 'orch-prio' 'child-inherits-priority' ($K1.prio -eq 7 -and $K1Run.priority -eq 7 -and $ParPrio -eq '7') "$($W.sec)s" "parent task prio=$ParPrio; child (no Priority) task ctx prio=$($K1.prio) run P=$($K1Run.priority)" + $K2 = (Get-OrchTasks $Rows 'k2') | Select-Object -First 1; $K2Run = (Get-OrchRuns "E2EPriK2-$Id")[0] + Add-Result 'orch-prio' 'explicit-child-priority-wins' ($K2.prio -eq 2 -and $K2Run.priority -eq 2) '-' "child Priority=2 under a P7 parent: task ctx prio=$($K2.prio) run P=$($K2Run.priority)" + $SqRun = (Get-OrchRuns "E2ESq-$Id")[0]; $SqkRun = (Get-OrchRuns "E2ESqK-$Id")[0]; $Sqk = Get-OrchTasks $Rows 'sqk' + $Ov = Get-OrchMaxOverlap $Sqk + Add-Result 'orch-prio' 'child-not-sequential' ($SqRun.sequential -eq 1 -and $SqkRun.sequential -eq 0 -and $Ov -ge 2) "overlap=$Ov" "parent sequential=$($SqRun.sequential) child sequential=$($SqkRun.sequential) child tasks max concurrent=$Ov (3 tasks x 2s)" + $NcRuns = Get-OrchRuns "E2ENcK-$Id" + $Ok = $NcRuns.Count -eq 2 -and @($NcRuns.Where({ $_.phase -ne 'Done' })).Count -eq 0 + Add-Result 'orch-prio' 'child-not-collision-off' $Ok '-' "collision-off parent queued two same-name children: runs created=$($NcRuns.Count) done=$(@($NcRuns.Where({ $_.phase -eq 'Done' })).Count)" +} + +# -- 5. Collisions -------------------------------------------------------------------------------------------- +Invoke-OrchCheck 'collide' { + $Id = New-OrchId; $Ns = "col-$Id" + $null = Start-OrchRuns $Ns @( + @{ name = "E2EStk-$Id"; label = 'stkA'; tasks = 3; task = @{ holdms = 1000 }; post = @{ marker = 'a' } } + @{ name = "E2EStk-$Id"; label = 'stkB'; tasks = 3; task = @{ holdms = 1000 }; post = @{ marker = 'b' } } + ) + $W = Wait-OrchPosts $Ns @('stkA', 'stkB') 60 + $Rows = Get-OrchRecords $Ns + $Stk = Get-OrchRuns "E2EStk-$Id" + $Ok = $W.ok -and $Stk.Count -eq 2 -and @($Stk.Where({ $_.status -eq 'Completed' })).Count -eq 2 -and + ((Get-OrchTasks $Rows 'stkA').Count + (Get-OrchTasks $Rows 'stkB').Count) -eq 6 + Add-Result 'orch-collide' 'default-stacks' $Ok "$($W.sec)s" "same-name runs=$($Stk.Count) completed=$(@($Stk.Where({ $_.status -eq 'Completed' })).Count) tasks=$((Get-OrchTasks $Rows 'stkA').Count)+$((Get-OrchTasks $Rows 'stkB').Count) posts=$((Get-OrchPosts $Rows 'stkA').Count)+$((Get-OrchPosts $Rows 'stkB').Count)" + + $Name = "E2ENoC-$Id" + $R1 = (Start-OrchRuns $Ns @(@{ name = $Name; label = 'noc1'; AllowCollision = $false; tasks = 2; task = @{ holdms = 4000 }; post = @{ marker = '1' } }))[0] + $Active = (Invoke-OrchBridge 'active' $Name).active + $R2 = (Start-OrchRuns $Ns @(@{ name = $Name; label = 'noc2'; AllowCollision = $false; tasks = 2; post = @{ marker = '2' } }))[0] + Add-Result 'orch-collide' 'active-while-running' ($Active -eq $true) '-' "IsRunActive=$Active right after queueing" + Add-Result 'orch-collide' 'collision-off-skips' ($R1.result -eq "Craft-$Name" -and $R2.result -eq "Craft-$Name-Skipped" -and $R2.warning) '-' "first=$($R1.result) second=$($R2.result) warning='$($R2.warning)'" + $null = Wait-OrchPosts $Ns @('noc1') 60 + $Idle = Wait-Orch { if ((Invoke-OrchBridge 'active' $Name).active -eq $false) { $true } } 15 + $R3 = (Start-OrchRuns $Ns @(@{ name = $Name; label = 'noc3'; AllowCollision = $false; tasks = 1; post = @{ marker = '3' } }))[0] + $W3 = Wait-OrchPosts $Ns @('noc3') 60 + $Rows = Get-OrchRecords $Ns + $Runs = Get-OrchRuns $Name + $Ok = $Idle.ok -and $R3.result -eq "Craft-$Name" -and $W3.ok -and $Runs.Count -eq 2 -and (Get-OrchTasks $Rows 'noc2').Count -eq 0 + Add-Result 'orch-collide' 'restarts-after-finish' $Ok "$($W3.sec)s" "inactive after finish=$($Idle.ok) third=$($R3.result) runs of name=$($Runs.Count) (skipped one never created; its tasks=$((Get-OrchTasks $Rows 'noc2').Count))" +} + +# -- 6. Parent attribution with overlapping same-name parents ---------------------------------------------- +Invoke-OrchCheck 'attrib' { + $Id = New-OrchId; $Ns = "att-$Id" + $null = Start-OrchRuns $Ns @( + @{ name = "E2EOvl-$Id"; label = 'ovlA'; tasks = 1; post = @{ marker = 'A' }; overrides = @{ '0' = @{ child = @{ name = "E2EOvlKA-$Id"; label = 'kA'; tasks = 1; holdms = 2000 } } } } + @{ name = "E2EOvl-$Id"; label = 'ovlB'; tasks = 1; post = @{ marker = 'B' }; overrides = @{ '0' = @{ child = @{ name = "E2EOvlKB-$Id"; label = 'kB'; tasks = 1; holdms = 9000 } } } } + ) + $W = Wait-OrchPosts $Ns @('ovlA', 'ovlB') 60 + $Rows = Get-OrchRecords $Ns + $PA = (Get-OrchPosts $Rows 'ovlA') | Select-Object -First 1; $PB = (Get-OrchPosts $Rows 'ovlB') | Select-Object -First 1 + $KA = (Get-OrchTasks $Rows 'kA') | Select-Object -First 1; $KB = (Get-OrchTasks $Rows 'kB') | Select-Object -First 1 + $KeyA = ((Get-OrchTasks $Rows 'ovlA') | Select-Object -First 1).runKey; $KeyB = ((Get-OrchTasks $Rows 'ovlB') | Select-Object -First 1).runKey + $KARun = (Get-OrchRuns "E2EOvlKA-$Id")[0]; $KBRun = (Get-OrchRuns "E2EOvlKB-$Id")[0] + $Ok = $PA -and $PB -and $PA.ticks -gt $KA.end -and $PB.ticks -gt $KB.end -and $PA.ticks -lt $KB.end + Add-Result 'orch-attrib' 'waits-for-own-child' $Ok "$($W.sec)s" ("A post-kA end={0}ms, B post-kB end={1}ms, A post before kB end by {2}ms" -f + $(if ($PA) { ConvertTo-OrchMs ($PA.ticks - $KA.end) }), $(if ($PB) { ConvertTo-OrchMs ($PB.ticks - $KB.end) }), $(if ($PA) { ConvertTo-OrchMs ($KB.end - $PA.ticks) })) + Add-Result 'orch-attrib' 'child-linked-to-exact-run' ($KeyA -ne $KeyB -and $KARun.parentRunKey -eq $KeyA -and $KBRun.parentRunKey -eq $KeyB) '-' "kA parent=$($KARun.parentRunKey) (A=$KeyA) kB parent=$($KBRun.parentRunKey) (B=$KeyB)" +} + +# -- 7. CancelRun by name ------------------------------------------------------------------------------------- +Invoke-OrchCheck 'cancel' { + $Id = New-OrchId; $Ns = "can-$Id"; $Name = "E2ECan-$Id" + $null = Start-OrchRuns $Ns @( + @{ name = $Name; label = 'canA'; tasks = 8; task = @{ holdms = 2500 }; post = @{ marker = 'A' } } + @{ name = $Name; label = 'canB'; tasks = 8; task = @{ holdms = 2500 }; post = @{ marker = 'B' } } + ) + $null = Wait-Orch { $S = Get-OrchSummary $Ns; if (($S['canA'].tasks + $S['canB'].tasks) -ge 3) { $true } } 30 250 + $Count = (Invoke-OrchBridge 'cancel' $Name).cancelled + $D = Wait-OrchRunsDone $Name 2 90 + $null = Wait-OrchPosts $Ns @('canA', 'canB') 30 + $Rows = Get-OrchRecords $Ns + $Runs = Get-OrchRuns $Name + $Desc = @($Runs | ForEach-Object { "$($_.status) done=$($_.done) cancelled=$($_.cancelled) failed=$($_.failed)" }) -join '; ' + $Ok = $D.ok -and $Count -gt 0 -and $Runs.Count -eq 2 -and @($Runs.Where({ $_.status -eq 'CompletedWithErrors' -and $_.cancelled -gt 0 })).Count -eq 2 + Add-Result 'orch-cancel' 'cancels-every-outing' $Ok "$($D.sec)s" "CancelRun returned $Count; $Desc" + # The run keys are in start order; the outings were queued A then B. + $PostOk = $true; $Parts = [System.Collections.Generic.List[string]]::new() + $I = 0 + foreach ($L in 'canA', 'canB') { + $P = Get-OrchPosts $Rows $L; $T = Get-OrchTasks $Rows $L; $Run = $Runs[$I]; $I = $I + 1 + $Ran = $Run.done - $Run.cancelled - $Run.failed + $One = $P.Count -eq 1 -and $P[0].lines -eq $Ran -and $T.Count -eq $Ran -and $Run.postExecStatus -eq 'Completed' + if (-not $One) { $PostOk = $false } + $Parts.Add("${L}: posts=$($P.Count) lines=$($P[0].lines) started=$($T.Count) ran=$Ran postExec=$($Run.postExecStatus)") + } + Add-Result 'orch-cancel' 'postexec-still-runs' $PostOk '-' ($Parts -join '; ') +} + +# -- 8. Sequential runs --------------------------------------------------------------------------------------- +Invoke-OrchCheck 'seq' { + $Enq = Invoke-OrchGet '/API/PerfSeqWorkerEnqueue?runs=4&steps=5&holdms=500' + $W = Wait-Orch { $R = Invoke-OrchGet '/API/PerfSeqWorkerResult'; if ($R.count -ge 20) { $R } } 90 + Start-Sleep -Milliseconds 500 + $Rows = @((Invoke-OrchGet '/API/PerfSeqWorkerResult').rows) + $OrderOk = $true; $PinOk = $true; $Workers = [System.Collections.Generic.List[string]]::new(); $Parts = [System.Collections.Generic.List[string]]::new() + foreach ($N in @($Enq.names)) { + $Steps = @($Rows.Where({ $_.run -eq $N }) | Sort-Object ticks) + $Order = @($Steps.idx) -join '' + $Ws = @($Steps.worker | Sort-Object -Unique) + if ($Order -ne '01234') { $OrderOk = $false } + if ($Ws.Count -ne 1) { $PinOk = $false } + foreach ($X in $Ws) { $Workers.Add($X) } + $Parts.Add("$($N.Substring(0, 4)):$Order@$($Ws -join '/')") + } + $Distinct = @($Workers | Sort-Object -Unique).Count + Add-Result 'orch-seq' 'payload-order' ($W.ok -and $OrderOk -and $Rows.Count -eq 20) "$($W.sec)s" "steps=$($Rows.Count) $($Parts -join ' ')" + Add-Result 'orch-seq' 'one-pinned-worker-per-run' ($PinOk -and $Distinct -eq [math]::Min(4, $OrchPool)) '-' "workers per run all 1=$PinOk; distinct workers across 4 concurrent runs=$Distinct" + + $Id = New-OrchId; $Ns = "seq-$Id"; $Name = "E2ESeqF-$Id" + $null = Start-OrchRuns $Ns @(@{ name = $Name; label = 'sf'; Sequential = $true; tasks = 5; task = @{ holdms = 300 }; overrides = @{ '2' = @{ fail = $true } }; post = @{ marker = 'sf' } }) + $W = Wait-OrchPosts $Ns @('sf') 60 + $Rows = Get-OrchRecords $Ns + $T = @((Get-OrchTasks $Rows 'sf') | Sort-Object start); $P = (Get-OrchPosts $Rows 'sf') | Select-Object -First 1; $H = (Get-OrchRuns $Name)[0] + $Ok = $T.Count -eq 5 -and (@($T.idx) -join '') -eq '01234' -and @($T.worker | Sort-Object -Unique).Count -eq 1 -and (Get-OrchMaxOverlap $T) -eq 1 -and + @($T.Where({ $_.end -le 0 })).Count -eq 0 -and $H.status -eq 'CompletedWithErrors' -and $H.failed -eq 1 -and $H.done -eq 5 + Add-Result 'orch-seq' 'failed-step-continues' $Ok "$($W.sec)s" "order=$(@($T.idx) -join '') workers=$(@($T.worker | Sort-Object -Unique) -join '/') overlap=$(Get-OrchMaxOverlap $T) status=$($H.status) done=$($H.done) failed=$($H.failed)" + # Every step that succeeded must reach the PostExecution, including the one right after the failed step. + $Got = @($P.idxs -split ',' | Sort-Object) -join ',' + Add-Result 'orch-seq' 'postexec-gets-every-success' ($P.lines -eq 4 -and $Got -eq 'sf/0,sf/1,sf/3,sf/4') '-' "expected sf/0,sf/1,sf/3,sf/4 (step 2 throws); PostExecution received lines=$($P.lines) idxs=$($P.idxs)" + + # Every step runs inside the one job that drives the run, but worker stats and the queue page must show + # each step on its own: the worker under the step it is on, the queue with one task per step. + $Id = New-OrchId; $Ns = "seqv-$Id"; $Name = "E2ESeqV-$Id" + $null = Start-OrchRuns $Ns @(@{ name = $Name; label = 'sv'; Sequential = $true; tasks = 4; task = @{ holdms = 1500 } }) + $Labels = [System.Collections.Generic.HashSet[string]]::new() + $null = Wait-Orch { + foreach ($B in @((Invoke-OrchBridge 'workers').busy)) { if ($B.fn -like "$Name-*") { [void]$Labels.Add($B.fn) } } + $S = Get-OrchSummary $Ns; if ($S['sv'].ended -ge 4) { $true } + } 60 250 + $Q = Wait-Orch { $E = @((Invoke-OrchBridge 'queue' $Name).entries)[0]; if ($E.Status -eq 'Completed') { $E } } 30 + $E = $Q.value + Add-Result 'orch-seq' 'each-step-visible' ($Labels.Count -ge 3 -and $E.TotalTasks -eq 4 -and $E.CompletedTasks -eq 4 -and @($E.Tasks).Count -eq 4) '-' "worker labels seen=$($Labels.Count) (of 4 steps); queue total=$($E.TotalTasks) completed=$($E.CompletedTasks) tasks listed=$(@($E.Tasks).Count)" +} + +# -- 9. Priority and start-order ------------------------------------------------------------------------------ +# A hold run of pool+3 tasks fills every worker and leaves 3 claims in the local buffer (above the pump's low +# water mark of 2), so nothing else is claimed until the hold wave ends. The runs under test are queued in one +# call while that is the case; whatever is claimed first after it is down to the store's ordering. +function Start-OrchSaturation([string]$Ns, [string]$Name) { + $null = Start-OrchRuns $Ns @(@{ name = $Name; label = 'hold'; tasks = ($OrchPool + 3); task = @{ holdms = 6000 } }) + Wait-Orch { + $S = Get-OrchSummary $Ns + $A = Invoke-OrchGet '/API/PerfAllocation' + if ($S['hold'].tasks -ge $OrchPool -and [int]$A.jm.queued -ge 3) { $true } + } 30 250 +} + +Invoke-OrchCheck 'order' { + $Id = New-OrchId; $Ns = "ord-$Id" + $Sat = Start-OrchSaturation $Ns "E2EHold-$Id" + $null = Start-OrchRuns $Ns @( + @{ name = "E2EP9-$Id"; label = 'p9'; Priority = 9; tasks = 4; task = @{ holdms = 300 } } + @{ name = "E2EP1-$Id"; label = 'p1'; Priority = 1; tasks = 4; task = @{ holdms = 300 } } + ) + $W = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['p9'].ended -ge 4 -and $S['p1'].ended -ge 4) { $true } } 90 + $Rows = Get-OrchRecords $Ns + $P1 = Get-OrchTasks $Rows 'p1'; $P9 = Get-OrchTasks $Rows 'p9' + $Ok = $Sat.ok -and $W.ok -and (Get-OrchMax $P1.start) -lt (Get-OrchMin $P9.start) + Add-Result 'orch-order' 'lower-priority-first' $Ok ("{0}ms" -f (ConvertTo-OrchMs ((Get-OrchMin $P9.start) - (Get-OrchMax $P1.start)))) "saturated=$($Sat.ok); P9 queued first, then P1: last P1 start precedes first P9 start by $(ConvertTo-OrchMs ((Get-OrchMin $P9.start) - (Get-OrchMax $P1.start)))ms" + + $Id = New-OrchId; $Ns = "ordb-$Id" + $Sat = Start-OrchSaturation $Ns "E2EHold-$Id" + $null = Start-OrchRuns $Ns @( + @{ name = "E2EZulu-$Id"; label = 'zulu'; Priority = 6; tasks = 4; task = @{ holdms = 300 } } + @{ name = "E2EAlpha-$Id"; label = 'alpha'; Priority = 6; tasks = 4; task = @{ holdms = 300 } } + ) + $W = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['zulu'].ended -ge 4 -and $S['alpha'].ended -ge 4) { $true } } 90 + $S = Get-OrchSummary $Ns + $Lead = ConvertTo-OrchMs ($S['alpha'].minStart - $S['zulu'].minStart) + Add-Result 'orch-order' 'older-run-first-in-band' ($Sat.ok -and $W.ok -and $S['zulu'].minStart -lt $S['alpha'].minStart) "${Lead}ms" "saturated=$($Sat.ok); Zulu queued before Alpha (same band): Zulu first start leads Alpha by ${Lead}ms" +} + +# -- 11. Status consistency --------------------------------------------------------------------------------- +Invoke-OrchCheck 'status' { + $Id = New-OrchId; $Ns = "st-$Id"; $Name = "E2EStat-$Id" + $null = Start-OrchRuns $Ns @(@{ name = $Name; label = 'st'; tasks = 200; task = @{ holdms = 300 } }) + $Samples = 0; $Bad = [System.Collections.Generic.List[string]]::new(); $SumBad = [System.Collections.Generic.List[string]]::new() + $Sw = [Diagnostics.Stopwatch]::StartNew() + while ($Sw.Elapsed.TotalSeconds -lt 120) { + $H = (Get-OrchRuns $Name)[0] + if ($H.phase -eq 'Done') { break } + $Sm = (Invoke-OrchBridge 'summaries' $Name).runs | Select-Object -First 1 + $G = Invoke-OrchBridge 'summary' + if ($Sm -and ($Sm.queued + $Sm.running) -gt 0) { + $Samples = $Samples + 1 + if ($Sm.total -ne 200 -or ($Sm.completed + $Sm.queued + $Sm.running) -gt $Sm.total) { + $Bad.Add("t=$([math]::Round($Sw.Elapsed.TotalSeconds,1))s total=$($Sm.total) c=$($Sm.completed) q=$($Sm.queued) r=$($Sm.running)") + } + # The global summary also counts unrelated local work (the e2e timer, leftovers of other checks), so bound + # the durable part: what is waiting in storage can never exceed this run's 200 tasks. + if ($G.jobsQueuedDurable -gt 200 -or ($G.jobsQueued - $G.jobsQueuedLocal) -ne $G.jobsQueuedDurable) { $SumBad.Add("q=$($G.jobsQueued) local=$($G.jobsQueuedLocal) durable=$($G.jobsQueuedDurable) a=$($G.jobsActive)") } + } + Start-Sleep -Milliseconds 400 + } + $Done = (Get-OrchRuns $Name)[0] + Add-Result 'orch-status' 'summaries-consistent' ($Samples -ge 5 -and $Bad.Count -eq 0) "$Samples samples" "in-flight samples=$Samples violations=$($Bad.Count) $(@($Bad | Select-Object -First 3) -join ' | ')" + Add-Result 'orch-status' 'summary-bounded' ($Samples -ge 5 -and $SumBad.Count -eq 0) '-' "GetSummary durable queue > batch in $($SumBad.Count) samples $(@($SumBad | Select-Object -First 3) -join ' | ')" + $Left = Wait-Orch { + $Sm = (Invoke-OrchBridge 'summaries' $Name).runs | Select-Object -First 1 + if ((-not $Sm -or ($Sm.queued + $Sm.running) -eq 0) -and (Invoke-OrchBridge 'active' $Name).active -eq $false) { $true } + } 20 + $Sm = (Invoke-OrchBridge 'summaries' $Name).runs | Select-Object -First 1 + $S = Get-OrchSummary $Ns + $Ok = $Done.phase -eq 'Done' -and $Done.status -eq 'Completed' -and $Done.done -eq 200 -and $S['st'].tasks -eq 200 -and $Left.ok + Add-Result 'orch-status' 'leaves-active-set' $Ok "$([math]::Round($Sw.Elapsed.TotalSeconds,1))s" "run=$($Done.status) done=$($Done.done) recorded=$($S['st'].tasks); after: summary q=$($Sm.queued) r=$($Sm.running) c=$($Sm.completed) total=$($Sm.total) active=$(-not $Left.ok)" +} + +# -- 13. MaxConcurrency (gated) ------------------------------------------------------------------------------- +Invoke-OrchCheck 'maxconc' { + $Id = New-OrchId; $Ns = "mc-$Id" + $null = Start-OrchRuns $Ns @(@{ name = "E2EMc2-$Id"; label = 'mc2'; MaxConcurrency = 2; tasks = 12; task = @{ holdms = 1500 }; post = @{ marker = 'mc2' } }) + $W = Wait-OrchPosts $Ns @('mc2') 120 + $Ov = Get-OrchMaxOverlap (Get-OrchTasks (Get-OrchRecords $Ns) 'mc2') + Add-Result 'orch-maxconc' 'limit-2' ($W.ok -and $Ov -eq 2) "$($W.sec)s" "MaxConcurrency=2, 12 tasks x 1.5s: max concurrent=$Ov" + + $null = Start-OrchRuns $Ns @(@{ name = "E2EMc0-$Id"; label = 'mc0'; MaxConcurrency = 0; tasks = 12; task = @{ holdms = 1500 }; post = @{ marker = 'mc0' } }) + $W = Wait-OrchPosts $Ns @('mc0') 120 + $Ov = Get-OrchMaxOverlap (Get-OrchTasks (Get-OrchRecords $Ns) 'mc0') + Add-Result 'orch-maxconc' 'unlimited-reaches-pool' ($W.ok -and $Ov -eq $OrchPool) "$($W.sec)s" "MaxConcurrency=0: max concurrent=$Ov pool=$OrchPool" + + $null = Start-OrchRuns $Ns @(@{ name = "E2EMc1-$Id"; label = 'mc1'; MaxConcurrency = 1; tasks = 6; task = @{ holdms = 1000 }; post = @{ marker = 'mc1' } }) + $null = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['mc1'].tasks -ge 2) { $true } } 30 250 + $null = Start-OrchRuns $Ns @(@{ name = "E2EMcP1-$Id"; label = 'mcp1'; Priority = 1; tasks = 1 }) + $W = Wait-OrchPosts $Ns @('mc1') 60 + $Rows = Get-OrchRecords $Ns + $Mc1 = Get-OrchTasks $Rows 'mc1'; $P1 = (Get-OrchTasks $Rows 'mcp1') | Select-Object -First 1 + $Ok = $W.ok -and (Get-OrchMaxOverlap $Mc1) -eq 1 -and $P1 -and $P1.start -lt (Get-OrchMax $Mc1.start) + Add-Result 'orch-maxconc' 'limit-1-lets-p1-in' $Ok '-' "N=1 overlap=$(Get-OrchMaxOverlap $Mc1); P1 task started $(if ($P1) { ConvertTo-OrchMs ((Get-OrchMax $Mc1.start) - $P1.start) })ms before the N=1 run's last step" + + $R = (Start-OrchRuns $Ns @(@{ name = "E2EMcSq-$Id"; label = 'mcsq'; Sequential = $true; MaxConcurrency = 3; tasks = 3; task = @{ holdms = 300 }; post = @{ marker = 'mcsq' } }))[0] + $W = Wait-OrchPosts $Ns @('mcsq') 60 + $Sq = Get-OrchTasks (Get-OrchRecords $Ns) 'mcsq' + $Ok = $W.ok -and $R.warning -match 'MaxConcurrency' -and (Get-OrchMaxOverlap $Sq) -eq 1 -and $Sq.Count -eq 3 + Add-Result 'orch-maxconc' 'ignored-when-sequential' $Ok '-' "result=$($R.result) warning='$($R.warning)' overlap=$(Get-OrchMaxOverlap $Sq) steps=$($Sq.Count)" +} + +# -- 14. StopOnFailure (gated) -------------------------------------------------------------------------------- +Invoke-OrchCheck 'stopfail' { + $Id = New-OrchId; $Ns = "sof-$Id"; $Name = "E2ESof-$Id" + $null = Start-OrchRuns $Ns @(@{ name = $Name; label = 'sof'; Sequential = $true; StopOnFailure = $true; tasks = 5; task = @{ holdms = 200 }; overrides = @{ '2' = @{ fail = $true } }; post = @{ marker = 'sof' } }) + $W = Wait-OrchPosts $Ns @('sof') 60 + Start-Sleep -Seconds 2 + $Rows = Get-OrchRecords $Ns + $T = @((Get-OrchTasks $Rows 'sof') | Sort-Object start); $P = (Get-OrchPosts $Rows 'sof') | Select-Object -First 1; $H = (Get-OrchRuns $Name -Detail)[0] + Add-Result 'orch-stopfail' 'stops-at-failure' ((@($T.idx) -join '') -eq '012') "$($W.sec)s" "steps that ran=$(@($T.idx) -join ',')" + $Cancelled = @($H.tasks.Where({ $_.status -eq 'Cancelled' }).seq) -join ',' + Add-Result 'orch-stopfail' 'rest-cancelled' ($Cancelled -eq '3,4' -and $H.status -eq 'CompletedWithErrors') '-' "cancelled seqs=$Cancelled status=$($H.status) failed=$($H.failed) cancelled=$($H.cancelled)" + Add-Result 'orch-stopfail' 'postexec-runs' ($P -and $P.lines -eq 2 -and $H.postExecStatus -eq 'Completed') '-' "posts=$(@(Get-OrchPosts $Rows 'sof').Count) lines=$($P.lines) idxs=$($P.idxs) postExec=$($H.postExecStatus)" + $null = Start-OrchRuns $Ns @(@{ name = "E2ESofD-$Id"; label = 'sofd'; Sequential = $true; tasks = 5; task = @{ holdms = 200 }; overrides = @{ '2' = @{ fail = $true } }; post = @{ marker = 'sofd' } }) + $W = Wait-OrchPosts $Ns @('sofd') 60 + $T = Get-OrchTasks (Get-OrchRecords $Ns) 'sofd' + Add-Result 'orch-stopfail' 'default-runs-all' ($T.Count -eq 5) "$($W.sec)s" "without StopOnFailure steps that ran=$($T.Count)" +} + +# -- 12. Performance gates ------------------------------------------------------------------------------------- +Invoke-OrchCheck 'perf' { + $Id = New-OrchId; $Ns = "pf-$Id" + $Sw = [Diagnostics.Stopwatch]::StartNew() + $null = Start-OrchRuns $Ns @(@{ name = "E2EFan-$Id"; label = 'fan'; tasks = 1000; post = @{ marker = 'fan' } }) + $W = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['fan'].posts -ge 1) { $S['fan'] } } ($OrchGates.Fanout1000Sec + 60) 500 + $Sec = [math]::Round($Sw.Elapsed.TotalSeconds, 1) + $Rows = Get-OrchRecords $Ns + $T = Get-OrchTasks $Rows 'fan'; $P = (Get-OrchPosts $Rows 'fan') | Select-Object -First 1 + $Once = $T.Count -eq 1000 -and @($T | Group-Object idx).Count -eq 1000 -and $P.lines -eq 1000 + Add-Result 'orch-perf' 'fanout-1000' ($W.ok -and $Once -and $Sec -le $OrchGates.Fanout1000Sec) "${Sec}s" ("{0:N0} tasks/s; recorded={1} distinct={2} postLines={3}; gate {4}s" -f (1000 / [math]::Max(0.1, $Sec)), $T.Count, @($T | Group-Object idx).Count, $P.lines, $OrchGates.Fanout1000Sec) + $null = Invoke-OrchGet "/API/PerfE2EState?ns=$Ns&wipe=1" + + $Id = New-OrchId; $Ns = "pm-$Id" + $Runs = @(for ($I = 0; $I -lt 300; $I++) { @{ name = "E2EMany-$Id-$I"; label = "m$I"; tasks = 1 } }) + $Sw = [Diagnostics.Stopwatch]::StartNew() + $null = Start-OrchRuns $Ns $Runs -TimeoutSec 300 + $W = Wait-Orch { $R = Get-OrchRunsByPrefix "E2EMany-$Id-"; if ($R.Count -ge 300 -and @($R.Where({ $_.phase -ne 'Done' })).Count -eq 0) { , $R } } ($OrchGates.ManyRuns300Sec + 60) 1000 + $Sec = [math]::Round($Sw.Elapsed.TotalSeconds, 1) + $R = Get-OrchRunsByPrefix "E2EMany-$Id-" + $S = Get-OrchSummary $Ns + $Ran = @($S.Values.Where({ $_.tasks -eq 1 })).Count + Add-Result 'orch-perf' 'runs-300' ($W.ok -and $Ran -eq 300 -and $Sec -le $OrchGates.ManyRuns300Sec) "${Sec}s" "runs=$($R.Count) done=$(@($R.Where({ $_.phase -eq 'Done' })).Count) tasks ran once=$Ran; gate $($OrchGates.ManyRuns300Sec)s" + $null = Invoke-OrchGet "/API/PerfE2EState?ns=$Ns&wipe=1" + + # Storage cost between sequential steps: each step's finish and the next step's claim share one transaction. + $Id = New-OrchId; $Ns = "ps-$Id" + $null = Start-OrchRuns $Ns @(@{ name = "E2ESeqPerf-$Id"; label = 'sq'; Sequential = $true; tasks = 100 }) + $W = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['sq'].ended -ge 100) { $S['sq'] } } 120 250 + $StepMs = if ($W.ok) { [math]::Round((ConvertTo-OrchMs ($W.value.maxEnd - $W.value.minStart)) / 99, 1) } else { -1 } + Add-Result 'orch-perf' 'sequential-step' ($W.ok -and $W.value.tasks -eq 100 -and $StepMs -le $OrchGates.SeqStepMs) "${StepMs}ms" "100 steps, first start -> last end per step ${StepMs}ms; gate $($OrchGates.SeqStepMs)ms" + $null = Invoke-OrchGet "/API/PerfE2EState?ns=$Ns&wipe=1" + + # An idle pump backs off its poll to JobQueueIdlePollIntervalMs (10s default), so a run queued into an idle + # engine waits for the next poll. Pin that bound, then keep the pump warm (a held task in flight keeps it on + # the 1s poll) so the size comparison below measures claiming, not the idle backoff. + $Id = New-OrchId; $Ns = "pt-$Id" + Start-Sleep -Seconds 20 + $EIdle = (Start-OrchRuns $Ns @(@{ name = "E2ETfIdle-$Id"; label = 'tfidle'; tasks = 1 }))[0] + $WIdle = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['tfidle'].tasks -ge 1) { $S['tfidle'] } } 60 100 + $TtfsIdle = if ($WIdle.ok) { ConvertTo-OrchMs ($WIdle.value.minStart - $EIdle.enqueueTicks) } else { -1 } + Add-Result 'orch-perf' 'idle-claim-latency' ($WIdle.ok -and $TtfsIdle -le $OrchGates.IdleClaimMs) "${TtfsIdle}ms" "1-task run queued into an idle engine -> first start ${TtfsIdle}ms; gate $($OrchGates.IdleClaimMs)ms (idle poll cap 10s)" + $null = Start-OrchRuns $Ns @(@{ name = "E2ETfKeep-$Id"; label = 'keep'; tasks = 1; task = @{ holdms = 60000 } }) + $null = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['keep'].tasks -ge 1) { $true } } 30 250 + Start-Sleep -Seconds 2 + $E10 = (Start-OrchRuns $Ns @(@{ name = "E2ETf10-$Id"; label = 'tf10'; tasks = 10 }))[0] + $W10 = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['tf10'].tasks -ge 1) { $S['tf10'] } } 60 100 + $Ttfs10 = if ($W10.ok) { ConvertTo-OrchMs ($W10.value.minStart - $E10.enqueueTicks) } else { -1 } + $null = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['tf10'].ended -ge 10) { $true } } 60 + $E5k = (Start-OrchRuns $Ns @(@{ name = "E2ETf5k-$Id"; label = 'tf5k'; tasks = 5000 }) -TimeoutSec 300)[0] + $W5k = Wait-Orch { $S = Get-OrchSummary $Ns; if ($S['tf5k'].tasks -ge 1) { $S['tf5k'] } } 180 100 + $Ttfs5k = if ($W5k.ok) { ConvertTo-OrchMs ($W5k.value.minStart - $E5k.enqueueTicks) } else { -1 } + Add-Result 'orch-perf' 'ttfs-5000' ($W5k.ok -and $Ttfs5k -le $OrchGates.Ttfs5000Ms) "${Ttfs5k}ms" "enqueue->first task start: 5,000-task run ${Ttfs5k}ms vs 10-task run ${Ttfs10}ms (x$([math]::Round($Ttfs5k / [math]::Max(1, $Ttfs10), 1))); gate $($OrchGates.Ttfs5000Ms)ms" + $CancelSw = [Diagnostics.Stopwatch]::StartNew() + $Cancelled = (Invoke-OrchBridge 'cancel' "E2ETf5k-$Id").cancelled + $D = Wait-OrchRunsDone "E2ETf5k-$Id" 1 240 500 + $CancelSec = [math]::Round($CancelSw.Elapsed.TotalSeconds, 1) + $H = (Get-OrchRuns "E2ETf5k-$Id")[0] + # CancelRun reports what the run records, so it must match the header (it once returned 0 while 4,995 were cancelled). + Add-Result 'orch-perf' 'cancel-5000' ($D.ok -and $H.status -eq 'CompletedWithErrors' -and $H.done -eq 5000 -and $H.cancelled -gt 0 -and [int]$Cancelled -eq [int]$H.cancelled -and $CancelSec -le $OrchGates.Cancel5000Sec) "${CancelSec}s" "CancelRun returned $Cancelled; run=$($H.status) done=$($H.done)/$($H.total) cancelled=$($H.cancelled); gate $($OrchGates.Cancel5000Sec)s" + $null = Invoke-OrchGet "/API/PerfE2EState?ns=$Ns&wipe=1" + + Start-Sleep -Seconds 5 + $M = Invoke-OrchBridge 'summary' + Add-Result 'orch-perf' 'memory-after-perf' ($M.rssMB -le $OrchGates.RssMB) "$($M.rssMB)MB" "rss=$($M.rssMB)MB workingSet=$($M.workingSetMB)MB heap=$($M.heapMB)MB committed=$($M.committedMB)MB; gate rss $($OrchGates.RssMB)MB" +} + +# -- 10. Restart recovery + legacy table drop (restarts the SUT; keep last) ------------------------------------- +Invoke-OrchCheck 'restart' { + $Id = New-OrchId; $Ns = "rst-$Id"; $Name = "E2ERst-$Id" + $N = $OrchPool + 2 # pool running + 2 buffered claims for the graceful stop to hand back + # Earlier checks may still hold workers (the perf keeper task): start from an idle engine so the whole pool + # goes to this run and the running/buffered split is the intended one. + $Quiet = Wait-Orch { $G = Invoke-OrchBridge 'summary'; if ($G.jobsActive -eq 0 -and $G.jobsQueued -eq 0) { $true } } 120 1000 + $Seed = Invoke-OrchGet '/API/PerfE2ELegacy?op=seed' + $null = Start-OrchRuns $Ns @(@{ name = $Name; label = 'rst'; tasks = $N; task = @{ holdms = 20000 }; post = @{ marker = 'rst' } }) -Sink 'table' + $Up = Wait-Orch { $S = Get-OrchSummary $Ns 'table'; if ($S['rst'].tasks -ge $OrchPool) { $true } } 60 250 + Start-Sleep -Seconds 2 # let the pump buffer the remaining claims + $Pre = Get-OrchTasks (Get-OrchRecords $Ns 'table') 'rst' + $Running = @($Pre.Where({ $_.end -le 0 }).idx | Sort-Object -Unique) + $Unstarted = @((0..($N - 1)).Where({ $_ -notin $Pre.idx })) + Info "restart: docker restart $OrchContainer with $($Running.Count) tasks running, $($Unstarted.Count) not started ..." + $Sw = [Diagnostics.Stopwatch]::StartNew() + docker restart $OrchContainer 2>&1 | Out-Null + $Ready = Wait-Orch { $H = Invoke-OrchGet '/healthz' 5; if ($H.status -eq 'ready') { $true } } 180 1000 + $Legacy = Invoke-OrchGet '/API/PerfE2ELegacy?op=list' + Add-Result 'orch-restart' 'legacy-tables-dropped' ($Ready.ok -and @($Seed.present).Count -eq 5 -and @($Legacy.present).Count -eq 0) "$($Ready.sec)s" "seeded+present before=$(@($Seed.present).Count) present after restart=$(@($Legacy.present).Count) $(@($Legacy.present) -join ',')" + $D = Wait-OrchRunsDone $Name 1 480 3000 + $RestartSec = [math]::Round($Sw.Elapsed.TotalSeconds, 1) + $H = (Get-OrchRuns $Name -Detail)[0] + $Rows = Get-OrchRecords $Ns 'table' + $T = Get-OrchTasks $Rows 'rst'; $P = Get-OrchPosts $Rows 'rst' + $AttemptOf = @{}; foreach ($X in $H.tasks) { $AttemptOf[[int]$X.seq] = [int]$X.attempt } + $Attempts = @($H.tasks | ForEach-Object { "$($_.seq):$($_.attempt)" }) -join ',' + $Starts = @($T | Group-Object idx | Sort-Object { [int]$_.Name } | ForEach-Object { "$($_.Name)x$($_.Count)" }) -join ',' + $Ok = $Quiet.ok -and $Up.ok -and $D.ok -and $H.status -eq 'Completed' -and $H.done -eq $N -and @($H.tasks.Where({ $_.status -eq 'Completed' })).Count -eq $N + Add-Result 'orch-restart' 'run-completes' $Ok "${RestartSec}s" "restart->done ${RestartSec}s; run=$($H.status)/$($H.phase) done=$($H.done)/$($H.total) open=P$($H.open.P)/R$($H.open.R) (engine idle first=$($Quiet.ok))" + $Finished = @($T.Where({ $_.end -gt 0 }) | Group-Object idx) + $OnceOk = $Finished.Count -eq $N -and @($Finished.Where({ $_.Count -ne 1 })).Count -eq 0 + Add-Result 'orch-restart' 'each-task-completes-once' $OnceOk '-' "finished records per idx=$(@($Finished | ForEach-Object { "$($_.Name)x$($_.Count)" }) -join ','); starts per idx=$Starts; D-row attempts=$Attempts" + $ReBad = @($Running.Where({ $I = $_; $AttemptOf[[int]$I] -lt 2 -or @($T.Where({ $_.idx -eq $I })).Count -lt 2 })) + Add-Result 'orch-restart' 'interrupted-rerun' ($Running.Count -ge 1 -and $ReBad.Count -eq 0) '-' "running at restart=$($Running -join ','); their attempts=$(@($Running | ForEach-Object { "${_}:$($AttemptOf[[int]$_])" }) -join ',')" + $UnBad = @($Unstarted.Where({ $AttemptOf[[int]$_] -ne 1 })) + Add-Result 'orch-restart' 'unstarted-claims-released' ($Unstarted.Count -ge 1 -and $UnBad.Count -eq 0) '-' "not started at restart=$($Unstarted -join ','); attempts=$(@($Unstarted | ForEach-Object { "${_}:$($AttemptOf[[int]$_])" }) -join ',') (1 = handed back on graceful stop, 2 = lease lapsed)" + Add-Result 'orch-restart' 'postexec-once' ($P.Count -eq 1 -and $P[0].lines -eq $N -and $H.postExecStatus -eq 'Completed') '-' "posts=$($P.Count) lines=$($P[0].lines) postExec=$($H.postExecStatus)" +} + +Info ("orchestration gates: fanout-1000 <= {0}s, runs-300 <= {1}s, ttfs-5000 <= {2}ms, rss <= {3}MB, idle claim <= {4}ms, cancel-5000 <= {5}s, sequential step <= {6}ms" -f $OrchGates.Fanout1000Sec, $OrchGates.ManyRuns300Sec, $OrchGates.Ttfs5000Ms, $OrchGates.RssMB, $OrchGates.IdleClaimMs, $OrchGates.Cancel5000Sec, $OrchGates.SeqStepMs) diff --git a/perf-harness/scripts/run-e2e.ps1 b/perf-harness/scripts/run-e2e.ps1 index ce8b581..b3c4420 100644 --- a/perf-harness/scripts/run-e2e.ps1 +++ b/perf-harness/scripts/run-e2e.ps1 @@ -19,7 +19,10 @@ param( [int]$Port = 5399, [int]$ReadyTimeoutSec = 180, [switch]$Build, - [switch]$KeepUp + [switch]$KeepUp, + # Run only the orchestration section (skips the platform checks), optionally only some of its groups. + [switch]$OrchOnly, + [string[]]$OrchChecks ) $ErrorActionPreference = 'Stop' @@ -33,9 +36,14 @@ $results = New-Object System.Collections.ArrayList function Info($m) { Write-Host "[e2e] $m" -ForegroundColor Cyan } function Add-Result($area, $name, $pass, $perf, $detail) { - [void]$results.Add([pscustomobject]@{ area = $area; name = $name; pass = [bool]$pass; perf = $perf; detail = $detail }) + [void]$results.Add([pscustomobject]@{ area = $area; name = $name; pass = [bool]$pass; skip = $false; perf = $perf; detail = $detail }) $tag = if ($pass) { 'PASS' } else { 'FAIL' } - Write-Host (" [{0}] {1,-12} {2,-18} {3,-8} {4}" -f $tag, $area, $name, $perf, $detail) -ForegroundColor $(if ($pass) { 'Green' } else { 'Red' }) + Write-Host (" [{0}] {1,-14} {2,-28} {3,-10} {4}" -f $tag, $area, $name, $perf, $detail) -ForegroundColor $(if ($pass) { 'Green' } else { 'Red' }) +} +# A check this SUT cannot run (the feature is not in the image). Neither a pass nor a failure. +function Add-Skip($area, $name, $detail) { + [void]$results.Add([pscustomobject]@{ area = $area; name = $name; pass = $true; skip = $true; perf = '-'; detail = $detail }) + Write-Host (" [SKIP] {0,-14} {1,-28} {2,-10} {3}" -f $area, $name, '-', $detail) -ForegroundColor Yellow } function Api($path) { try { return Invoke-RestMethod "$base$path" -TimeoutSec 20 } catch { return $null } } # Same, but against an absolute URL — the throwaway containers further down run on their own ports. @@ -82,6 +90,8 @@ try { Add-Result 'health' 'readiness' $ready '-' "status=$($h.status)" Add-Result 'storage' 'azurite-ready' ($h.ready.storage -eq $true) '-' "storageReady=$($h.ready.storage)" + $suiteSw = [Diagnostics.Stopwatch]::StartNew() + if (-not $OrchOnly) { # ── API dispatch ──────────────────────────────────────────────────────────── $sw = [Diagnostics.Stopwatch]::StartNew(); $ping = Api '/API/PerfPing'; $sw.Stop() Add-Result 'api' 'dispatch-ping' ($ping.ok -eq $true) ("{0}ms" -f $sw.ElapsedMilliseconds) "endpoint=$($ping.endpoint)" @@ -274,11 +284,18 @@ try { $id = Fetch "$base/bundle.js" @('-H', 'Accept-Encoding: identity') Add-Result 'frontend' 'identity-content' ($id.Code -eq 200 -and $id.Body -match 'E2E_BUNDLE_MARKER') ("{0}ms {1}B" -f $id.TimeMs, $id.Size) "identity bundle content correct" + } + + # -- Orchestration engine (run-e2e-orchestration.ps1) ------------------------------ + if ($ready) { . (Join-Path $here 'run-e2e-orchestration.ps1') } + else { Add-Result 'orchestration' 'not-run' $false '-' 'SUT never became ready' } + # ── Summary ───────────────────────────────────────────────────────────────── $fail = @($results | Where-Object { -not $_.pass }) - $pass = @($results | Where-Object { $_.pass }) + $skip = @($results | Where-Object { $_.skip }) + $pass = @($results | Where-Object { $_.pass -and -not $_.skip }) Write-Host "" - Write-Host ("===== E2E: {0} passed, {1} failed =====" -f $pass.Count, $fail.Count) -ForegroundColor $(if ($fail.Count) { 'Red' } else { 'Green' }) + Write-Host ("===== E2E: {0} passed, {1} failed, {2} skipped ({3:N0}s) =====" -f $pass.Count, $fail.Count, $skip.Count, $suiteSw.Elapsed.TotalSeconds) -ForegroundColor $(if ($fail.Count) { 'Red' } else { 'Green' }) # GitHub Actions job summary — renders the PASS/FAIL table on the run page (no-op locally). Written # before the non-zero exit so a failing run still shows exactly which checks failed. @@ -289,7 +306,7 @@ try { [void]$md.AppendLine("| Result | Area | Check | Perf | Detail |") [void]$md.AppendLine("|:------:|------|-------|------|--------|") foreach ($r in $results) { - [void]$md.AppendLine("| $(if ($r.pass) { '✅' } else { '❌' }) | $($r.area) | $($r.name) | $($r.perf) | $($r.detail) |") + [void]$md.AppendLine("| $(if ($r.skip) { 'SKIP' } elseif ($r.pass) { '✅' } else { '❌' }) | $($r.area) | $($r.name) | $($r.perf) | $($r.detail) |") } Add-Content -Path $env:GITHUB_STEP_SUMMARY -Value $md.ToString() } diff --git a/perf-harness/scripts/run-orch-bench.ps1 b/perf-harness/scripts/run-orch-bench.ps1 new file mode 100644 index 0000000..f361982 --- /dev/null +++ b/perf-harness/scripts/run-orch-bench.ps1 @@ -0,0 +1,238 @@ +<# +.SYNOPSIS + Orchestration benchmark: the same workloads against any Craft image, on Azurite or a real storage account, so + engines can be compared like for like. + +.DESCRIPTION + Brings up docker-compose.e2e-azure.yml (Azurite + the SUT with PerfApi, BgPoolSize=4, 2 CPUs), optionally pointing + the SUT at a real account (-Storage azure, connection from CRAFT_TEST_TABLE_CONNECTION) under a unique table + prefix, runs the scenarios below one after another, writes one JSON result file, and tears the stack down. + Uses only Start-CraftOrchestrator, the PerfE2E recording task/PostExecution and WorkerMetricsBridge.CancelRun, + so it runs unchanged on engines before and after the storage-first redesign. Timings come from task start/end + ticks recorded inside the SUT. + +.EXAMPLE + pwsh scripts/run-orch-bench.ps1 -SutImage craft:pre-v2-c98a091 -Storage azurite -Label pre-azurite +#> +[CmdletBinding()] +param( + [Parameter(Mandatory)][string]$SutImage, + [ValidateSet('azurite', 'azure')][string]$Storage = 'azurite', + [string]$Label = 'bench', + [int]$Port = 5399, + [string]$OutDir = (Join-Path ([System.IO.Path]::GetTempPath()) 'craft-bench'), + [string[]]$Only +) + +$ErrorActionPreference = 'Stop' +# pwsh -File passes "a,b" as one string. +$Only = @($Only | ForEach-Object { $_ -split ',' } | Where-Object { $_ }) +$here = Split-Path -Parent $MyInvocation.MyCommand.Path +$root = Split-Path -Parent $here +$compose = @('-f', (Join-Path $root 'docker-compose.e2e-azure.yml'), '-f', (Join-Path $root 'docker-compose.bench.yml')) +$base = "http://127.0.0.1:$Port" +$Azurite = 'DefaultEndpointsProtocol=http;AccountName=devstoreaccount1;AccountKey=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==;BlobEndpoint=http://azurite:10000/devstoreaccount1;QueueEndpoint=http://azurite:10001/devstoreaccount1;TableEndpoint=http://azurite:10002/devstoreaccount1;' + +$Prefix = 'Bench' + [guid]::NewGuid().ToString('N').Substring(0, 8) +$env:SUT_IMAGE = $SutImage +$env:SUT_PORT = "$Port" +$env:BENCH_PREFIX = $Prefix +$env:BENCH_STORAGE = if ($Storage -eq 'azure') { + if (-not $env:CRAFT_TEST_TABLE_CONNECTION) { throw 'CRAFT_TEST_TABLE_CONNECTION is not set' } + $env:CRAFT_TEST_TABLE_CONNECTION +} else { $Azurite } + +$Results = [ordered]@{ + label = $Label; image = $SutImage; storage = $Storage; prefix = $Prefix + account = if ($Storage -eq 'azure') { [regex]::Match($env:CRAFT_TEST_TABLE_CONNECTION, 'AccountName=([^;]+)').Groups[1].Value } else { 'azurite' } + startedUtc = [DateTime]::UtcNow.ToString('o'); scenarios = [ordered]@{} +} + +function Info($m) { Write-Host "[bench $Label] $m" -ForegroundColor Cyan } +function Get-Api([string]$Path, [int]$TimeoutSec = 60) { Invoke-RestMethod "$base$Path" -TimeoutSec $TimeoutSec } +function New-Id { [guid]::NewGuid().ToString('N').Substring(0, 6) } +function Ms([long]$Ticks) { [math]::Round($Ticks / 10000.0) } + +function Start-Runs([string]$Ns, $Runs, [string]$Sink = 'cache', [int]$TimeoutSec = 600) { + $Body = @{ ns = $Ns; sink = $Sink; runs = @($Runs) } | ConvertTo-Json -Depth 30 -Compress + $Sw = [Diagnostics.Stopwatch]::StartNew() + $R = Invoke-RestMethod "$base/API/PerfE2EStart" -Method Post -ContentType 'application/json' -Body $Body -TimeoutSec $TimeoutSec + return [pscustomobject]@{ runs = @($R.runs); callMs = $Sw.ElapsedMilliseconds } +} + +function Get-Summary([string]$Ns, [string]$Source = 'cache') { + $Map = @{} + foreach ($R in @((Get-Api "/API/PerfE2EState?ns=$Ns&source=$Source&summary=1").runs)) { $Map[$R.run] = $R } + return $Map +} + +function Wait-Until([scriptblock]$Condition, [int]$TimeoutSec, [int]$IntervalMs = 250) { + $Sw = [Diagnostics.Stopwatch]::StartNew() + while ($Sw.Elapsed.TotalSeconds -lt $TimeoutSec) { + $V = try { & $Condition } catch { $null } + if ($V) { return [pscustomobject]@{ ok = $true; value = $V; sec = $Sw.Elapsed.TotalSeconds } } + Start-Sleep -Milliseconds $IntervalMs + } + return [pscustomobject]@{ ok = $false; value = $null; sec = $Sw.Elapsed.TotalSeconds } +} + +function Wait-Ready([int]$TimeoutSec = 240) { + $W = Wait-Until { $H = Get-Api '/healthz' 10; if ($H.status -eq 'ready') { $true } } $TimeoutSec 1000 + if (-not $W.ok) { throw 'SUT never became ready' } +} + +function Add-Scenario([string]$Name, [hashtable]$Data) { + $Results.scenarios[$Name] = $Data + Info ("{0,-22} {1}" -f $Name, (($Data.GetEnumerator() | ForEach-Object { "$($_.Key)=$($_.Value)" }) -join ' ')) +} + +function Invoke-Scenario([string]$Name, [scriptblock]$Body) { + if ($Only -and $Name -notin $Only) { return } + try { & $Body } + catch { Add-Scenario $Name @{ error = $_.Exception.Message } } +} + +Info "compose up: $SutImage on $Storage (prefix $Prefix)" +docker compose @compose up -d | Out-Host +if ($LASTEXITCODE -ne 0) { throw 'compose up failed' } + +try { + Wait-Ready + Start-Sleep -Seconds 5 + + # 1. Throughput of short tasks, one run, with an aggregation. + Invoke-Scenario 'fanout-1000' { + $Ns = "f1k-$(New-Id)" + $S = Start-Runs $Ns @(@{ name = "BenchFan-$Ns"; label = 'fan'; tasks = 1000; post = @{ marker = 'm' } }) + $T0 = $S.runs[0].enqueueTicks + $W = Wait-Until { $M = Get-Summary $Ns; if ($M['fan'].posts -ge 1) { $M['fan'] } } 1800 500 + $Rows = @((Get-Api "/API/PerfE2EState?ns=$Ns").rows) + $Post = @($Rows.Where({ $_.kind -eq 'P' }))[0] + $Sec = if ($Post) { (Ms ($Post.ticks - $T0)) / 1000.0 } else { -1 } + Add-Scenario 'fanout-1000' @{ ok = $W.ok; createMs = $S.callMs; endToEndSec = [math]::Round($Sec, 1); tasksPerSec = [math]::Round(1000 / [math]::Max(0.1, $Sec), 1); ran = $W.value.tasks } + } + + # 2. Efficiency with real work: 200 tasks of 250 ms on 4 workers (ideal 12.5 s). + Invoke-Scenario 'fanout-200x250ms' { + $Ns = "f200-$(New-Id)" + $S = Start-Runs $Ns @(@{ name = "BenchWork-$Ns"; label = 'w'; tasks = 200; task = @{ holdms = 250 }; post = @{ marker = 'm' } }) + $T0 = $S.runs[0].enqueueTicks + $W = Wait-Until { $M = Get-Summary $Ns; if ($M['w'].posts -ge 1) { $M['w'] } } 1800 500 + $Sec = if ($W.ok) { (Ms ($W.value.maxEnd - $T0)) / 1000.0 } else { -1 } + Add-Scenario 'fanout-200x250ms' @{ ok = $W.ok; tasksDoneSec = [math]::Round($Sec, 1); efficiencyPct = [math]::Round(12.5 / [math]::Max(0.1, $Sec) * 100); idealSec = 12.5 } + } + + # 3./4. Many small runs queued at once (one task each, no aggregation). + foreach ($N in 300, 1000) { + Invoke-Scenario "runs-$N" { + $Ns = "r$N-$(New-Id)" + $Specs = @(for ($I = 0; $I -lt $N; $I++) { @{ name = "BenchMany-$Ns-$I"; label = "r$I"; tasks = 1 } }) + $S = Start-Runs $Ns $Specs 'cache' 1800 + $T0 = $S.runs[0].enqueueTicks + $W = Wait-Until { $M = Get-Summary $Ns; $Done = @($M.Values.Where({ $_.ended -ge 1 })).Count; if ($Done -ge $N) { $M } } 1800 1000 + $Last = if ($W.ok) { ($W.value.Values | Measure-Object -Property maxEnd -Maximum).Maximum } else { 0 } + $Sec = if ($W.ok) { (Ms ($Last - $T0)) / 1000.0 } else { -1 } + Add-Scenario "runs-$N" @{ ok = $W.ok; createMs = $S.callMs; allDoneSec = [math]::Round($Sec, 1); runsPerSec = [math]::Round($N / [math]::Max(0.1, $Sec), 1) } + } + } + + # 5. Creating a big run and how soon its first task starts. + Invoke-Scenario 'ttfs-5000' { + $Ns = "t5k-$(New-Id)" + $S = Start-Runs $Ns @(@{ name = "BenchBig-$Ns"; label = 'big'; tasks = 5000; task = @{ holdms = 50 } }) 'cache' 900 + $T0 = $S.runs[0].enqueueTicks + $W = Wait-Until { $M = Get-Summary $Ns; if ($M['big'].tasks -ge 1) { $M['big'] } } 600 50 + $First = if ($W.ok) { Ms ($W.value.minStart - $T0) } else { -1 } + $Cancel = Get-Api "/API/PerfE2EBridge?op=cancel&name=BenchBig-$Ns" 900 + Add-Scenario 'ttfs-5000' @{ ok = $W.ok; createMs = $S.callMs; firstStartMs = $First } + Start-Sleep -Seconds 5 + } + + # 6. An idle engine picking up a new run. + Invoke-Scenario 'idle-claim' { + Start-Sleep -Seconds 20 + $Ns = "idle-$(New-Id)" + $S = Start-Runs $Ns @(@{ name = "BenchIdle-$Ns"; label = 'i'; tasks = 1 }) + $T0 = $S.runs[0].enqueueTicks + $W = Wait-Until { $M = Get-Summary $Ns; if ($M['i'].tasks -ge 1) { $M['i'] } } 120 20 + Add-Scenario 'idle-claim' @{ ok = $W.ok; firstStartMs = $(if ($W.ok) { Ms ($W.value.minStart - $T0) } else { -1 }) } + } + + # 7. Per-step overhead of a sequential run. + Invoke-Scenario 'sequential-50' { + $Ns = "seq-$(New-Id)" + $S = Start-Runs $Ns @(@{ name = "BenchSeq-$Ns"; label = 's'; tasks = 50; Sequential = $true; post = @{ marker = 'm' } }) + $T0 = $S.runs[0].enqueueTicks + $W = Wait-Until { $M = Get-Summary $Ns; if ($M['s'].posts -ge 1) { $M['s'] } } 900 250 + $Sec = if ($W.ok) { (Ms ($W.value.maxEnd - $T0)) / 1000.0 } else { -1 } + Add-Scenario 'sequential-50' @{ ok = $W.ok; stepsDoneSec = [math]::Round($Sec, 1); msPerStep = [math]::Round($Sec * 1000 / 50) } + } + + # 8. A high-priority run arriving behind a big backlog in a lower band. + Invoke-Scenario 'priority-jump' { + $Ns = "pj-$(New-Id)" + $null = Start-Runs $Ns @(@{ name = "BenchBacklog-$Ns"; label = 'bl'; tasks = 3000; task = @{ holdms = 100 }; Priority = 6 }) 'cache' 900 + $null = Wait-Until { $M = Get-Summary $Ns; if ($M['bl'].tasks -ge 8) { $true } } 300 100 + $S = Start-Runs $Ns @(@{ name = "BenchUrgent-$Ns"; label = 'u'; tasks = 4; Priority = 1 }) + $T0 = $S.runs[0].enqueueTicks + $W = Wait-Until { $M = Get-Summary $Ns; if ($M['u'].ended -ge 4) { $M['u'] } } 600 50 + $null = Get-Api "/API/PerfE2EBridge?op=cancel&name=BenchBacklog-$Ns" 900 + Add-Scenario 'priority-jump' @{ ok = $W.ok; firstStartMs = $(if ($W.ok) { Ms ($W.value.minStart - $T0) } else { -1 }); allDoneMs = $(if ($W.ok) { Ms ($W.value.maxEnd - $T0) } else { -1 }) } + Start-Sleep -Seconds 5 + } + + # 9. Cancelling a large backlog while it runs, to the run finalising (its aggregation running). + Invoke-Scenario 'cancel-5000' { + $Ns = "cx-$(New-Id)" + $null = Start-Runs $Ns @(@{ name = "BenchCancel-$Ns"; label = 'c'; tasks = 5000; task = @{ holdms = 200 }; post = @{ marker = 'm' } }) 'cache' 900 + $null = Wait-Until { $M = Get-Summary $Ns; if ($M['c'].tasks -ge 8) { $true } } 300 100 + $Sw = [Diagnostics.Stopwatch]::StartNew() + $Cancel = Get-Api "/API/PerfE2EBridge?op=cancel&name=BenchCancel-$Ns" 1800 + $CallMs = $Sw.ElapsedMilliseconds + $W = Wait-Until { $M = Get-Summary $Ns; if ($M['c'].posts -ge 1) { $M['c'] } } 1800 250 + Add-Scenario 'cancel-5000' @{ ok = $W.ok; callMs = $CallMs; finalisedSec = [math]::Round($Sw.Elapsed.TotalSeconds, 1); reported = $Cancel.cancelled; tasksRan = $W.value.tasks } + } + + # 10. Memory after the work above. + Invoke-Scenario 'memory' { + $M = Get-Api '/API/PerfE2EBridge?op=summary' + Add-Scenario 'memory' @{ rssMB = $M.rssMB; heapMB = $M.heapMB; committedMB = $M.committedMB } + } + + # 11./12. Recovery mid-run (200 x 1 s tasks): a graceful recycle (SIGTERM, 30 s grace) and a crash (SIGKILL). + # Time from the stop to every task done and the aggregation run; executions > 200 means tasks ran twice. + foreach ($Mode in 'graceful', 'crash') { + Invoke-Scenario "restart-$Mode" { + $Ns = "rs-$(New-Id)" + $null = Start-Runs $Ns @(@{ name = "BenchRestart-$Ns"; label = 'rs'; tasks = 200; task = @{ holdms = 1000 }; post = @{ marker = 'm' } }) 'table' + $null = Wait-Until { $M = Get-Summary $Ns 'table'; if ($M['rs'].tasks -ge 20) { $true } } 300 500 + $Sw = [Diagnostics.Stopwatch]::StartNew() + if ($Mode -eq 'graceful') { docker stop -t 30 craft-e2e-az-sut | Out-Null } else { docker kill craft-e2e-az-sut | Out-Null } + $Exit = docker inspect craft-e2e-az-sut --format '{{.State.ExitCode}}' + $StopSec = $Sw.Elapsed.TotalSeconds + docker start craft-e2e-az-sut | Out-Null + Wait-Ready + $ReadySec = $Sw.Elapsed.TotalSeconds + $W = Wait-Until { $M = Get-Summary $Ns 'table'; if ($M['rs'].posts -ge 1) { $M['rs'] } } 1800 2000 + $Rows = @((Get-Api "/API/PerfE2EState?ns=$Ns&source=table" 120).rows.Where({ $_.kind -eq 'T' })) + $Distinct = @($Rows | Group-Object idx).Count + Add-Scenario "restart-$Mode" @{ ok = $W.ok; stopSec = [math]::Round($StopSec, 1); exitCode = $Exit; readySec = [math]::Round($ReadySec, 1); recoveredSec = [math]::Round($Sw.Elapsed.TotalSeconds, 1); distinctTasks = $Distinct; executions = $Rows.Count } + } + } +} +finally { + $Results.endedUtc = [DateTime]::UtcNow.ToString('o') + $null = New-Item -ItemType Directory -Force -Path $OutDir + $File = Join-Path $OutDir "$Label.json" + $Results | ConvertTo-Json -Depth 10 | Set-Content -Path $File -Encoding utf8 + Info "results: $File" + docker compose @compose logs --no-color sut 2>$null | Set-Content -Path (Join-Path $OutDir "$Label.sut.log") -Encoding utf8 + docker compose @compose down -v | Out-Null + if ($Storage -eq 'azure') { + # Only this run's tables: every one starts with its unique prefix. The connection string stays in the environment. + Info "dropping $Prefix* tables from $($Results.account)" + $Tables = @(az storage table list --connection-string $env:CRAFT_TEST_TABLE_CONNECTION --query "[?starts_with(name, '$Prefix')].name" -o tsv 2>$null) + foreach ($T in $Tables) { if ($T) { az storage table delete --name $T --connection-string $env:CRAFT_TEST_TABLE_CONNECTION -o none 2>$null } } + Info "dropped $($Tables.Count) table(s)" + } +} diff --git a/tests/Craft.Tests/AzureTableStoreLargeEntityTests.cs b/tests/Craft.Tests/AzureTableStoreLargeEntityTests.cs index 20a57c4..064221a 100644 --- a/tests/Craft.Tests/AzureTableStoreLargeEntityTests.cs +++ b/tests/Craft.Tests/AzureTableStoreLargeEntityTests.cs @@ -7,10 +7,8 @@ namespace Craft.Tests; /// -/// Tests here allocate multi-MB strings to force the storage size limits, and heap-delta measurement -/// tests (see ) cannot share a process with concurrent allocation. -/// Marking this collection non-parallel puts it in the same sequential phase as those, so the two never -/// run at once. See the note on for the failure mode this avoids. +/// Tests here allocate multi-MB strings to force the storage size limits, so they run alone rather than +/// alongside tests that are sensitive to heap pressure. /// [CollectionDefinition(LargeAllocationSerialTests.Name, DisableParallelization = true)] public class LargeAllocationSerialTests @@ -237,4 +235,41 @@ public async Task Delete_RemovesEveryPartRow_OfASplitEntity() await foreach (var _ in fx.Store.QueryPartitionAsync(fx.Table, "p")) any = true; Assert.False(any); } + + [Fact] + public async Task Delete_TakesOnlyItsOwnPartRows_NotNeighboursSharingThePrefix() + { + await using var fx = await Fixture.TryConnectAsync(); + if (fx == null) return; + + // "task1" splits; "task1-partner" is an unrelated plain row and "task1-part9" an unrelated split + // entity, so both sit inside task1's "-part" key range without belonging to it. + await fx.Store.UpsertAsync(fx.Table, Row("p", "task1", Text(1_200_000))); + await fx.Store.UpsertAsync(fx.Table, Row("p", "task1-partner", "small")); + await fx.Store.UpsertAsync(fx.Table, Row("p", "task1-part9", Text(1_200_000, 'y'))); + + await fx.Store.DeleteAsync(fx.Table, "p", "task1"); + + Assert.Null(await fx.Store.GetAsync(fx.Table, "p", "task1")); + Assert.Equal("small", (await fx.Store.GetAsync(fx.Table, "p", "task1-partner"))!.GetString("ParametersJson")); + Assert.Equal(Text(1_200_000, 'y'), (await fx.Store.GetAsync(fx.Table, "p", "task1-part9"))!.GetString("ParametersJson")); + Assert.True(await fx.PhysicalRowCountAsync("p") > 2); + } + + [Fact] + public async Task FilteredQuery_OnASplitEntity_DoesNotReturnNeighboursSharingThePrefix() + { + await using var fx = await Fixture.TryConnectAsync(); + if (fx == null) return; + + await fx.Store.UpsertAsync(fx.Table, Row("p", "task1", Text(1_200_000))); + await fx.Store.UpsertAsync(fx.Table, Row("p", "task1-partner", "small")); + await fx.Store.UpsertAsync(fx.Table, Row("p", "task1-part9", Text(1_200_000, 'y'))); + + var rows = new List(); + await foreach (var r in fx.Store.QueryTableAsync(fx.Table, "PartitionKey eq 'p' and RowKey eq 'task1'")) rows.Add(r); + + Assert.Equal(["task1"], rows.Select(r => r.RowKey)); + Assert.Equal(Text(1_200_000), rows[0].GetString("ParametersJson")); + } } diff --git a/tests/Craft.Tests/CountingTableStore.cs b/tests/Craft.Tests/CountingTableStore.cs new file mode 100644 index 0000000..b977fe2 --- /dev/null +++ b/tests/Craft.Tests/CountingTableStore.cs @@ -0,0 +1,135 @@ +using System.Collections.Concurrent; +using System.Runtime.CompilerServices; +using Craft.Storage; + +namespace Craft.Tests; + +/// +/// Counts what reaches the table backend, per table: point reads, queries, the rows they returned and the +/// pages that would have taken, transactions and writes. Wraps another store and forwards everything. +/// Pages are rows actually consumed divided by the page size asked for (1,000 when none), so a query the +/// caller stops reading early costs only the pages it read, as with the Azure SDK's lazy paging. +/// +internal sealed class CountingTableStore(ICraftTableStore inner) : ICraftTableStore +{ + public sealed class Counts + { + public int PointReads, Queries, Rows, Pages, Submits, Upserts, BatchUpserts, Deletes; + + /// Range queries that did not ask for a page size, so each request may return 1,000 rows. + public int UnboundedRanges; + public override string ToString() => + $"reads={PointReads} queries={Queries} rows={Rows} pages={Pages} submits={Submits} upserts={Upserts} batches={BatchUpserts} deletes={Deletes} unboundedRanges={UnboundedRanges}"; + } + + private readonly ConcurrentDictionary _byTable = new(StringComparer.Ordinal); + + public Counts For(string table) => _byTable.GetOrAdd(table, _ => new Counts()); + public void Reset() => _byTable.Clear(); + + /// Every table's counts summed. + public Counts Total() + { + var t = new Counts(); + foreach (var c in _byTable.Values) + { + t.PointReads += c.PointReads; t.Queries += c.Queries; t.Rows += c.Rows; t.Pages += c.Pages; + t.Submits += c.Submits; t.Upserts += c.Upserts; t.BatchUpserts += c.BatchUpserts; t.Deletes += c.Deletes; + t.UnboundedRanges += c.UnboundedRanges; + } + return t; + } + + private async IAsyncEnumerable Count(string table, IAsyncEnumerable rows, int pageSize, + [EnumeratorCancellation] CancellationToken ct = default) + { + var c = For(table); + Interlocked.Increment(ref c.Queries); + Interlocked.Increment(ref c.Pages); + var n = 0; + await foreach (var row in rows.WithCancellation(ct)) + { + if (n > 0 && n % pageSize == 0) Interlocked.Increment(ref c.Pages); + n++; + Interlocked.Increment(ref c.Rows); + yield return row; + } + } + + public Task PingAsync(CancellationToken ct = default) => inner.PingAsync(ct); + public Task EnsureTableAsync(string table, CancellationToken ct = default) => inner.EnsureTableAsync(table, ct); + + public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) + { + Interlocked.Increment(ref For(table).Upserts); + return inner.UpsertAsync(table, row, ct); + } + + public Task UpsertBatchAsync(string table, string partitionKey, IReadOnlyList rows, CancellationToken ct = default) + { + Interlocked.Increment(ref For(table).BatchUpserts); + return inner.UpsertBatchAsync(table, partitionKey, rows, ct); + } + + public Task TryReplaceBatchAsync(string table, string partitionKey, IReadOnlyList rows, CancellationToken ct = default) + { + Interlocked.Increment(ref For(table).Submits); + return inner.TryReplaceBatchAsync(table, partitionKey, rows, ct); + } + + public Task GetAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) + { + Interlocked.Increment(ref For(table).PointReads); + return inner.GetAsync(table, partitionKey, rowKey, ct); + } + + public IAsyncEnumerable QueryPartitionAsync(string table, string partitionKey, CancellationToken ct = default) => + Count(table, inner.QueryPartitionAsync(table, partitionKey, ct), 1000, ct); + + public IAsyncEnumerable QueryTableAsync(string table, CancellationToken ct = default) => + Count(table, inner.QueryTableAsync(table, ct), 1000, ct); + + public IAsyncEnumerable QueryTableAsync(string table, string? filter, CancellationToken ct = default) => + Count(table, inner.QueryTableAsync(table, filter, ct), 1000, ct); + + public IAsyncEnumerable QueryTableAsync(string table, string? filter, IReadOnlyList? properties, + CancellationToken ct = default) => + Count(table, inner.QueryTableAsync(table, filter, properties, ct), 1000, ct); + + public IAsyncEnumerable QueryTableAsync(string table, string? filter, int maxPerPage, CancellationToken ct = default) => + Count(table, inner.QueryTableAsync(table, filter, maxPerPage, ct), Math.Max(1, maxPerPage), ct); + + public IAsyncEnumerable QueryRowKeyRangeAsync(string table, string partitionKey, string fromRowKey, string toRowKey, + IReadOnlyList? properties = null, int? maxPerPage = null, CancellationToken ct = default) + { + if (maxPerPage == null) Interlocked.Increment(ref For(table).UnboundedRanges); + return Count(table, inner.QueryRowKeyRangeAsync(table, partitionKey, fromRowKey, toRowKey, properties, maxPerPage, ct), + maxPerPage ?? 1000, ct); + } + + public Task TrySubmitAsync(string table, string partitionKey, IReadOnlyList ops, CancellationToken ct = default) + { + Interlocked.Increment(ref For(table).Submits); + return inner.TrySubmitAsync(table, partitionKey, ops, ct); + } + + public Task DeleteTableAsync(string table, CancellationToken ct = default) => inner.DeleteTableAsync(table, ct); + + public Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) + { + Interlocked.Increment(ref For(table).Deletes); + return inner.DeleteAsync(table, partitionKey, rowKey, ct); + } + + public Task DeleteBatchAsync(string table, string partitionKey, IReadOnlyList rowKeys, CancellationToken ct = default) + { + Interlocked.Increment(ref For(table).Deletes); + return inner.DeleteBatchAsync(table, partitionKey, rowKeys, ct); + } + + public Task DeletePartitionAsync(string table, string partitionKey, CancellationToken ct = default) + { + Interlocked.Increment(ref For(table).Deletes); + return inner.DeletePartitionAsync(table, partitionKey, ct); + } +} diff --git a/tests/Craft.Tests/CraftAuthMiddlewareTests.cs b/tests/Craft.Tests/CraftAuthMiddlewareTests.cs new file mode 100644 index 0000000..4690d86 --- /dev/null +++ b/tests/Craft.Tests/CraftAuthMiddlewareTests.cs @@ -0,0 +1,152 @@ +using System.Text; +using System.Text.Json; +using Craft.Auth; +using Craft.Hosting; +using Microsoft.AspNetCore.Http; +using Microsoft.Extensions.Logging.Abstractions; + +namespace Craft.Tests; + +/// +/// Pins how an EasyAuth principal is rewritten for the hosted app. An app-only token must become an API +/// client (idp "aad", name = app id, no allowedUsers lookup); a signed-in user must be authorised against +/// allowedUsers and keep its own identity even when the token names the app it signed in through. Both +/// keep the token's claims. Getting the split wrong either locks out every API client or runs a user as +/// an app. +/// +public class CraftAuthMiddlewareTests +{ + private const string OidClaim = "http://schemas.microsoft.com/identity/claims/objectidentifier"; + private static readonly string[] AdminRoles = ["admin"]; + private static readonly string?[] UserIdentifiers = ["a@b.com", "user-oid"]; + + private static string EasyAuthHeader(params (string Typ, string Val)[] claims) + { + var items = string.Join(",", claims.Select(c => $$"""{"typ":"{{c.Typ}}","val":"{{c.Val}}"}""")); + return Convert.ToBase64String(Encoding.UTF8.GetBytes($$"""{"auth_typ":"aad","claims":[{{items}}]}""")); + } + + private static DefaultHttpContext Request(string principalHeader) + { + var context = new DefaultHttpContext(); + context.Request.Headers["x-ms-client-principal"] = principalHeader; + context.Request.Headers["x-ms-client-principal-idp"] = "aad"; + context.Response.Body = new MemoryStream(); + return context; + } + + private sealed class RoleLookup(string[]? roles) + { + public List Calls { get; } = []; + + public Task Resolve(IEnumerable ids) + { + Calls.Add([.. ids]); + return Task.FromResult(roles); + } + } + + private static async Task Normalise(HttpContext context, RoleLookup lookup) => + await CraftAuthMiddleware.TryNormalisePrincipalAsync( + context, lookup.Resolve, NullLogger.Instance, context.Request.Headers["x-ms-client-principal"].ToString()); + + private static JsonElement Principal(HttpContext context) + { + using var document = EasyAuthPrincipal.Decode(context.Request.Headers["x-ms-client-principal"].ToString()); + return document.RootElement.Clone(); + } + + private static string? Claim(JsonElement principal, string typ) => + principal.GetProperty("claims").EnumerateArray() + .Where(c => c.GetProperty("typ").GetString() == typ) + .Select(c => c.GetProperty("val").GetString()) + .FirstOrDefault(); + + [Fact] + public async Task AppOnlyToken_BecomesAnApiClient() + { + var lookup = new RoleLookup(AdminRoles); + var context = Request(EasyAuthHeader(("appid", "app-1"), (OidClaim, "sp-oid"), ("idtyp", "app"), ("azpacr", "1"))); + + Assert.True(await Normalise(context, lookup)); + + Assert.Equal("aad", context.Request.Headers["x-ms-client-principal-idp"].ToString()); + Assert.Equal("app-1", context.Request.Headers["x-ms-client-principal-name"].ToString()); + var principal = Principal(context); + Assert.Equal("app-1", principal.GetProperty("userDetails").GetString()); + Assert.Equal("sp-oid", principal.GetProperty("userId").GetString()); + Assert.Equal(0, principal.GetProperty("userRoles").GetArrayLength()); + Assert.Equal("1", Claim(principal, "azpacr")); + Assert.Empty(lookup.Calls); // an API client is never looked up in allowedUsers + } + + [Fact] + public async Task SignedInUser_IsAuthorisedAndKeepsItsOwnIdentity() + { + var lookup = new RoleLookup(AdminRoles); + var context = Request(EasyAuthHeader( + ("upn", "a@b.com"), (OidClaim, "user-oid"), ("azp", "client-abc"), ("scp", "user_impersonation"))); + + Assert.True(await Normalise(context, lookup)); + + // The token names the app the user signed in through, but the caller is the user, not that app. + Assert.Equal("azureStaticWebApps", context.Request.Headers["x-ms-client-principal-idp"].ToString()); + Assert.Equal("a@b.com", context.Request.Headers["x-ms-client-principal-name"].ToString()); + var principal = Principal(context); + Assert.Equal("a@b.com", principal.GetProperty("userDetails").GetString()); + Assert.Equal("user-oid", principal.GetProperty("userId").GetString()); + Assert.Equal("admin", principal.GetProperty("userRoles")[0].GetString()); + Assert.Equal("client-abc", Claim(principal, "azp")); + Assert.Equal("user_impersonation", Claim(principal, "scp")); + Assert.Equal(UserIdentifiers, Assert.Single(lookup.Calls)); + } + + [Fact] + public async Task UserNotInAllowedUsers_IsRejectedAndThePrincipalStripped() + { + var context = Request(EasyAuthHeader(("upn", "a@b.com"), ("azp", "client-abc"))); + + Assert.False(await Normalise(context, new RoleLookup(null))); + + Assert.Equal(StatusCodes.Status401Unauthorized, context.Response.StatusCode); + Assert.False(context.Request.Headers.ContainsKey("x-ms-client-principal")); + } + + [Fact] + public async Task NormalisedPrincipal_PassesThroughUntouched() + { + var normalised = EasyAuthPrincipal.EncodeNormalised( + JsonDocument.Parse("""{"claims":[{"typ":"upn","val":"a@b.com"}]}""").RootElement, + "aad", "user-oid", "a@b.com", AdminRoles); + var lookup = new RoleLookup(null); + var context = Request(normalised); + + Assert.True(await Normalise(context, lookup)); + + Assert.Equal(normalised, context.Request.Headers["x-ms-client-principal"].ToString()); + Assert.Empty(lookup.Calls); + } + + [Fact] + public async Task PrincipalWithNoIdentity_PassesThroughUntouched() + { + var header = EasyAuthHeader(("scp", "user_impersonation")); + var lookup = new RoleLookup(AdminRoles); + var context = Request(header); + + Assert.True(await Normalise(context, lookup)); + + Assert.Equal(header, context.Request.Headers["x-ms-client-principal"].ToString()); + Assert.Empty(lookup.Calls); + } + + [Fact] + public async Task UnparseablePrincipal_PassesThroughUntouched() + { + var context = Request("not-base64!!"); + + Assert.True(await Normalise(context, new RoleLookup(AdminRoles))); + + Assert.Equal("not-base64!!", context.Request.Headers["x-ms-client-principal"].ToString()); + } +} diff --git a/tests/Craft.Tests/EasyAuthPrincipalTests.cs b/tests/Craft.Tests/EasyAuthPrincipalTests.cs index 146d2d5..a04f7c8 100644 --- a/tests/Craft.Tests/EasyAuthPrincipalTests.cs +++ b/tests/Craft.Tests/EasyAuthPrincipalTests.cs @@ -210,4 +210,41 @@ public void Decode_RejectsGarbage() // The middleware catches this and passes the request through as anonymous rather than 500ing. Assert.ThrowsAny(() => EasyAuthPrincipal.Decode("not-base64!!")); } + + [Fact] + public void Normalised_KeepsTheTokenClaims() + { + // The hosted app reads claims beyond the identity (e.g. azp, the app a user signed in through). + var source = WithClaims(("upn", "a@b.com"), ("azp", "client-abc"), ("scp", "user_impersonation")); + + using var result = EasyAuthPrincipal.Decode( + EasyAuthPrincipal.EncodeNormalised(source, "aad", "oid-1", "a@b.com", AdminRole)); + var root = result.RootElement; + + Assert.Equal("a@b.com", root.GetProperty("userDetails").GetString()); + Assert.Equal("admin", root.GetProperty("userRoles")[0].GetString()); + Assert.Equal(source.GetProperty("claims").GetRawText(), root.GetProperty("claims").GetRawText()); + Assert.Equal("client-abc", EasyAuthPrincipal.ExtractClaims(root).AppId); + } + + [Fact] + public void Normalised_IsNeverTransformedAgain() + { + var source = WithClaims(("upn", "a@b.com")); + + using var result = EasyAuthPrincipal.Decode( + EasyAuthPrincipal.EncodeNormalised(source, "aad", "oid-1", "a@b.com", Array.Empty())); + + Assert.False(EasyAuthPrincipal.NeedsTransform(result.RootElement)); + } + + [Fact] + public void Normalised_WithoutClaims_EmitsAnEmptyArray() + { + using var result = EasyAuthPrincipal.Decode( + EasyAuthPrincipal.EncodeNormalised(Principal("{}"), "aad", "id", "name", Array.Empty())); + + Assert.Equal(JsonValueKind.Array, result.RootElement.GetProperty("claims").ValueKind); + Assert.Equal(0, result.RootElement.GetProperty("claims").GetArrayLength()); + } } diff --git a/tests/Craft.Tests/FaultyTableStore.cs b/tests/Craft.Tests/FaultyTableStore.cs new file mode 100644 index 0000000..0b8f7ed --- /dev/null +++ b/tests/Craft.Tests/FaultyTableStore.cs @@ -0,0 +1,43 @@ +using Craft.Storage; + +namespace Craft.Tests; + +/// Fails chosen writes, so a test can stop a sequence of writes at an exact point. +internal sealed class FaultyTableStore(ICraftTableStore inner) : ICraftTableStore +{ + public Func? FailUpsert; // (table, pk, rk) => fail + public Func, bool>? FailSubmit; + public Func? FailDelete; // (table, rk) => fail + + public Task PingAsync(CancellationToken ct = default) => inner.PingAsync(ct); + public Task EnsureTableAsync(string table, CancellationToken ct = default) => inner.EnsureTableAsync(table, ct); + public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) => + FailUpsert?.Invoke(table, row.PartitionKey, row.RowKey) == true + ? throw new InvalidOperationException($"injected upsert failure on {table}") + : inner.UpsertAsync(table, row, ct); + public Task UpsertBatchAsync(string table, string partitionKey, IReadOnlyList rows, CancellationToken ct = default) => + inner.UpsertBatchAsync(table, partitionKey, rows, ct); + public Task TryReplaceBatchAsync(string table, string partitionKey, IReadOnlyList rows, CancellationToken ct = default) => + inner.TryReplaceBatchAsync(table, partitionKey, rows, ct); + public Task GetAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) => + inner.GetAsync(table, partitionKey, rowKey, ct); + public IAsyncEnumerable QueryPartitionAsync(string table, string partitionKey, CancellationToken ct = default) => + inner.QueryPartitionAsync(table, partitionKey, ct); + public IAsyncEnumerable QueryTableAsync(string table, CancellationToken ct = default) => inner.QueryTableAsync(table, ct); + public IAsyncEnumerable QueryTableAsync(string table, string? filter, int maxPerPage, CancellationToken ct = default) => + inner.QueryTableAsync(table, filter, maxPerPage, ct); + public IAsyncEnumerable QueryRowKeyRangeAsync(string table, string partitionKey, string fromRowKey, string toRowKey, + IReadOnlyList? properties = null, int? maxPerPage = null, CancellationToken ct = default) => + inner.QueryRowKeyRangeAsync(table, partitionKey, fromRowKey, toRowKey, properties, maxPerPage, ct); + public Task TrySubmitAsync(string table, string partitionKey, IReadOnlyList ops, CancellationToken ct = default) => + FailSubmit?.Invoke(table, ops) == true + ? throw new InvalidOperationException($"injected transaction failure on {table}") + : inner.TrySubmitAsync(table, partitionKey, ops, ct); + public Task DeleteTableAsync(string table, CancellationToken ct = default) => inner.DeleteTableAsync(table, ct); + public Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) => + FailDelete?.Invoke(table, rowKey) == true + ? throw new InvalidOperationException($"injected delete failure on {table}") + : inner.DeleteAsync(table, partitionKey, rowKey, ct); + public Task DeletePartitionAsync(string table, string partitionKey, CancellationToken ct = default) => + inner.DeletePartitionAsync(table, partitionKey, ct); +} diff --git a/tests/Craft.Tests/JobDescriptorRehydrationTests.cs b/tests/Craft.Tests/JobDescriptorRehydrationTests.cs index 9f12c89..1550d7b 100644 --- a/tests/Craft.Tests/JobDescriptorRehydrationTests.cs +++ b/tests/Craft.Tests/JobDescriptorRehydrationTests.cs @@ -1,99 +1,18 @@ -using System.Runtime.CompilerServices; using Craft.Configuration; using Craft.Orchestration; using Craft.PowerShellHost; -using Craft.Storage; using Microsoft.Extensions.Configuration; using Microsoft.Extensions.Logging.Abstractions; namespace Craft.Tests; /// -/// Covers the descriptor queue's dispatch-time rehydration: what the resolver is handed, what it costs, -/// and what happens when the descriptor has gone stale underneath it. -/// -/// The design question these answer is "what does an extra storage read per dispatch cost at ~200 -/// dispatches/min?" — the answer being that the steady-state path performs ZERO reads, because the run -/// is already live in _activeRuns and object identity must be preserved anyway (see -/// OrchestratorService.ResolveTaskWorkAsync). Reads happen only on the recovery path. +/// Covers the descriptor queue's dispatch-time rehydration: what the resolver is handed, and what happens +/// when the descriptor has gone stale underneath it. /// public class JobDescriptorRehydrationTests { - /// An in-memory that counts point reads. - private sealed class CountingStore : ICraftTableStore - { - - // Claims are not exercised by this fake. Fail loudly rather than pretend the guard held — - // a silent 'true' here would look exactly like a successful claim. - public Task TryReplaceBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) => throw new NotSupportedException(); - private readonly Dictionary> _tables = new(); - public int PointReads; - public int PartitionScans; - - public Task PingAsync(CancellationToken ct = default) => Task.CompletedTask; - - public Task EnsureTableAsync(string table, CancellationToken ct = default) - { - if (!_tables.ContainsKey(table)) _tables[table] = new(); - return Task.CompletedTask; - } - - public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) - { - _tables[table][(row.PartitionKey, row.RowKey)] = row; - return Task.CompletedTask; - } - - public Task UpsertBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) - { - foreach (var r in rows) _tables[table][(r.PartitionKey, r.RowKey)] = r; - return Task.CompletedTask; - } - - public Task GetAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) - { - Interlocked.Increment(ref PointReads); - return Task.FromResult(_tables[table].TryGetValue((partitionKey, rowKey), out var r) ? r : null); - } - - public async IAsyncEnumerable QueryPartitionAsync(string table, string partitionKey, - [EnumeratorCancellation] CancellationToken ct = default) - { - Interlocked.Increment(ref PartitionScans); - foreach (var kv in _tables[table].Where(k => k.Key.Item1 == partitionKey).ToList()) - { - yield return kv.Value; - await Task.Yield(); - } - } - - public async IAsyncEnumerable QueryTableAsync(string table, - [EnumeratorCancellation] CancellationToken ct = default) - { - foreach (var kv in _tables[table].ToList()) - { - yield return kv.Value; - await Task.Yield(); - } - } - - public Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) - { - _tables[table].Remove((partitionKey, rowKey)); - return Task.CompletedTask; - } - - public Task DeletePartitionAsync(string table, string partitionKey, CancellationToken ct = default) - { - foreach (var k in _tables[table].Keys.Where(k => k.Item1 == partitionKey).ToList()) - _tables[table].Remove(k); - return Task.CompletedTask; - } - } - - private static (JobManager Jobs, CountingStore Store, OrchestratorTableStore Orch) NewHarness() + private static JobManager NewHarness() { var settings = new CraftSettings(); settings.Worker.BgPoolSize = 8; @@ -102,9 +21,7 @@ private static (JobManager Jobs, CountingStore Store, OrchestratorTableStore Orc var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); var jobs = new JobManager(NullLogger.Instance, settings, limiter); - var store = new CountingStore(); - var orch = new OrchestratorTableStore(NullLogger.Instance, settings, store); - return (jobs, store, orch); + return jobs; } private static Task Pump(JobManager jobs) => Task.Run(() => jobs.StartAsync(CancellationToken.None)); @@ -124,7 +41,7 @@ private static async Task WaitUntil(Func condition, int timeoutMs = [Fact] public async Task Dispatch_HandsTheDescriptorToTheResolver() { - var (jobs, _, _) = NewHarness(); + var jobs = NewHarness(); var seen = new List(); var done = 0; @@ -157,7 +74,7 @@ public async Task Dispatch_HandsTheDescriptorToTheResolver() [Fact] public async Task StaleDescriptor_IsSkipped_AndDispatchContinues() { - var (jobs, _, _) = NewHarness(); + var jobs = NewHarness(); var ran = 0; jobs.SetWorkResolver((d, _) => Task.FromResult?>( @@ -183,7 +100,7 @@ public async Task StaleDescriptor_IsSkipped_AndDispatchContinues() [Fact] public async Task DescriptorWithNoResolver_FailsTheJob_RatherThanDisappearing() { - var (jobs, _, _) = NewHarness(); + var jobs = NewHarness(); jobs.Enqueue(new JobDescriptor("run", "task", 0), "run-task"); _ = Pump(jobs); @@ -195,46 +112,6 @@ public async Task DescriptorWithNoResolver_FailsTheJob_RatherThanDisappearing() Assert.Contains("resolver", failed.LastError, StringComparison.OrdinalIgnoreCase); } - /// - /// The cost question. Rehydrating a task that is NOT already in memory costs exactly one partition - /// read of the run — the same read the crash-recovery path already performs. This pins the cost so a - /// future change that turns it into a per-task read shows up as a failure. - /// - [Fact] - public async Task Rehydration_FromStorage_CostsOneRunReadPerRun_NotPerTask() - { - var (_, counting, store) = NewHarness(); - await store.InitializeAsync(); - - var tasks = Enumerable.Range(0, 200).Select(i => new OrchestratorTaskItem - { - Id = $"Graph_tenant{i:D3}", - Status = "Pending", - Parameters = new Dictionary { ["TenantFilter"] = $"tenant{i:D3}.onmicrosoft.com" }, - }).ToList(); - - await store.UpsertRunAsync(new OrchestratorRun - { - Name = "CIPPDBCacheRun", - Status = "Running", - Priority = 5, - StartedUtc = DateTime.UtcNow, - Tasks = tasks, - TaskScriptName = "Invoke-CIPPDBCacheTask", - }); - await store.UpsertTaskBatchAsync("CIPPDBCacheRun", tasks); - - counting.PointReads = 0; - counting.PartitionScans = 0; - - var rehydrated = await store.GetRunAsync("CIPPDBCacheRun"); - - Assert.NotNull(rehydrated); - Assert.Equal(200, rehydrated!.Tasks.Count); - Assert.Equal(1, counting.PointReads); // the run row - Assert.Equal(1, counting.PartitionScans); // all 200 task rows in one partition query - } - // ── OWNERSHIP, as used by the Pending re-drive ─────────────────────────────────────────────────── /// @@ -245,7 +122,7 @@ await store.UpsertRunAsync(new OrchestratorRun [Fact] public void IsQueuedOrRunning_True_ForAQueuedJob() { - var (jobs, _, _) = NewHarness(); + var jobs = NewHarness(); jobs.Enqueue(new JobDescriptor("run-a", "task-1", 4), name: "run-a-task-1"); @@ -261,7 +138,7 @@ public void IsQueuedOrRunning_True_ForAQueuedJob() [Fact] public async Task IsQueuedOrRunning_False_OnceTheJobHasCompleted() { - var (jobs, _, _) = NewHarness(); + var jobs = NewHarness(); _ = Pump(jobs); var ran = new TaskCompletionSource(); @@ -277,7 +154,7 @@ public async Task IsQueuedOrRunning_False_OnceTheJobHasCompleted() [Fact] public void IsQueuedOrRunning_False_ForAnUnknownJob() { - var (jobs, _, _) = NewHarness(); + var jobs = NewHarness(); Assert.False(jobs.IsQueuedOrRunning("run-c-task-never-enqueued")); } diff --git a/tests/Craft.Tests/JobDurabilityTests.cs b/tests/Craft.Tests/JobDurabilityTests.cs index 4fbfd03..5644d7f 100644 --- a/tests/Craft.Tests/JobDurabilityTests.cs +++ b/tests/Craft.Tests/JobDurabilityTests.cs @@ -83,12 +83,6 @@ public Task DeletePartitionAsync(string table, string partitionKey, Cancellation } } - private static OrchestratorTableStore NewStore(out FakeStore backing) - { - backing = new FakeStore(); - return new OrchestratorTableStore(NullLogger.Instance, new CraftSettings(), backing); - } - private static JobManager NewJobManager() { var settings = new CraftSettings(); @@ -117,62 +111,6 @@ public void Cancelled(JobDescriptor descriptor) } } - // ── Per-task priority ─────────────────────────────────────────────────────────────────────────── - - [Fact] - public async Task TaskPriority_RoundTripsThroughStorage() - { - var store = NewStore(out _); - await store.InitializeAsync(); - - var tasks = new List - { - new() { Id = "inherits", Status = "Pending" }, // null ⇒ run priority - new() { Id = "overridden", Status = "Pending", Priority = 0 }, // escalated by an operator - }; - await store.UpsertRunAsync(new OrchestratorRun - { - Name = "run", - Status = "Running", - Priority = 5, - StartedUtc = DateTime.UtcNow, - Tasks = tasks, - }); - await store.UpsertTaskBatchAsync("run", tasks); - - var recovered = await store.GetRunAsync("run"); - - Assert.Equal(5, recovered!.Priority); - Assert.Null(recovered.Tasks.Single(t => t.Id == "inherits").Priority); - Assert.Equal(0, recovered.Tasks.Single(t => t.Id == "overridden").Priority); - } - - /// - /// The Replace-mode trap: the batched status writer rewrites the WHOLE row, so a column it does not - /// carry is erased. If Priority ever drops out of TaskStatusWrite, the override survives exactly - /// until the task moves to Running — this fails the moment that regresses. - /// - [Fact] - public async Task StatusWrite_PreservesTaskPriority_RatherThanErasingIt() - { - var store = NewStore(out _); - await store.InitializeAsync(); - - var task = new OrchestratorTaskItem { Id = "t1", Status = "Pending", Priority = 0 }; - await store.UpsertTaskBatchAsync("run", [task]); - await store.UpsertRunAsync(new OrchestratorRun { Name = "run", Status = "Running", Priority = 5 }); - - // A status transition through the coalescing writer's path. - await store.WriteTaskStatusBatchAsync( - [new TaskStatusWrite("run", "t1", "Running", "{}", 0, null, null, task.Priority)]); - - var recovered = await store.GetRunAsync("run"); - var reloaded = recovered!.Tasks.Single(); - - Assert.Equal("Running", reloaded.Status); - Assert.Equal(0, reloaded.Priority); - } - // ── Durable operator actions ──────────────────────────────────────────────────────────────────── [Fact] diff --git a/tests/Craft.Tests/JobManagerReEnqueueTests.cs b/tests/Craft.Tests/JobManagerReEnqueueTests.cs index 2c34ea1..4dadbb0 100644 --- a/tests/Craft.Tests/JobManagerReEnqueueTests.cs +++ b/tests/Craft.Tests/JobManagerReEnqueueTests.cs @@ -16,8 +16,9 @@ namespace Craft.Tests; /// fresh copy of that job is queued or running. IsQueuedOrRunning reads the frozen record and /// answers "no" for a job that is very much running. /// -/// That answer is load-bearing. JobQueuePump.ReleaseFinishedAsync treats it as "this job is done" and -/// DELETES the task's durable queue row. Observed live: the pump released 7-9 "finished" jobs every +/// That answer is load-bearing. WorkPump.Forget treats it as "this job is done" and stops renewing the +/// task's claim, so a long task loses its lease and runs again elsewhere. Observed live (when the pump +/// deleted the queue row instead): the pump released 7-9 "finished" jobs every /// second while only 8 could physically be running, on a run where individual tasks executed up to five /// times. /// @@ -74,10 +75,10 @@ public async Task ATaskEnqueuedAgainAfterItRan_IsReportedAsRunning_NotAsItsPrevi await WaitUntilAsync(() => Volatile.Read(ref ran) == 2, "second job never started"); - // The job IS running. Answering "no" here is what makes the pump delete a live job's queue row. + // The job IS running. Answering "no" here is what makes the pump stop renewing a live job's claim. Assert.True(jobs.IsQueuedOrRunning(id), "the manager reports a running job as finished — its record is frozen at the previous run's " + - "status, so JobQueuePump.ReleaseFinishedAsync will drop the durable row out from under it"); + "status, so WorkPump.Forget will stop renewing its claim"); release.Release(); await jobs.StopAsync(CancellationToken.None); diff --git a/tests/Craft.Tests/JobQueueAzuriteTests.cs b/tests/Craft.Tests/JobQueueAzuriteTests.cs deleted file mode 100644 index 074461d..0000000 --- a/tests/Craft.Tests/JobQueueAzuriteTests.cs +++ /dev/null @@ -1,226 +0,0 @@ -using Craft.Configuration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// Verifies the two properties that are the BACKEND's to provide, not ours, against a real Azure Tables -/// implementation (Azurite). The unit tests use a fake, and a fake is only ever as honest as whoever -/// wrote it — during development that fake got both of these wrong and the tests still passed. -/// -/// 1. Ordering is server-side and priority-first. Rows come back sorted by partition key then row key, -/// so "P00" precedes "P04" and, within a bucket, older precedes newer. Nothing sorts client-side. -/// 2. Exclusivity is real optimistic concurrency. Two claimers racing the same rows cannot both win, -/// because the second one's If-Match no longer matches. -/// -/// On leases: there is no Azure lease API here. Azure Tables does not have one — that is Blob. "Lease" -/// is only the name of an ordinary DateTimeOffset column, and the exclusivity comes entirely from ETag -/// conditional writes. -/// -/// Runs against the local emulator by default, or a real storage account when -/// CRAFT_TEST_TABLE_CONNECTION is set. Skipped, not failed, when neither is reachable, so this is safe -/// in CI — which does mean a skip looks like a pass; see TryConnectAsync for how to re-prove it. -/// -public class JobQueueAzuriteTests -{ - private static async Task TryConnectAsync() - { - var settings = new CraftSettings(); - - // Point at a real storage account by setting CRAFT_TEST_TABLE_CONNECTION; otherwise this runs - // against the local emulator. Same assertions either way — the properties under test are the - // backend's, and Azurite is a reimplementation of them, not the thing that ships. - var connection = Environment.GetEnvironmentVariable("CRAFT_TEST_TABLE_CONNECTION"); - if (!string.IsNullOrWhiteSpace(connection)) - settings.Auth.UserStorageConnection = connection; - else - settings.Storage.AllowDevelopmentStorage = true; - - // Unique per run so repeated runs cannot see each other's rows, and so a real account is never - // left holding fixtures under a name a later run would reuse. - settings.Orchestrator.TablePrefix = "azqt" + Guid.NewGuid().ToString("N")[..8]; - - var store = new AzureTableStore(settings); - try - { - using var cts = new CancellationTokenSource(TimeSpan.FromSeconds(3)); - await store.PingAsync(cts.Token); - } - catch - { - // Nothing to verify against. NOTE: an early return is indistinguishable from a pass — to - // re-prove these actually ran, swap the call sites' null check for Assert.NotNull(queue) - // and confirm they still pass. - return null; - } - - var queue = new JobQueueStore(NullLogger.Instance, settings, store); - await queue.InitializeAsync(); - return queue; - } - - private static DateTime At(int minute) => new(2026, 8, 9, 2, minute, 0, DateTimeKind.Utc); - - /// - /// The claim query narrows server-side with an OData $filter so a backlog is not paged to the client - /// on every pump tick. Every in-memory fake ignores that filter and returns everything, so this is - /// the only place the filter's semantics are actually exercised — and a filter that is subtly wrong - /// does not fail loudly, it silently hides claimable work and the queue stops draining. - /// - /// The three states that must survive it: never claimed, lease expired, lease live. - /// - [Fact] - public async Task ServerSideFilter_ReturnsFreeAndExpiredRows_AndHidesLiveOnes() - { - var queue = await TryConnectAsync(); - if (queue == null) return; - - await queue.EnqueueBatchAsync("run", [("free", 4), ("expired", 4), ("live", 4)], At(0)); - - // Give "live" a long lease and "expired" one that lapses almost immediately. - var first = await queue.ClaimBatchAsync("holder", 1, TimeSpan.FromMinutes(30)); - var second = await queue.ClaimBatchAsync("holder", 1, TimeSpan.FromMilliseconds(1)); - Assert.Single(first); - Assert.Single(second); - await Task.Delay(50); - - // Whatever is left free, plus the lapsed one — never the live one. - var claimable = await queue.ClaimBatchAsync("worker", 10, TimeSpan.FromMinutes(30)); - var ids = claimable.Select(c => c.TaskId).ToHashSet(StringComparer.Ordinal); - - Assert.Equal(2, ids.Count); - Assert.Contains(second[0].TaskId, ids); // lease lapsed -> reclaimable - Assert.DoesNotContain(first[0].TaskId, ids); // lease live -> hidden - } - - /// - /// The run-scoped reads (remove, release, queued-ids) narrow on RunName server-side. Every - /// in-memory fake ignores the filter and returns everything, so this is the only place the - /// predicate is actually evaluated — and getting it wrong is silent: a filter that matches nothing - /// makes cleanup delete nothing and de-duplication see nothing, both of which look like success. - /// - [Fact] - public async Task ServerSideRunFilter_ScopesToOneRun() - { - var queue = await TryConnectAsync(); - if (queue == null) return; - - await queue.EnqueueBatchAsync("run-a", [("a1", 4), ("a2", 4)], At(0)); - await queue.EnqueueBatchAsync("run-b", [("b1", 4)], At(0)); - - Assert.Equal(["a1", "a2"], - (await queue.GetQueuedTaskIdsAsync("run-a")).OrderBy(x => x, StringComparer.Ordinal)); - - // Removing one run must not touch the other. - await queue.RemoveRunAsync("run-a"); - - Assert.Empty(await queue.GetQueuedTaskIdsAsync("run-a")); - Assert.Equal(["b1"], await queue.GetQueuedTaskIdsAsync("run-b")); - } - - /// A run name containing a quote must not break the filter or leak into it. - [Fact] - public async Task ServerSideRunFilter_HandlesAQuoteInTheRunName() - { - var queue = await TryConnectAsync(); - if (queue == null) return; - - const string Odd = "run-o'brien"; - await queue.EnqueueBatchAsync(Odd, [("t1", 4)], At(0)); - await queue.EnqueueBatchAsync("run-plain", [("t2", 4)], At(0)); - - Assert.Equal(["t1"], await queue.GetQueuedTaskIdsAsync(Odd)); - Assert.Equal(["t2"], await queue.GetQueuedTaskIdsAsync("run-plain")); - } - - [Fact] - public async Task BackendReturnsWorkHighestPriorityFirst() - { - var queue = await TryConnectAsync(); - if (queue == null) return; - - // Inserted in the least helpful order: bulk P4 work first, urgent P0 last, idle P6 early. If any - // of this came back in insertion order the priority scheme is broken. - await queue.EnqueueBatchAsync("StandardsApply", - [("std-a", 4), ("std-b", 4)], At(30)); - await queue.EnqueueAsync("StandardsApply", "std-c", 4, At(1)); - await queue.EnqueueAsync("DbCache", "cache-0", 6, At(0)); - await queue.EnqueueAsync("AuditLogIngest", "audit-0", 0, At(59)); - - // Claim one at a time so the order the backend hands work out is directly observable. - var order = new List(); - for (var i = 0; i < 5; i++) - { - var claimed = await queue.ClaimBatchAsync($"w{i}", 1, TimeSpan.FromMinutes(20)); - order.Add(Assert.Single(claimed).TaskId); - } - - // P0 first despite being queued last; P6 last despite being queued early. - Assert.Equal("audit-0", order[0]); - Assert.Equal("cache-0", order[4]); - - // The three P4 tasks fill the middle, ahead of P6 and behind P0. Schema v2 keys per (run, task), - // so ORDER within a priority is no longer oldest-first — only that they all rank between the two. - var middle = order.GetRange(1, 3); - Assert.Contains("std-a", middle); - Assert.Contains("std-b", middle); - Assert.Contains("std-c", middle); - } - - [Fact] - public async Task TwoClaimersRacingTheSameRowsCannotBothWin() - { - var queue = await TryConnectAsync(); - if (queue == null) return; - - await queue.EnqueueBatchAsync("run", - Enumerable.Range(0, 6).Select(i => ($"task-{i:D2}", 4)).ToList(), At(0)); - - // Both read the same rows before either writes, so both hold the same ETags. - var a = queue.ClaimBatchAsync("worker-a", 6, TimeSpan.FromMinutes(20)); - var b = queue.ClaimBatchAsync("worker-b", 6, TimeSpan.FromMinutes(20)); - var results = await Task.WhenAll(a, b); - - var winners = results.Where(r => r.Count > 0).ToList(); - Assert.Single(winners); - Assert.Equal(6, winners[0].Count); - - // And the rows really are owned now, not merely reported as claimed. - Assert.Empty(await queue.ClaimBatchAsync("worker-c", 6, TimeSpan.FromMinutes(20))); - } - - [Fact] - public async Task ExpiredLeaseIsReclaimableAndALiveOneIsNot() - { - var queue = await TryConnectAsync(); - if (queue == null) return; - - await queue.EnqueueAsync("run", "task-0", 4, At(0)); - - // Long lease for the "still held" assertions. They only need the lease to outlive a storage - // round-trip, and a short one makes that a race: this test used to claim for 2 seconds and - // then assert liveness through two more round-trips, so a loaded backend expired the lease - // before the assertion ran and the test failed claiming exclusivity was broken. Nothing here - // is waiting out this lease, so minutes cost nothing. - var held = await queue.ClaimBatchAsync("worker-a", 1, TimeSpan.FromMinutes(5)); - Assert.Single(held); - - // Live lease: nobody else may take it. - Assert.Empty(await queue.ClaimBatchAsync("worker-b", 1, TimeSpan.FromMinutes(20))); - - // Renewal keeps ownership rather than releasing it — the long-running-task path. - Assert.True(await queue.RenewAsync(held, "worker-a", TimeSpan.FromMinutes(5))); - Assert.Empty(await queue.ClaimBatchAsync("worker-b", 1, TimeSpan.FromMinutes(20))); - - // Renewal sets the deadline against the real clock, so renewing SHORT lets it lapse. This is - // the one deliberately time-bound step; the wait is several times the lease so a slow - // backend makes it later, not flakier. - Assert.True(await queue.RenewAsync(held, "worker-a", TimeSpan.FromSeconds(1))); - await Task.Delay(TimeSpan.FromSeconds(4)); - - // Lapsed: reclaimable, with no intervention from the holder. This is the crash-recovery path. - var reclaimed = await queue.ClaimBatchAsync("worker-b", 1, TimeSpan.FromMinutes(20)); - Assert.Equal("task-0", Assert.Single(reclaimed).TaskId); - } -} diff --git a/tests/Craft.Tests/JobQueueDispatchableTests.cs b/tests/Craft.Tests/JobQueueDispatchableTests.cs deleted file mode 100644 index 3821a71..0000000 --- a/tests/Craft.Tests/JobQueueDispatchableTests.cs +++ /dev/null @@ -1,106 +0,0 @@ -using Craft.Configuration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// The index table answers "does this run still have queued work" from a single partition, which is -/// what keeps the re-drive and finalize paths off a full-table scan. But the index can OUTLIVE the -/// queue rows it points at, and when it does it lies: it reports a task queued that no pump will ever -/// claim, so an orphan re-drive that trusts the index skips it and the run stalls indefinitely with -/// the task Pending — resume logs "Dispatched 0 tasks (N already queued)" and the orphan watchdog -/// never fires, because GetQueuedTaskIdsAsync (index-only) keeps reporting the task queued while the -/// queue table has no runnable row for it. Clearing the stale index row is the only thing that lets -/// the watchdog see it as orphaned again. -/// -/// is the fix: it verifies each candidate -/// against the queue TABLE, returning only tasks the pump can actually still claim. These tests pin -/// the three divergence shapes it has to catch. -/// -public class JobQueueDispatchableTests -{ - private static readonly TimeSpan Lease = TimeSpan.FromMinutes(20); - private const string QueueTable = "OrchestratorQueue"; - - private static (JobQueueStore Queue, RunRemainingCounterTests.ConditionalStore Backing) NewQueue() - { - var backing = new RunRemainingCounterTests.ConditionalStore(); - var queue = new JobQueueStore(NullLogger.Instance, new CraftSettings(), backing); - return (queue, backing); - } - - [Fact] - public async Task NormallyQueuedTasksAreAllDispatchable() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueBatchAsync("R", [("a", 4), ("b", 4), ("c", 4)], DateTime.UtcNow); - - var dispatchable = await queue.GetDispatchableTaskIdsAsync("R", ["a", "b", "c"]); - - Assert.Equal(3, dispatchable.Count); - Assert.Contains("a", dispatchable); - Assert.Contains("b", dispatchable); - Assert.Contains("c", dispatchable); - } - - [Fact] - public async Task AnIndexRowWithNoQueueRowIsNotDispatchable_ButTheIndexStillListsIt() - { - var (queue, backing) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueBatchAsync("R", [("a", 4), ("b", 4)], DateTime.UtcNow); - - // Delete ONLY b's queue row, leaving its index row — the exact divergence a crash between the - // two deletes, or a run carried in from a pre-pump build, leaves behind. - await backing.DeleteAsync(QueueTable, JobQueueStore.Bucket(4), JobQueueStore.BuildRowKey("R", "b")); - - // The index — what the old re-drive trusted — still reports both as queued. - var indexView = await queue.GetQueuedTaskIdsAsync("R"); - Assert.Contains("a", indexView); - Assert.Contains("b", indexView); - - // The queue-verified view sees b for the ghost it is. - var dispatchable = await queue.GetDispatchableTaskIdsAsync("R", ["a", "b"]); - Assert.Contains("a", dispatchable); - Assert.DoesNotContain("b", dispatchable); - } - - [Fact] - public async Task AQueueRowOwnedWithNoLeaseIsNotDispatchable() - { - var (queue, backing) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueBatchAsync("R", [("a", 4)], DateTime.UtcNow); - - // Owned with no LeaseUntil: neither "Owner eq ''" nor "LeaseUntil lt now", so the claim filter - // can never match it and the pump will never dispatch it — a ghost as surely as a missing row. - var key = JobQueueStore.BuildRowKey("R", "a"); - var row = await backing.GetAsync(QueueTable, JobQueueStore.Bucket(4), key); - Assert.NotNull(row); - row!["Owner"] = "dead-instance"; - row["LeaseUntil"] = (DateTimeOffset?)null; - await backing.UpsertAsync(QueueTable, row); - - var dispatchable = await queue.GetDispatchableTaskIdsAsync("R", ["a"]); - Assert.Empty(dispatchable); - } - - [Fact] - public async Task AQueueRowUnderALeaseStaysDispatchable() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueBatchAsync("R", [("a", 4)], DateTime.UtcNow); - - // A live claim: owned under a lease. The pump (or its lapse) will run it, so the re-drive must - // NOT treat it as orphaned — doing so would reset a row a worker is actively holding and could - // run the task a second time. - var claimed = await queue.ClaimBatchAsync("worker-a", 8, Lease); - Assert.Single(claimed); - - var dispatchable = await queue.GetDispatchableTaskIdsAsync("R", ["a"]); - Assert.Contains("a", dispatchable); - } -} diff --git a/tests/Craft.Tests/JobQueueIndexBackfillTests.cs b/tests/Craft.Tests/JobQueueIndexBackfillTests.cs deleted file mode 100644 index 3041822..0000000 --- a/tests/Craft.Tests/JobQueueIndexBackfillTests.cs +++ /dev/null @@ -1,201 +0,0 @@ -using Craft.Configuration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// The one-time build of the run index, for a queue table that predates it. -/// -/// This is the only part of the index change that touches a queue nobody wrote through the new enqueue -/// path, and it runs against the worst queue in the estate: the instance that motivated the change was -/// carrying ~743,000 rows across 12,028 runs. Two properties matter and neither is observable from the -/// happy path. -/// -/// 1. It actually indexes what is already there. Without this every pre-existing run reads as having -/// no queued tasks, and the orphan re-drive re-queues a backlog that already has rows — the -/// duplicate-execution failure the queue exists to prevent. -/// 2. It runs exactly once, ever. It is a full pass over the queue table, awaited before the service -/// serves traffic, so a version that re-ran on every start would add a multi-minute stall to every -/// restart of the largest instances. -/// -public class JobQueueIndexBackfillTests -{ - private static DateTime At(int minute) => new(2026, 8, 9, 2, minute, 0, DateTimeKind.Utc); - - private static (JobQueueStore Queue, RunRemainingCounterTests.ConditionalStore Store, string QueueTable) NewQueue() - { - var settings = new CraftSettings(); - var store = new RunRemainingCounterTests.ConditionalStore(); - var queue = new JobQueueStore(NullLogger.Instance, settings, store); - return (queue, store, $"{settings.Orchestrator.TablePrefix}Queue"); - } - - /// - /// A queue row exactly as the v1 code wrote it: the legacy time-prefixed key - /// ({ticks:D19}-{run}-{task}), no QueuedUtc property, no index row anywhere. This is what - /// the v2 migration must re-key. - /// - private static Task SeedLegacyRowAsync(RunRemainingCounterTests.ConditionalStore store, string queueTable, - string runName, string taskId, int priority, DateTime queuedUtc) - { - var legacyKey = $"{queuedUtc.Ticks.ToString("D19", System.Globalization.CultureInfo.InvariantCulture)}-{runName}-{taskId}"; - return store.UpsertAsync(queueTable, - new StoreRow(JobQueueStore.Bucket(priority), legacyKey) - { - Properties = - { - ["RunName"] = runName, - ["TaskId"] = taskId, - ["Priority"] = priority, - ["Owner"] = "", - ["LeaseUntil"] = (DateTimeOffset?)null, - } - }); - } - - [Fact] - public async Task BuildsTheIndexForAQueueThatPredatesIt() - { - var (queue, store, queueTable) = NewQueue(); - - await SeedLegacyRowAsync(store, queueTable, "run-a", "task-0", 4, At(0)); - await SeedLegacyRowAsync(store, queueTable, "run-a", "task-1", 4, At(0)); - await SeedLegacyRowAsync(store, queueTable, "run-b", "task-9", 0, At(1)); - - await queue.InitializeAsync(); - - Assert.Equal(["task-0", "task-1"], - (await queue.GetQueuedTaskIdsAsync("run-a")).OrderBy(x => x, StringComparer.Ordinal)); - Assert.Equal(["task-9"], await queue.GetQueuedTaskIdsAsync("run-b")); - } - - /// - /// v2 re-keys a legacy time-prefixed row to the deterministic {run}|{task} scheme, carrying its - /// enqueue time into the QueuedUtc property, deleting the old row, and leaving the task claimable once. - /// - [Fact] - public async Task MigratesLegacyRowsToTheDeterministicKeyScheme() - { - var (queue, store, queueTable) = NewQueue(); - await SeedLegacyRowAsync(store, queueTable, "run-a", "task-0", 4, At(3)); - - await queue.InitializeAsync(); - - var rows = new List(); - await foreach (var row in store.QueryTableAsync(queueTable)) rows.Add(row); - - // The legacy row is gone; one row remains, keyed deterministically and carrying QueuedUtc. - var only = Assert.Single(rows); - Assert.Equal(JobQueueStore.BuildRowKey("run-a", "task-0"), only.RowKey); - Assert.Equal(new DateTimeOffset(At(3), TimeSpan.Zero), only.GetDateTimeOffset("QueuedUtc")); - - // And it is still claimable, exactly once. - Assert.Equal("task-0", - Assert.Single(await queue.ClaimBatchAsync("w", 8, TimeSpan.FromMinutes(20))).TaskId); - } - - [Fact] - public async Task RunsOnceAndNeverAgain() - { - var (queue, store, queueTable) = NewQueue(); - - await SeedLegacyRowAsync(store, queueTable, "run-a", "task-0", 4, At(0)); - await queue.InitializeAsync(); - Assert.Equal(["task-0"], await queue.GetQueuedTaskIdsAsync("run-a")); - - // A second un-indexed row, then a fresh store instance over the SAME tables so the in-process - // _initialized latch cannot be what short-circuits it. Only the persisted marker can. - await SeedLegacyRowAsync(store, queueTable, "run-a", "task-1", 4, At(2)); - - var second = new JobQueueStore(NullLogger.Instance, new CraftSettings(), store); - await second.InitializeAsync(); - - // Deliberately asserting the un-indexed row is NOT picked up: that is what proves the backfill - // short-circuited rather than silently re-running. A rebuild would report both tasks. - Assert.Equal(["task-0"], await second.GetQueuedTaskIdsAsync("run-a")); - } - - [Fact] - public async Task IndexedRowsSurviveARemoveRun() - { - var (queue, store, queueTable) = NewQueue(); - - await SeedLegacyRowAsync(store, queueTable, "run-a", "task-0", 4, At(0)); - await SeedLegacyRowAsync(store, queueTable, "run-b", "task-9", 4, At(0)); - await queue.InitializeAsync(); - - await queue.RemoveRunAsync("run-a"); - - Assert.Empty(await queue.GetQueuedTaskIdsAsync("run-a")); - Assert.Equal(["task-9"], await queue.GetQueuedTaskIdsAsync("run-b")); - - // The backfilled queue rows must be gone too, not just their index entries — otherwise a - // cancelled run's work stays claimable. - var left = new List(); - await foreach (var row in store.QueryTableAsync(queueTable)) - { - if (row.GetString("RunName") == "run-a") left.Add(row.RowKey); - } - Assert.Empty(left); - } - - /// - /// Run names carry a user-supplied scheduled-task name, and Azure Tables rejects '/', '\', '#', '?' - /// and control characters in a key. A real one in production is - /// "UserTaskOrchestrator_AllTenants: Alert on Huntress Rogue Apps detected-{guid}"; nothing stops the - /// next one containing a slash. Unescaped, the index write throws and the run has no index at all. - /// - [Theory] - [InlineData("run/with/slashes")] - [InlineData("run?with=query")] - [InlineData(@"run\with\backslash")] - [InlineData("run#with-hash")] - [InlineData("run%already-escaped")] - [InlineData("UserTaskOrchestrator_AllTenants: Alert on 100% CPU? detected-abc123")] - public async Task RunNamesWithKeyIllegalCharactersRoundTrip(string runName) - { - var (queue, _, _) = NewQueue(); - await queue.InitializeAsync(); - - await queue.EnqueueBatchAsync(runName, [("task-0", 4)], At(0)); - await queue.EnqueueBatchAsync("run-plain", [("task-9", 4)], At(0)); - - var partition = JobQueueStore.IndexPartition(runName); - Assert.DoesNotContain(partition, c => c is '/' or '\\' or '#' or '?' || char.IsControl(c)); - - Assert.Equal(["task-0"], await queue.GetQueuedTaskIdsAsync(runName)); - Assert.Equal(["task-9"], await queue.GetQueuedTaskIdsAsync("run-plain")); - } - - /// Distinct run names must not collide once escaped, or one run's cleanup drops another's. - [Fact] - public async Task EscapingDoesNotCollideAcrossRunNames() - { - // "a/b" escapes to "a%2Fb"; a run literally named "a%2Fb" must land somewhere else, which is why - // '%' is itself escaped. - Assert.NotEqual(JobQueueStore.IndexPartition("a/b"), JobQueueStore.IndexPartition("a%2Fb")); - - var (queue, _, _) = NewQueue(); - await queue.InitializeAsync(); - - await queue.EnqueueBatchAsync("a/b", [("slash", 4)], At(0)); - await queue.EnqueueBatchAsync("a%2Fb", [("literal", 4)], At(0)); - - Assert.Equal(["slash"], await queue.GetQueuedTaskIdsAsync("a/b")); - Assert.Equal(["literal"], await queue.GetQueuedTaskIdsAsync("a%2Fb")); - } - - [Fact] - public void IndexRowKeySplitsAtTheBucketBoundary_EvenWithAPipeInTheRowKey() - { - // The queue row key embeds the run name, so a '|' can appear inside it. Splitting on the first - // '|' rather than at the fixed bucket width would address the wrong queue row. - const string QueueRowKey = "0000000638000000000000000-run|odd-task-0"; - var split = JobQueueStore.SplitIndexRowKey(JobQueueStore.IndexRowKey("P04", QueueRowKey)); - - Assert.NotNull(split); - Assert.Equal("P04", split!.Value.Bucket); - Assert.Equal(QueueRowKey, split.Value.QueueRowKey); - } -} diff --git a/tests/Craft.Tests/JobQueuePumpBackoffTests.cs b/tests/Craft.Tests/JobQueuePumpBackoffTests.cs deleted file mode 100644 index 2375fa8..0000000 --- a/tests/Craft.Tests/JobQueuePumpBackoffTests.cs +++ /dev/null @@ -1,276 +0,0 @@ -using Craft.Configuration; -using Craft.Orchestration; -using Craft.PowerShellHost; -using Craft.Storage; -using Microsoft.Extensions.Configuration; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -[CollectionDefinition(Name, DisableParallelization = true)] -public class PumpTiming -{ - public const string Name = "job-queue-pump-timing"; -} - -/// -/// The pump polls storage for work, so an idle instance was scanning the queue table once a second -/// forever. Backing off fixes that, but the interval is also a hard throughput ceiling — a refill hands -/// over at most batchSize jobs per tick, so a flat 10s poll caps the whole system at batchSize/10 tasks -/// per second. On a 7,336-task fan-out that is hours of waiting no matter how fast the tasks are. -/// -/// So the backoff has to key off the right signal. "Claimed nothing this tick" is NOT idleness: the -/// pump also claims nothing while its buffer is above the low-water mark, which is precisely when a -/// busy run is about to need its next batch. Idle means claimed nothing AND holding nothing. -/// -/// These tests pin both directions — that a quiet pump slows down, and that a working one does not. -/// They count scans in sub-second wall-clock windows, so they run alone: sharing a small CI runner with -/// parallel collections starved the 100ms ticks enough to read as a backoff. -/// -[Collection(PumpTiming.Name)] -public class JobQueuePumpBackoffTests -{ - private static (JobQueuePump Pump, JobQueueStore Queue, JobManager Jobs) NewPump( - int pollMs, int idlePollMs, int backlog = 0, int batch = 4, int lowWater = 2, int? poolSize = null) - { - var settings = new CraftSettings(); - // Separate knobs on purpose: with batch == pool every claimed job starts immediately, the buffer - // empties into the running state and the pump keeps claiming. A pool SMALLER than the batch is - // what leaves work sitting in the buffer — the state where the pump holds work but claims none. - settings.Worker.BgPoolSize = poolSize ?? batch; - - var config = new ConfigurationBuilder().AddInMemoryCollection(new Dictionary - { - ["JobQueueBatchSize"] = batch.ToString(System.Globalization.CultureInfo.InvariantCulture), - ["JobQueueLowWaterMark"] = lowWater.ToString(System.Globalization.CultureInfo.InvariantCulture), - ["JobQueuePollIntervalMs"] = pollMs.ToString(System.Globalization.CultureInfo.InvariantCulture), - ["JobQueueIdlePollIntervalMs"] = idlePollMs.ToString(System.Globalization.CultureInfo.InvariantCulture), - }).Build(); - - var backing = new CountingStore(); - var queue = new JobQueueStore(NullLogger.Instance, settings, backing); - queue.InitializeAsync().GetAwaiter().GetResult(); - - if (backlog > 0) - { - queue.EnqueueBatchAsync("run", - Enumerable.Range(0, backlog).Select(i => ($"task-{i:D5}", 4)).ToList(), - new DateTime(2026, 8, 9, 2, 0, 0, DateTimeKind.Utc)).GetAwaiter().GetResult(); - } - - backing.Scans = 0; // ignore the enqueue traffic - - var repo = new ScriptRepository(NullLogger.Instance, settings); - var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); - var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); - var jobs = new JobManager(NullLogger.Instance, settings, limiter); - - var pump = new JobQueuePump(NullLogger.Instance, queue, jobs, config, settings); - return (pump, queue, jobs); - } - - private static async Task PumpFor(JobQueuePump pump, int ms) - { - await pump.StartAsync(CancellationToken.None); - await Task.Delay(ms); - await Task.WhenAny(pump.StopAsync(CancellationToken.None), Task.Delay(3000)); - } - - /// Counts table scans, which is exactly what the backoff is meant to reduce. - private sealed class CountingStore : ICraftTableStore - { - private readonly Dictionary> _tables = new(); - private readonly object _sync = new(); - - public int Scans; - - public Task PingAsync(CancellationToken ct = default) => Task.CompletedTask; - - public Task EnsureTableAsync(string table, CancellationToken ct = default) - { - lock (_sync) { if (!_tables.ContainsKey(table)) _tables[table] = new(); } - return Task.CompletedTask; - } - - public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) - { - lock (_sync) - { - if (!_tables.ContainsKey(table)) _tables[table] = new(); - _tables[table][(row.PartitionKey, row.RowKey)] = row; - } - return Task.CompletedTask; - } - - public Task UpsertBatchAsync(string table, string pk, IReadOnlyList rows, CancellationToken ct = default) - { - lock (_sync) - { - if (!_tables.ContainsKey(table)) _tables[table] = new(); - foreach (var r in rows) _tables[table][(r.PartitionKey, r.RowKey)] = r; - } - return Task.CompletedTask; - } - - public Task TryReplaceBatchAsync(string table, string pk, IReadOnlyList rows, - CancellationToken ct = default) - { - lock (_sync) - { - if (!_tables.ContainsKey(table)) _tables[table] = new(); - foreach (var r in rows) _tables[table][(r.PartitionKey, r.RowKey)] = r; - } - return Task.FromResult(true); - } - - public Task GetAsync(string table, string pk, string rk, CancellationToken ct = default) - { - lock (_sync) - return Task.FromResult(_tables.TryGetValue(table, out var t) - && t.TryGetValue((pk, rk), out var r) ? r : null); - } - - public async IAsyncEnumerable QueryPartitionAsync(string table, string pk, - [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) - { - List snap; - lock (_sync) - snap = _tables.TryGetValue(table, out var t) - ? t.Where(k => k.Key.Item1 == pk).Select(k => k.Value).ToList() - : new List(); - foreach (var r in snap) { yield return r; await Task.Yield(); } - } - - public async IAsyncEnumerable QueryTableAsync(string table, - [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) - { - Interlocked.Increment(ref Scans); - List snap; - lock (_sync) snap = _tables.TryGetValue(table, out var t) ? t.Values.ToList() : new List(); - foreach (var r in snap) { yield return r; await Task.Yield(); } - } - - public Task DeleteAsync(string table, string pk, string rk, CancellationToken ct = default) - { - lock (_sync) { if (_tables.TryGetValue(table, out var t)) t.Remove((pk, rk)); } - return Task.CompletedTask; - } - - public Task DeletePartitionAsync(string table, string pk, CancellationToken ct = default) - { - lock (_sync) - { - if (!_tables.TryGetValue(table, out var t)) return Task.CompletedTask; - foreach (var k in t.Keys.Where(k => k.Item1 == pk).ToList()) t.Remove(k); - } - return Task.CompletedTask; - } - } - - [Fact] - public async Task AnIdlePump_BacksOff_InsteadOfScanningEveryTick() - { - // Empty queue: every tick claims nothing and holds nothing. - var (pump, queue, _) = NewPump(pollMs: 100, idlePollMs: 2000); - var backing = (CountingStore)GetBacking(queue); - - await PumpFor(pump, 900); - - // At a flat 100ms this window is ~9 scans. Doubling (100/200/400/800/1600...) allows about 4. - Assert.True(backing.Scans <= 5, - $"idle pump scanned storage {backing.Scans} times in 900ms — it is not backing off"); - Assert.True(backing.Scans >= 1, "idle pump never polled at all"); - } - - /// - /// The regression a naive backoff would cause, measured where it actually shows: throughput while a - /// consumer is draining the buffer. - /// - /// A refill hands over at most batchSize jobs, and only happens once per tick, so the poll interval - /// is a hard ceiling of batchSize/interval. If the pump backed off because a tick claimed nothing — - /// which it legitimately does whenever the buffer is above the low-water mark — a busy run would be - /// throttled to batchSize per IDLE interval instead. Here that is the difference between ~9 refills - /// in the window and ~4. - /// - /// Scans rather than polls is the right unit: RefillAsync returns without touching storage while the - /// buffer is full, so a scan happens exactly when the pump needed more work. - /// - [Fact] - public async Task APumpWhoseBufferIsDraining_RefillsAtTheBaseInterval() - { - var (pump, queue, jobs) = NewPump(pollMs: 100, idlePollMs: 2000, backlog: 200); - var backing = (CountingStore)GetBacking(queue); - - // A consumer, so the buffer actually draws down and refills are needed — without one the pump - // fills once and correctly never scans again. - // The window opens once the consumer has run a job, so its startup is not billed to the pump. - var draining = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); - jobs.SetWorkResolver((_, _) => - { - draining.TrySetResult(); - return Task.FromResult?>(_ => Task.CompletedTask); - }); - _ = Task.Run(() => jobs.StartAsync(CancellationToken.None)); - - await pump.StartAsync(CancellationToken.None); - await draining.Task.WaitAsync(TimeSpan.FromSeconds(10)); - var before = backing.Scans; - await Task.Delay(900); - var scans = backing.Scans - before; - await Task.WhenAny(pump.StopAsync(CancellationToken.None), Task.Delay(3000)); - await jobs.StopAsync(CancellationToken.None); - - Assert.True(scans >= 6, - $"a draining buffer was only refilled {scans} times in 900ms — the pump backed off " + - "while work was flowing, which caps throughput at batchSize per idle interval"); - } - - /// - /// The case that decides the idle SIGNAL, as opposed to the backoff curve. - /// - /// While long tasks run, the buffer stays above the low-water mark and the pump claims nothing tick - /// after tick. Treating "claimed nothing" as idle backs the loop off during exactly that stretch, and - /// the delay is then paid at the worst moment: the buffer finally drains and the refill that should - /// have taken one base interval takes an idle one. That is why idleness requires holding nothing as - /// well as claiming nothing. - /// - /// Long tasks are modelled by blocking the consumer, then released so the buffer drains. The window - /// after release is short enough that a backed-off pump cannot have polled in it at all. - /// - [Fact] - public async Task AfterALongStretchOfHoldingWork_TheNextRefillIsStillPrompt() - { - // Batch 8 into a pool of 1: one job runs, seven sit in the buffer above the low-water mark, so - // the pump claims nothing while still holding claims. That is the stretch under test. - var (pump, queue, jobs) = NewPump(pollMs: 100, idlePollMs: 3000, backlog: 200, batch: 8, poolSize: 1); - var backing = (CountingStore)GetBacking(queue); - - var release = new SemaphoreSlim(0); - jobs.SetWorkResolver((_, _) => Task.FromResult?>( - async _ => await release.WaitAsync(TimeSpan.FromSeconds(10), CancellationToken.None))); - _ = Task.Run(() => jobs.StartAsync(CancellationToken.None)); - - await pump.StartAsync(CancellationToken.None); - - // Tasks are stuck, so the buffer stays full and nothing is claimed for many ticks. - await Task.Delay(800); - var scansWhileHolding = backing.Scans; - - // Let everything finish; the buffer now drains and the pump must top it up at the base interval. - release.Release(1000); - await Task.Delay(300); - var scansAfterDrain = backing.Scans - scansWhileHolding; - - await Task.WhenAny(pump.StopAsync(CancellationToken.None), Task.Delay(3000)); - await jobs.StopAsync(CancellationToken.None); - - Assert.True(scansAfterDrain >= 2, - $"only {scansAfterDrain} refill(s) in the 300ms after the buffer drained — the pump had backed " + - "off during the stretch where it held work but claimed nothing, so the next batch was late"); - } - - private static object GetBacking(JobQueueStore queue) => - typeof(JobQueueStore).GetField("_store", - System.Reflection.BindingFlags.NonPublic | System.Reflection.BindingFlags.Instance)! - .GetValue(queue)!; -} diff --git a/tests/Craft.Tests/JobQueuePumpTests.cs b/tests/Craft.Tests/JobQueuePumpTests.cs deleted file mode 100644 index 60f5cf5..0000000 --- a/tests/Craft.Tests/JobQueuePumpTests.cs +++ /dev/null @@ -1,224 +0,0 @@ -using Craft.Configuration; -using Craft.Orchestration; -using Craft.PowerShellHost; -using Craft.Storage; -using Microsoft.Extensions.Configuration; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// The pump that keeps the in-memory queue small by feeding it from storage a batch at a time. -/// -/// The property under test is the one the whole exercise is for: however deep the backlog, this process -/// holds a worker-pool-sized buffer and no more. A run of 7,336 tasks is 7,336 rows in storage and a -/// handful of objects here. -/// -/// It is a separate pump rather than a change to the dispatch loop on purpose — that loop owns the -/// limiter slot lifecycle whose invariants were written to close a leak that wedged production for 28 -/// hours, and feeding it through the enqueue path it already has leaves all of that untouched. -/// -public class JobQueuePumpTests -{ - private static (JobQueuePump Pump, JobQueueStore Queue, JobManager Jobs) NewPump( - int batch = 4, int lowWater = 2, int backlog = 0) - { - var settings = new CraftSettings(); - settings.Worker.BgPoolSize = batch; - - var config = new ConfigurationBuilder().AddInMemoryCollection(new Dictionary - { - ["JobQueueBatchSize"] = batch.ToString(System.Globalization.CultureInfo.InvariantCulture), - ["JobQueueLowWaterMark"] = lowWater.ToString(System.Globalization.CultureInfo.InvariantCulture), - ["JobQueuePollIntervalMs"] = "100", - }).Build(); - - var backing = new RunRemainingCounterTests.ConditionalStore(); - var queue = new JobQueueStore(NullLogger.Instance, settings, backing); - queue.InitializeAsync().GetAwaiter().GetResult(); - - if (backlog > 0) - { - queue.EnqueueBatchAsync("StandardsApply", - Enumerable.Range(0, backlog).Select(i => ($"task-{i:D5}", 4)).ToList(), - new DateTime(2026, 8, 9, 2, 0, 0, DateTimeKind.Utc)).GetAwaiter().GetResult(); - } - - var repo = new ScriptRepository(NullLogger.Instance, settings); - var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); - var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); - var jobs = new JobManager(NullLogger.Instance, settings, limiter); - - var pump = new JobQueuePump(NullLogger.Instance, queue, jobs, config, settings); - return (pump, queue, jobs); - } - - /// Run the pump without a dispatch loop consuming, so the buffer state is observable. - private static async Task PumpFor(JobQueuePump pump, int ms) - { - await pump.StartAsync(CancellationToken.None); - await Task.Delay(ms); - await Task.WhenAny(pump.StopAsync(CancellationToken.None), Task.Delay(3000)); - } - - /// - /// THE GUARANTEE. A backlog far larger than the pool must not end up in memory. Before this, all - /// 7,336 descriptors of a StandardsApply run sat in the process; now the table holds them. - /// - [Fact] - public async Task DeepBacklogNeverLandsInMemoryAllAtOnce() - { - var (pump, _, jobs) = NewPump(batch: 4, lowWater: 2, backlog: 500); - - await PumpFor(pump, 500); - - // Nothing is consuming, so the buffer sits at whatever one refill put there — never the backlog. - Assert.InRange(jobs.QueuedCount, 1, 8); - } - - [Fact] - public async Task RefillsOnlyOnceTheBufferHasDrawnDown() - { - var (pump, _, jobs) = NewPump(batch: 4, lowWater: 2, backlog: 100); - - await PumpFor(pump, 400); - var afterFirst = jobs.QueuedCount; - - // Above the low-water mark, so repeated cycles must not keep claiming. - Assert.InRange(afterFirst, 1, 8); - - await PumpFor(pump, 400); - Assert.InRange(jobs.QueuedCount, 1, 8); - } - - [Fact] - public async Task ClaimsNothingWhenTheQueueIsEmpty() - { - var (pump, _, jobs) = NewPump(backlog: 0); - - await PumpFor(pump, 300); - - Assert.Equal(0, jobs.QueuedCount); - } - - /// - /// Claimed rows stay in storage until the work is done. Deleting on claim would take the task with it - /// if this instance died holding the batch — the lease, not deletion, is what stops a double run. - /// - [Fact] - public async Task ClaimedRowsSurviveUntilTheWorkFinishes() - { - var (pump, queue, jobs) = NewPump(batch: 3, lowWater: 2, backlog: 3); - - await PumpFor(pump, 300); - Assert.True(jobs.QueuedCount > 0, "the pump should have claimed a batch"); - - // Still owned by us and still present, so nobody else can take them... - Assert.Empty(await queue.ClaimBatchAsync("someone-else", 3, TimeSpan.FromMinutes(20))); - - // ...and once the leases lapse they are reclaimable, which is the crash-recovery path. - var reclaimed = await queue.ClaimBatchAsync("someone-else", 3, TimeSpan.FromMinutes(20)); - Assert.Empty(reclaimed); - } - - [Fact] - public async Task SurvivesAStoreThatThrows() - { - var settings = new CraftSettings(); - settings.Worker.BgPoolSize = 4; - var config = new ConfigurationBuilder().AddInMemoryCollection(new Dictionary - { - ["JobQueuePollIntervalMs"] = "100", - }).Build(); - - var repo = new ScriptRepository(NullLogger.Instance, settings); - var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); - var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); - var jobs = new JobManager(NullLogger.Instance, settings, limiter); - - // A store whose every read throws — a pump that dies here would look exactly like an empty queue. - var queue = new JobQueueStore(NullLogger.Instance, settings, new ThrowingStore()); - var pump = new JobQueuePump(NullLogger.Instance, queue, jobs, config, settings); - - var ex = await Record.ExceptionAsync(() => PumpFor(pump, 350)); - - Assert.Null(ex); - } - - /// - /// The wake signal: an enqueue must start the pump now, not on its next poll tick. With a - /// deliberately long poll interval, a task queued after the pump has gone quiet is still claimed - /// promptly — which can only happen if the enqueue woke it. Without the signal this waits the full - /// poll interval, which on a cold system had backed off toward its idle ceiling. - /// - [Fact] - public async Task AnEnqueueWakesThePumpBeforeTheNextPollTick() - { - var settings = new CraftSettings(); - settings.Worker.BgPoolSize = 4; - var config = new ConfigurationBuilder().AddInMemoryCollection(new Dictionary - { - ["JobQueueBatchSize"] = "4", - ["JobQueueLowWaterMark"] = "2", - // Long enough that a poll-driven claim would miss the assertion window by an order of magnitude. - ["JobQueuePollIntervalMs"] = "5000", - ["JobQueueIdlePollIntervalMs"] = "5000", - }).Build(); - - var backing = new RunRemainingCounterTests.ConditionalStore(); - var queue = new JobQueueStore(NullLogger.Instance, settings, backing); - await queue.InitializeAsync(); - - var repo = new ScriptRepository(NullLogger.Instance, settings); - var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); - var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); - var jobs = new JobManager(NullLogger.Instance, settings, limiter); - var pump = new JobQueuePump(NullLogger.Instance, queue, jobs, config, settings); - - await pump.StartAsync(CancellationToken.None); - try - { - // Let the first (empty) cycle run and the pump settle into its long wait. - await Task.Delay(200); - Assert.Equal(0, jobs.QueuedCount); - - await queue.EnqueueBatchAsync("run", - new[] { ("task-1", 4) }, - new DateTime(2026, 8, 9, 2, 0, 0, DateTimeKind.Utc)); - - // Well under the 5s poll: only the wake can explain a claim this fast. - var claimed = await WaitUntilAsync(() => jobs.QueuedCount > 0, TimeSpan.FromMilliseconds(1500)); - Assert.True(claimed, - "the pump did not claim the enqueued task within 1.5s despite a 5s poll — the wake signal did not fire"); - } - finally - { - await Task.WhenAny(pump.StopAsync(CancellationToken.None), Task.Delay(3000)); - } - } - - private static async Task WaitUntilAsync(Func condition, TimeSpan timeout) - { - var sw = System.Diagnostics.Stopwatch.StartNew(); - while (sw.Elapsed < timeout) - { - if (condition()) return true; - await Task.Delay(20); - } - return condition(); - } - - private sealed class ThrowingStore : ICraftTableStore - { - public Task PingAsync(CancellationToken ct = default) => Task.CompletedTask; - public Task EnsureTableAsync(string table, CancellationToken ct = default) => Task.CompletedTask; - public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) => throw new InvalidOperationException("store down"); - public Task UpsertBatchAsync(string table, string pk, IReadOnlyList rows, CancellationToken ct = default) => throw new InvalidOperationException("store down"); - public Task TryReplaceBatchAsync(string table, string pk, IReadOnlyList rows, CancellationToken ct = default) => throw new InvalidOperationException("store down"); - public Task GetAsync(string table, string pk, string rk, CancellationToken ct = default) => throw new InvalidOperationException("store down"); - public IAsyncEnumerable QueryPartitionAsync(string table, string pk, CancellationToken ct = default) => throw new InvalidOperationException("store down"); - public IAsyncEnumerable QueryTableAsync(string table, CancellationToken ct = default) => throw new InvalidOperationException("store down"); - public Task DeleteAsync(string table, string pk, string rk, CancellationToken ct = default) => throw new InvalidOperationException("store down"); - public Task DeletePartitionAsync(string table, string pk, CancellationToken ct = default) => throw new InvalidOperationException("store down"); - } -} diff --git a/tests/Craft.Tests/JobQueueRetentionTests.cs b/tests/Craft.Tests/JobQueueRetentionTests.cs deleted file mode 100644 index 678919c..0000000 --- a/tests/Craft.Tests/JobQueueRetentionTests.cs +++ /dev/null @@ -1,280 +0,0 @@ -using Craft.Configuration; -using Craft.Orchestration; -using Craft.PowerShellHost; -using Microsoft.Extensions.Configuration; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// Heap-delta measurement cannot share a process with concurrently-allocating tests — xUnit runs -/// collections in parallel, and another collection's allocations land in the same -/// GC.GetTotalMemory window. Observed swinging the same measurement from 725 to 1181 B/task. -/// This collection runs alone. -/// -[CollectionDefinition(RetentionMeasurement.Name, DisableParallelization = true)] -public class RetentionMeasurement -{ - public const string Name = "retention-measurement"; -} - -/// -/// Measures what the JobManager queue actually retains per queued job. -/// -/// The claim under test is that queue entries pin whole graphs and are -/// therefore a primary driver of heap growth under a deep backlog (production peaked at 783 queued with -/// 3.7-hour waits). Retention is measured, not assumed: each case builds the graph, settles the GC, and -/// diffs GC.GetTotalMemory(forceFullCollection: true). -/// -/// The measurement deliberately holds ONE run graph alive across every case, mirroring -/// OrchestratorService._activeRuns — which pins the run from DispatchPendingTasks -/// (TryAdd) until FinalizeRunAsync (TryRemove), i.e. for the entire time its tasks sit queued. -/// So the reported delta is the queue's MARGINAL cost, which is the only part a descriptor rewrite -/// can actually reclaim. -/// -[Collection(RetentionMeasurement.Name)] -public class JobQueueRetentionTests -{ - private const int Jobs = 5000; - - /// Bytes below which a per-job cost is not worth a redesign on a 2398MB heap cap. - private const int NoiseFloorBytesPerJob = 32; - - private static JobManager NewJobManager(int bgPoolSize = 8) - { - var settings = new CraftSettings(); - settings.Worker.BgPoolSize = bgPoolSize; - var config = new ConfigurationBuilder().AddInMemoryCollection([]).Build(); - var repo = new ScriptRepository(NullLogger.Instance, settings); - var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); - var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); - return new JobManager(NullLogger.Instance, settings, limiter); - } - - /// A task payload shaped like a real CIPP fan-out item (tenant + collection + queue refs). - private static OrchestratorTaskItem NewTask(int i) => new() - { - Id = $"Graph_tenant{i:D4}.onmicrosoft.com", - Status = "Pending", - Parameters = new Dictionary - { - ["TenantFilter"] = $"tenant{i:D4}.onmicrosoft.com", - ["CollectionType"] = "Graph", - ["Name"] = $"CIPPDBCacheRun-Graph_tenant{i:D4}.onmicrosoft.com", - ["FunctionName"] = "Invoke-CIPPDBCacheTask", - ["QueueName"] = "cippdbcache", - ["BatchNumber"] = 1L, - }, - }; - - private static OrchestratorRun NewRun(int taskCount) - { - var run = new OrchestratorRun - { - Name = "CIPPDBCacheRun", - Status = "Running", - Priority = 5, - StartedUtc = DateTime.UtcNow, - TaskScriptName = "Invoke-CIPPDBCacheTask", - }; - for (var i = 0; i < taskCount; i++) run.Tasks.Add(NewTask(i)); - return run; - } - - private static void Settle() - { - for (var i = 0; i < 3; i++) - { - GC.Collect(2, GCCollectionMode.Forced, blocking: true, compacting: true); - GC.WaitForPendingFinalizers(); - } - } - - /// - /// Heap bytes retained by whatever returns, over and above the state - /// already alive when it is called. - /// - /// Best-of-N, taking the MINIMUM: measurement error here is one-sided. Anything else allocating - /// during the window — the test host, finalizers, JIT on the first pass — can only inflate the - /// delta, never shrink it below the true retained size. So the smallest observation is the closest - /// to truth, and averaging would bake the noise in. - /// - private static long RetainedBytes(Func build, int trials = 3) - { - var best = long.MaxValue; - - for (var i = 0; i < trials; i++) - { - // Drop the previous trial's graph and settle BEFORE sampling the floor. Without this the - // prior graph is still rooted at `before` and reclaimed inside the window, which drives the - // delta negative — that is how this reported "0 bytes" for a 3.6MB run graph. - Held = null; - Settle(); - - var before = GC.GetTotalMemory(forceFullCollection: true); - Held = build(); - Settle(); - var after = GC.GetTotalMemory(forceFullCollection: true); - - var delta = after - before; - if (delta > 0) best = Math.Min(best, delta); - } - - Held = null; - Assert.NotEqual(long.MaxValue, best); // every trial was non-positive ⇒ the measurement is broken - return best; - } - - /// - /// Roots the graph under measurement across the sampling window. A static field rather than a local - /// so the JIT cannot treat it as dead before GC.GetTotalMemory runs, and so the previous - /// trial's graph can be explicitly released. - /// - private static object? Held; - - /// - /// TODAY: every queued job carries a closure capturing (run, task, taskPath, this) — the shape - /// built by OrchestratorService.DispatchSingleTask. - /// - [Fact] - public void ClosureQueue_MarginalRetentionPerJob_IsMeasured() - { - var run = NewRun(Jobs); // held alive throughout, exactly as _activeRuns does - const string taskPath = "/app/API/Modules/CIPP/Invoke-CIPPDBCacheTask.ps1"; - var sink = new object(); // stands in for the captured `this` (OrchestratorService) - - var bytes = RetainedBytes(() => - { - var jm = NewJobManager(); - foreach (var task in run.Tasks) - { - var captured = task; - jm.Enqueue( - name: $"{run.Name}-{captured.Id}", - priority: run.Priority, - runName: run.Name, - work: _ => - { - // Same captures as the production closure: run graph, task, script path, service. - GC.KeepAlive(run); - GC.KeepAlive(captured); - GC.KeepAlive(taskPath); - GC.KeepAlive(sink); - return Task.CompletedTask; - }); - } - return jm; - }); - - GC.KeepAlive(run); - var perJob = bytes / (double)Jobs; - Assert.Equal(Jobs, NewJobManagerQueueDepthProbe(run)); // sanity: all enqueued, none dispatched - Assert.InRange(perJob, NoiseFloorBytesPerJob, 4096); - TestOutput($"closure queue: {bytes:N0} bytes for {Jobs:N0} jobs = {perJob:N0} bytes/job"); - } - - /// - /// AFTER: the queue carries only — (runName, taskId, priority) — and the - /// work is rehydrated at dispatch. No closure, no delegate, no _pendingWork entry, and no - /// Guid-suffixed second id string. - /// - [Fact] - public void DescriptorQueue_RetainsLessPerJob_ThanTheClosureQueue() - { - var run = NewRun(Jobs); - const string taskPath = "/app/API/Modules/CIPP/Invoke-CIPPDBCacheTask.ps1"; - var sink = new object(); - - var closureBytes = RetainedBytes(() => - { - var jm = NewJobManager(); - foreach (var task in run.Tasks) - { - var captured = task; - jm.Enqueue($"{run.Name}-{captured.Id}", run.Priority, _ => - { - GC.KeepAlive(run); GC.KeepAlive(captured); GC.KeepAlive(taskPath); GC.KeepAlive(sink); - return Task.CompletedTask; - }, run.Name); - } - return jm; - }); - - var descriptorBytes = RetainedBytes(() => - { - var jm = NewJobManager(); - foreach (var task in run.Tasks) - jm.Enqueue(new JobDescriptor(run.Name, task.Id, run.Priority), $"{run.Name}-{task.Id}"); - return jm; - }); - - GC.KeepAlive(run); - var closurePerJob = closureBytes / (double)Jobs; - var descriptorPerJob = descriptorBytes / (double)Jobs; - var saved = 1 - (descriptorPerJob / closurePerJob); - - TestOutput($"BEFORE (closure): {closurePerJob:N0} B/job ({closureBytes:N0} total)"); - TestOutput($"AFTER (descriptor): {descriptorPerJob:N0} B/job ({descriptorBytes:N0} total)"); - TestOutput($"reduction: {saved:P0} ({closurePerJob - descriptorPerJob:N0} B/job)"); - - Assert.True(descriptorBytes < closureBytes, - $"descriptor queue ({descriptorBytes:N0}B) must retain less than the closure queue ({closureBytes:N0}B)"); - } - - /// - /// The load-bearing measurement, and the one that falsifies "the queue is the memory problem". - /// - /// The run graph a closure captures is ALREADY pinned by _activeRuns from - /// DispatchPendingTasks to FinalizeRunAsync, so a descriptor rewrite reclaims the - /// queue's own bookkeeping and nothing else. Measured, both are the same order of magnitude - /// (~730 B), and at production's observed peak of 783 queued jobs the ENTIRE queue is well under - /// 1 MB — 0.02% of the 2398 MB DOTNET_GCHeapHardLimit. The queue cannot be what OOMs the - /// container; this test fails if that ever stops being true. - /// - [Fact] - public void QueueRetention_AtProductionPeak_IsNegligibleAgainstTheHeapCap() - { - const int ProductionPeakQueueDepth = 783; - const long HeapHardLimitBytes = 0x95E00000; // build/Dockerfile DOTNET_GCHeapHardLimit = 2398 MB - - var runBytes = RetainedBytes(() => NewRun(Jobs)); - - var run = NewRun(Jobs); - var queueBytes = RetainedBytes(() => - { - var jm = NewJobManager(); - foreach (var task in run.Tasks) - { - var captured = task; - jm.Enqueue($"{run.Name}-{captured.Id}", run.Priority, _ => - { - GC.KeepAlive(run); GC.KeepAlive(captured); - return Task.CompletedTask; - }, run.Name); - } - return jm; - }); - GC.KeepAlive(run); - - var perJob = queueBytes / (double)Jobs; - var atPeak = (long)(perJob * ProductionPeakQueueDepth); - var shareOfCap = atPeak / (double)HeapHardLimitBytes; - - TestOutput($"run graph ({Jobs:N0} tasks): {runBytes:N0} bytes ({runBytes / (double)Jobs:N0} B/task, pinned by _activeRuns)"); - TestOutput($"queue bookkeeping: {queueBytes:N0} bytes ({perJob:N0} B/job, reclaimable)"); - TestOutput($"at production peak ({ProductionPeakQueueDepth} queued): {atPeak:N0} bytes = {shareOfCap:P3} of the 2398MB cap"); - - Assert.True(shareOfCap < 0.01, - $"queue at peak is {shareOfCap:P2} of the heap cap ({atPeak:N0}B) — if this ever exceeds 1% " + - "the queue really has become a memory driver and this analysis needs redoing"); - } - - private static int NewJobManagerQueueDepthProbe(OrchestratorRun run) - { - var jm = NewJobManager(); - foreach (var t in run.Tasks) jm.Enqueue(t.Id, run.Priority, _ => Task.CompletedTask, run.Name); - return jm.QueuedCount; - } - - private static void TestOutput(string message) => Console.WriteLine($"[retention] {message}"); -} diff --git a/tests/Craft.Tests/JobQueueRunLookupTests.cs b/tests/Craft.Tests/JobQueueRunLookupTests.cs deleted file mode 100644 index b5e68bd..0000000 --- a/tests/Craft.Tests/JobQueueRunLookupTests.cs +++ /dev/null @@ -1,142 +0,0 @@ -using Craft.Configuration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// The lookup that tells a re-drive "waiting" from "lost". -/// -/// RedrivePendingTasks used to decide a Pending task was orphaned when the JobManager did not have it -/// queued or running. Under the pump that is what a BACKLOG is — the pump buffers a worker-pool-sized -/// slice and leaves the rest in storage — so the re-drive re-queued the whole un-started backlog every -/// 60 seconds. Queue RowKeys are prefixed with the enqueue timestamp, so each pass added another row -/// for the same task instead of updating the first, and every copy was independently claimable. -/// -/// Measured live before the fix, on one 124-task run: re-drives of 92, 60, 60, 52, 44, 36 tasks on -/// consecutive ticks, six queue rows for a single task exactly 60s apart, and that task executing six -/// times. -/// -public class JobQueueRunLookupTests -{ - private static JobQueueStore NewQueue() - { - var settings = new CraftSettings(); - var queue = new JobQueueStore(NullLogger.Instance, settings, - new RunRemainingCounterTests.ConditionalStore()); - queue.InitializeAsync().GetAwaiter().GetResult(); - return queue; - } - - private static DateTime At(int minute) => new(2026, 8, 9, 2, minute, 0, DateTimeKind.Utc); - - [Fact] - public async Task QueuedTaskIds_ReportsWaitingWork_SoABacklogIsNotMistakenForOrphans() - { - var queue = NewQueue(); - await queue.EnqueueBatchAsync("run-a", - [("task-0", 4), ("task-1", 4), ("task-2", 4)], At(0)); - await queue.EnqueueBatchAsync("other-run", [("task-9", 4)], At(0)); - - var ids = await queue.GetQueuedTaskIdsAsync("run-a"); - - Assert.Equal(["task-0", "task-1", "task-2"], ids.OrderBy(x => x, StringComparer.Ordinal)); - Assert.DoesNotContain("task-9", ids); - } - - [Fact] - public async Task AClaimedTaskIsStillReported_BecauseItIsRunning_NotLost() - { - // The re-drive must not re-queue work that has been handed to a worker: it is the case that - // produced duplicate rows for tasks already executing. - var queue = NewQueue(); - await queue.EnqueueBatchAsync("run-a", [("task-0", 4), ("task-1", 4)], At(0)); - - var claimed = await queue.ClaimBatchAsync("worker-a", 1, TimeSpan.FromMinutes(20)); - Assert.Single(claimed); - - var ids = await queue.GetQueuedTaskIdsAsync("run-a"); - - Assert.Contains(claimed[0].TaskId, ids); - Assert.Equal(2, ids.Count); - } - - [Fact] - public async Task ATaskWhoseRowIsGone_IsNotReported_SoARealOrphanIsStillRecoverable() - { - // The guard has to stay specific — a task whose row genuinely vanished must still be re-driven, - // which is the whole reason the re-drive exists. - var queue = NewQueue(); - await queue.EnqueueBatchAsync("run-a", [("task-0", 4), ("task-1", 4)], At(0)); - - var claimed = await queue.ClaimBatchAsync("worker-a", 2, TimeSpan.FromMinutes(20)); - await queue.RemoveAsync(claimed.Single(c => c.TaskId == "task-0")); - - var ids = await queue.GetQueuedTaskIdsAsync("run-a"); - - Assert.DoesNotContain("task-0", ids); - Assert.Contains("task-1", ids); - } - - [Fact] - public async Task ARunWithNothingQueued_ReportsNothing() - { - var queue = NewQueue(); - Assert.Empty(await queue.GetQueuedTaskIdsAsync("run-with-no-rows")); - } - - // ── Releasing the claims a crashed process was holding ──────────────────────────────────────── - - /// - /// After a crash the dead process's claims are still live as far as storage is concerned, so nothing - /// can pick those rows up until the lease lapses — up to 30 minutes by default. Since re-dispatch - /// now declines to write duplicate rows for tasks that already have one, the run simply stalls. - /// Recovery has to hand the claims back. - /// - [Fact] - public async Task ReleasingARunsClaims_MakesItsRowsClaimableAgain() - { - var queue = NewQueue(); - await queue.EnqueueBatchAsync("crashed-run", [("task-0", 4), ("task-1", 4)], At(0)); - - // A long lease, as the pump takes: without a release these are untouchable for its full duration. - var held = await queue.ClaimBatchAsync("dead-worker", 2, TimeSpan.FromMinutes(30)); - Assert.Equal(2, held.Count); - Assert.Empty(await queue.ClaimBatchAsync("new-worker", 2, TimeSpan.FromMinutes(30))); - - var released = await queue.ReleaseRunClaimsAsync("crashed-run"); - - Assert.Equal(2, released); - Assert.Equal(2, (await queue.ClaimBatchAsync("new-worker", 2, TimeSpan.FromMinutes(30))).Count); - } - - /// Releasing frees the existing rows rather than adding more — the duplicate-row bug again. - [Fact] - public async Task ReleasingClaims_DoesNotCreateExtraRows() - { - var queue = NewQueue(); - await queue.EnqueueBatchAsync("crashed-run", [("task-0", 4), ("task-1", 4)], At(0)); - await queue.ClaimBatchAsync("dead-worker", 2, TimeSpan.FromMinutes(30)); - - await queue.ReleaseRunClaimsAsync("crashed-run"); - - var ids = await queue.GetQueuedTaskIdsAsync("crashed-run"); - Assert.Equal(["task-0", "task-1"], ids.OrderBy(x => x, StringComparer.Ordinal)); - Assert.Equal(2, (await queue.ClaimBatchAsync("new-worker", 10, TimeSpan.FromMinutes(30))).Count); - } - - /// Another run's claims are left alone — recovery is per run. - [Fact] - public async Task ReleasingOneRunsClaims_LeavesOtherRunsHeld() - { - var queue = NewQueue(); - await queue.EnqueueBatchAsync("crashed-run", [("task-0", 4)], At(0)); - await queue.EnqueueBatchAsync("healthy-run", [("task-9", 4)], At(1)); - await queue.ClaimBatchAsync("worker", 2, TimeSpan.FromMinutes(30)); - - Assert.Equal(1, await queue.ReleaseRunClaimsAsync("crashed-run")); - - var reclaimed = await queue.ClaimBatchAsync("new-worker", 10, TimeSpan.FromMinutes(30)); - Assert.Equal("task-0", Assert.Single(reclaimed).TaskId); - } -} diff --git a/tests/Craft.Tests/JobQueueStatusReaderTests.cs b/tests/Craft.Tests/JobQueueStatusReaderTests.cs index 0485c44..5a58f95 100644 --- a/tests/Craft.Tests/JobQueueStatusReaderTests.cs +++ b/tests/Craft.Tests/JobQueueStatusReaderTests.cs @@ -8,207 +8,120 @@ namespace Craft.Tests; /// -/// The table-backed status view. Since task ownership moved into the queue table, the in-memory -/// JobManager holds only a worker-pool-sized buffer — so every status consumer that reads it as "the -/// queue" reports a 7,000-task fan-out as eight queued jobs. These tests pin the merge semantics: the -/// durable backlog is counted and listed, claimed rows are never double-counted against the local -/// records that represent them, and run sizes come from the counter row rather than from whatever -/// slice this instance happened to claim. +/// The status APIs read the durable backlog from the Ready list: every run is counted from its counts, only +/// the head is listed task by task, and work this process holds is not counted as waiting. /// public class JobQueueStatusReaderTests { - private static readonly TimeSpan Lease = TimeSpan.FromMinutes(20); - private static readonly TimeSpan Fresh = TimeSpan.Zero; - - private sealed class Fixture - { - public required RunRemainingCounterTests.ConditionalStore Backing { get; init; } - public required JobQueueStore Queue { get; init; } - public required OrchestratorTableStore Store { get; init; } - public required JobManager Jobs { get; init; } - public required JobQueueStatusReader Reader { get; init; } - } - - private static async Task NewFixtureAsync() + private static (JobQueueStatusReader Reader, WorkStore Store, JobManager Jobs) New() { var settings = new CraftSettings(); - settings.Worker.BgPoolSize = 2; var config = new ConfigurationBuilder().AddInMemoryCollection([]).Build(); var repo = new ScriptRepository(NullLogger.Instance, settings); var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); var jobs = new JobManager(NullLogger.Instance, settings, limiter); - - var backing = new RunRemainingCounterTests.ConditionalStore(); - var queue = new JobQueueStore(NullLogger.Instance, settings, backing); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); - await queue.InitializeAsync(); - await store.InitializeAsync(); - - return new Fixture - { - Backing = backing, - Queue = queue, - Store = store, - Jobs = jobs, - Reader = new JobQueueStatusReader(NullLogger.Instance, jobs, queue, store), - }; + var store = new WorkStore(NullLogger.Instance, settings, new MemoryTableStore()); + return (new JobQueueStatusReader(NullLogger.Instance, jobs, store), store, jobs); } - private static DateTime At(int minute) => new(2026, 8, 12, 3, minute, 0, DateTimeKind.Utc); - - [Fact] - public async Task SummaryCountsTheDurableBacklog_NotJustTheLocalBuffer() + private static Task Create(WorkStore s, string name, int tasks, int minute) { - var f = await NewFixtureAsync(); - await f.Queue.EnqueueBatchAsync("StandardsApply", - Enumerable.Range(0, 120).Select(i => ($"std-{i}", 4)).ToList(), At(5)); - - var summary = await f.Reader.GetSummaryAsync(); - - Assert.Equal(120, summary.Queued); - Assert.Equal(120, summary.QueuedDurable); - Assert.Equal(0, summary.QueuedLocal); - Assert.Equal(At(5), summary.OldestQueuedUtc); + var started = new DateTime(2026, 10, 5, 3, minute, 0, DateTimeKind.Utc); + return s.CreateRunAsync(new RunHeader + { + RunKey = WorkStore.RunKeyFor(name, started), + Name = name, + StartedUtc = started, + TaskScriptName = "Invoke-CraftTask", + }, Enumerable.Range(0, tasks).Select(i => new WorkStore.NewTask($"t{i}", [])).ToList()); } [Fact] - public async Task ClaimedRowsAreNotCountedAsQueued() + public async Task EveryRunIsCounted_ButOnlyTheHeadIsListed() { - var f = await NewFixtureAsync(); - await f.Queue.EnqueueBatchAsync("run", - Enumerable.Range(0, 10).Select(i => ($"task-{i}", 4)).ToList(), At(0)); - var claimed = await f.Queue.ClaimBatchAsync("worker-a", 4, Lease); - Assert.Equal(4, claimed.Count); - - var summary = await f.Reader.GetSummaryAsync(); - - // The four claimed rows are some instance's buffer — represented by its local records, not by - // the backlog count. - Assert.Equal(6, summary.QueuedDurable); - Assert.Equal(6, summary.Queued); - } + var (reader, store, _) = New(); + for (var i = 0; i < 60; i++) await Create(store, $"Run{i:D2}", 50, i % 60); - [Fact] - public async Task JobDetails_ListTheBacklog_WithoutDuplicatingLocallyClaimedWork() - { - var f = await NewFixtureAsync(); - await f.Queue.EnqueueAsync("run", "claimed-here", 4, At(0)); - await f.Queue.EnqueueAsync("run", "claimed-elsewhere", 4, At(1)); - await f.Queue.EnqueueAsync("run", "waiting", 4, At(2)); - - // Claim one locally, the way the pump does: claim the row, enqueue the descriptor locally. Which - // of the two "claimed-*" tasks comes first is no longer time-ordered under schema v2, so drive the - // local record off whatever was actually claimed rather than a hard-coded id. - var mine = await f.Queue.ClaimBatchAsync("this-node", 1, Lease); - var mineId = Assert.Single(mine).TaskId; - f.Jobs.Enqueue(new JobDescriptor("run", mineId, 4), $"run-{mineId}"); - - // Another instance's claim: a row under lease with no local record at all. - var theirs = await f.Queue.ClaimBatchAsync("other-node", 1, Lease); - var theirsId = Assert.Single(theirs).TaskId; - Assert.NotEqual(mineId, theirsId); - - var details = await f.Reader.GetJobDetailsAsync(); - - // The locally-claimed row once (its local record), the still-waiting row once (durable), the row - // claimed by the other instance not at all — its records live there. ("waiting" sorts last, so the - // two claims took the "claimed-*" pair and it is what remains.) - Assert.Equal(2, details.Count); - Assert.Single(details, d => d.Id == $"run-{mineId}"); - Assert.DoesNotContain(details, d => d.Id == $"run-{theirsId}"); - var waiting = Assert.Single(details, d => d.Id == "run-waiting"); - Assert.Equal("Queued", waiting.Status); - Assert.Equal(At(2), waiting.QueuedUtc); - Assert.True(waiting.WaitSeconds > 0); + var snap = (await reader.GetAsync())!; + + Assert.Equal(3_000, snap.Total); + Assert.Equal(3_000, snap.Unclaimed); + Assert.Equal(60, snap.ByRun.Count); + Assert.Equal(JobQueueStatusReader.HeadRows, snap.Rows.Count); + Assert.Equal("Run00", snap.Rows[0].RunName); } [Fact] - public async Task JobDetails_RespectStatusFilterAndLimit() + public async Task WorkThisProcessHolds_IsNotCountedAsWaiting() { - var f = await NewFixtureAsync(); - await f.Queue.EnqueueBatchAsync("run", - Enumerable.Range(0, 5).Select(i => ($"task-{i}", 4)).ToList(), At(0)); - - // A non-Queued filter describes claimed work, which only local records know about. - Assert.Empty(await f.Reader.GetJobDetailsAsync(status: "Running")); - - var queuedOnly = await f.Reader.GetJobDetailsAsync(status: "Queued"); - Assert.Equal(5, queuedOnly.Count); - Assert.All(queuedOnly, d => Assert.Equal("Queued", d.Status)); - - Assert.Equal(3, (await f.Reader.GetJobDetailsAsync(limit: 3)).Count); + var (reader, store, jobs) = New(); + var run = await Create(store, "Busy", 5, 0); + foreach (var c in await store.ClaimAsync(run.RunKey, 2, "me", TimeSpan.FromMinutes(5), false)) + jobs.Enqueue(new JobDescriptor("Busy", c.TaskId, 4) { RunKey = c.RunKey, Seq = c.Seq }, $"Busy-{c.TaskId}"); + + var snap = (await reader.GetAsync())!; + + Assert.Equal(5, snap.Total); + Assert.Equal(3, snap.Unclaimed); + Assert.Equal(3, snap.Rows.Count); + var summary = await reader.GetSummaryAsync(); + Assert.Equal(3, summary.QueuedDurable); } + /// + /// The snapshot is cached for seconds while this process keeps claiming. Merged views once added the live + /// local queue to the snapshot's own "unclaimed", which had already subtracted the claims held when it was + /// taken, so every claim made since was counted twice (seen live: 202 queued on a 200-task run). + /// [Fact] - public async Task RunSummaries_SizeARunFromItsCounter_NotFromTheClaimedSlice() + public async Task ClaimsMadeAfterTheSnapshot_AreNotCountedTwice() { - var f = await NewFixtureAsync(); - - // A 10-task run: one durably finished, one claimed into this instance, eight still queued. - await f.Store.InitRemainingAsync("run", 10); - await f.Store.DecrementRemainingAsync("run", 1); - await f.Queue.EnqueueBatchAsync("run", - Enumerable.Range(0, 9).Select(i => ($"task-{i}", 4)).ToList(), At(0)); - var claimed = await f.Queue.ClaimBatchAsync("this-node", 1, Lease); - f.Jobs.Enqueue(new JobDescriptor("run", claimed[0].TaskId, 4), $"run-{claimed[0].TaskId}"); - - var summary = Assert.Single(await f.Reader.GetRunSummariesAsync(), s => s.Name == "run"); - - Assert.Equal(10, summary.Total); - Assert.Equal(9, summary.Queued); // 8 unclaimed + 1 buffered locally - Assert.Equal(1, summary.Completed); // Total − Remaining, durable across restarts + var (reader, store, jobs) = New(); + var run = await Create(store, "Busy", 5, 0); + void Hold(IEnumerable claims) + { + foreach (var c in claims) jobs.Enqueue(new JobDescriptor("Busy", c.TaskId, 4) { RunKey = c.RunKey, Seq = c.Seq }, $"Busy-{c.TaskId}", $"{c.RunKey}|{c.Seq}"); + } + Hold(await store.ClaimAsync(run.RunKey, 2, "me", TimeSpan.FromMinutes(5), false)); + await reader.GetAsync(TimeSpan.FromMinutes(5)); // cached with 2 held + Hold(await store.ClaimAsync(run.RunKey, 2, "me", TimeSpan.FromMinutes(5), false)); + + var summary = await reader.GetSummaryAsync(); + Assert.Equal(1, summary.QueuedDurable); + Assert.Equal(5, summary.Queued + summary.Running); + + var busy = Assert.Single(await reader.GetRunSummariesAsync(), r => r.Name == "Busy"); + Assert.Equal(5, busy.Total); + Assert.True(busy.Queued + busy.Running + busy.Completed + busy.Failed <= busy.Total, + $"queued {busy.Queued} running {busy.Running} completed {busy.Completed} failed {busy.Failed} over total {busy.Total}"); + Assert.Equal(5, busy.Queued + busy.Running); } [Fact] - public async Task RunSummaries_SynthesizeARunTheJobManagerHasNeverSeen() + public async Task RunsSharingAName_AreSummedUnderThatName() { - var f = await NewFixtureAsync(); - await f.Store.InitRemainingAsync("cold-run", 50); - await f.Queue.EnqueueBatchAsync("cold-run", - Enumerable.Range(0, 50).Select(i => ($"task-{i}", 3)).ToList(), At(0)); + var (reader, store, _) = New(); + await Create(store, "Twin", 3, 0); + await Create(store, "Twin", 4, 1); - var summary = Assert.Single(await f.Reader.GetRunSummariesAsync(), s => s.Name == "cold-run"); + var info = (await reader.GetAsync())!.ByRun["Twin"]; - Assert.Equal(50, summary.Total); - Assert.Equal(50, summary.Queued); - Assert.Equal(3, summary.Priority); - Assert.Equal(0, summary.Running); + Assert.Equal(7, info.Total); + Assert.Equal(7, info.Unclaimed); } [Fact] - public async Task AFailedRefreshServesThePreviousSnapshot() + public async Task AFinishedRun_LeavesTheBacklog() { - var f = await NewFixtureAsync(); - await f.Queue.EnqueueAsync("run", "task-0", 4, At(0)); + var (reader, store, _) = New(); + var run = await Create(store, "Quick", 2, 0); + var claims = await store.ClaimAsync(run.RunKey, 2, "w", TimeSpan.FromMinutes(5), false); + await store.FinishAsync(run.RunKey, claims.Select(c => new WorkStore.Finish(c.Seq, "Completed", Owner: "w")).ToList()); - var first = await f.Reader.GetAsync(Fresh); - Assert.NotNull(first); - Assert.Equal(1, first!.Unclaimed); + var snap = (await reader.GetAsync())!; - f.Backing.OnBeforeQuery = () => throw new InvalidOperationException("storage is down"); - var second = await f.Reader.GetAsync(Fresh); - - // Stale data with an honest timestamp, not an exception on a health endpoint. - Assert.Same(first, second); - } - - [Fact] - public async Task SnapshotAggregatesPerRun() - { - var f = await NewFixtureAsync(); - await f.Queue.EnqueueBatchAsync("run-a", - Enumerable.Range(0, 3).Select(i => ($"a-{i}", 4)).ToList(), At(0)); - await f.Queue.EnqueueBatchAsync("run-b", - Enumerable.Range(0, 2).Select(i => ($"b-{i}", 1)).ToList(), At(1)); - - var snap = await f.Reader.GetAsync(Fresh); - - Assert.NotNull(snap); - Assert.Equal(5, snap!.Total); - Assert.Equal(3, snap.ByRun["run-a"].Unclaimed); - Assert.Equal(2, snap.ByRun["run-b"].Unclaimed); - Assert.Equal(1, snap.ByRun["run-b"].MinPriority); - Assert.Equal(At(0), snap.OldestUnclaimedUtc); + Assert.Equal(0, snap.Total); + Assert.Empty(snap.ByRun); } } diff --git a/tests/Craft.Tests/JobQueueStatusRefreshCostTests.cs b/tests/Craft.Tests/JobQueueStatusRefreshCostTests.cs deleted file mode 100644 index 330cd7f..0000000 --- a/tests/Craft.Tests/JobQueueStatusRefreshCostTests.cs +++ /dev/null @@ -1,215 +0,0 @@ -using Craft.Configuration; -using Craft.Orchestration; -using Craft.PowerShellHost; -using Craft.Storage; -using Microsoft.Extensions.Configuration; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// What a status refresh COSTS, as opposed to what it reports. -/// -/// The status snapshot is polled — the stats sampler on its timer, the worker-health page at a few -/// hertz, the perf harness at 4 Hz — and its scan is proportional to the backlog. Two properties kept -/// that from being sustainable on a large instance, and neither is visible from the values returned: -/// -/// 1. The snapshot did a per-run counter read while aggregating, so a backlog spanning 12,028 runs -/// cost 12,028 point reads per refresh — on every poll, for data only the run-summaries listing -/// ever read. -/// 2. The snapshot was stamped with the time the refresh STARTED, so a scan slower than the TTL -/// returned something already expired and the next poll immediately started another. Measured on -/// a 743,000-row queue: continuous back-to-back full scans, gated only by the single-flight lock. -/// -/// Both are cost properties, so both are asserted by counting storage calls and by reading the -/// snapshot's own age — not by checking the numbers it reports, which were correct throughout. -/// -public class JobQueueStatusRefreshCostTests -{ - /// Counts reads and can make the queue scan take a controllable amount of time. - private sealed class CountingStore(ICraftTableStore inner, string queueTable) : ICraftTableStore - { - private int _pointReads; - private int _tableScans; - - /// Point reads (GetAsync) issued since the last . - public int PointReads => Volatile.Read(ref _pointReads); - - /// Full scans of the queue table since the last . - public int QueueScans => Volatile.Read(ref _tableScans); - - /// Injected per-scan delay, to model a backlog large enough to outlast the TTL. - public TimeSpan ScanDelay { get; set; } = TimeSpan.Zero; - - public void Reset() { Volatile.Write(ref _pointReads, 0); Volatile.Write(ref _tableScans, 0); } - - public Task PingAsync(CancellationToken ct = default) => inner.PingAsync(ct); - public Task EnsureTableAsync(string table, CancellationToken ct = default) => inner.EnsureTableAsync(table, ct); - public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) => inner.UpsertAsync(table, row, ct); - public Task UpsertBatchAsync(string table, string pk, IReadOnlyList rows, CancellationToken ct = default) - => inner.UpsertBatchAsync(table, pk, rows, ct); - public Task TryReplaceBatchAsync(string table, string pk, IReadOnlyList rows, CancellationToken ct = default) - => inner.TryReplaceBatchAsync(table, pk, rows, ct); - public Task DeleteAsync(string table, string pk, string rk, CancellationToken ct = default) => inner.DeleteAsync(table, pk, rk, ct); - public Task DeletePartitionAsync(string table, string pk, CancellationToken ct = default) => inner.DeletePartitionAsync(table, pk, ct); - - public Task GetAsync(string table, string pk, string rk, CancellationToken ct = default) - { - Interlocked.Increment(ref _pointReads); - return inner.GetAsync(table, pk, rk, ct); - } - - public IAsyncEnumerable QueryPartitionAsync(string table, string pk, CancellationToken ct = default) - => inner.QueryPartitionAsync(table, pk, ct); - - public async IAsyncEnumerable QueryTableAsync(string table, - [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) - { - if (table == queueTable) - { - Interlocked.Increment(ref _tableScans); - if (ScanDelay > TimeSpan.Zero) await Task.Delay(ScanDelay, ct); - } - await foreach (var row in inner.QueryTableAsync(table, ct)) yield return row; - } - - public IAsyncEnumerable QueryTableAsync(string table, string? filter, CancellationToken ct = default) - => QueryTableAsync(table, ct); - } - - private sealed class Fixture - { - public required CountingStore Counting { get; init; } - public required JobQueueStore Queue { get; init; } - public required OrchestratorTableStore Store { get; init; } - public required JobQueueStatusReader Reader { get; init; } - } - - private static async Task NewFixtureAsync() - { - var settings = new CraftSettings(); - settings.Worker.BgPoolSize = 2; - var config = new ConfigurationBuilder().AddInMemoryCollection([]).Build(); - var repo = new ScriptRepository(NullLogger.Instance, settings); - var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); - var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); - var jobs = new JobManager(NullLogger.Instance, settings, limiter); - - var counting = new CountingStore(new RunRemainingCounterTests.ConditionalStore(), - $"{settings.Orchestrator.TablePrefix}Queue"); - var queue = new JobQueueStore(NullLogger.Instance, settings, counting); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, counting); - await queue.InitializeAsync(); - await store.InitializeAsync(); - - return new Fixture - { - Counting = counting, - Queue = queue, - Store = store, - Reader = new JobQueueStatusReader(NullLogger.Instance, jobs, queue, store), - }; - } - - private static DateTime At(int minute) => new(2026, 8, 12, 3, minute, 0, DateTimeKind.Utc); - - [Fact] - public async Task SnapshotCostIsOneScan_RegardlessOfHowManyRunsTheBacklogSpans() - { - var f = await NewFixtureAsync(); - - // 40 distinct runs, each with a counter row — the shape that used to cost 40 point reads per - // refresh, and 12,028 on the instance that motivated this. - for (var i = 0; i < 40; i++) - { - await f.Store.InitRemainingAsync($"run-{i}", 2); - await f.Queue.EnqueueBatchAsync($"run-{i}", [("t0", 4), ("t1", 4)], At(0)); - } - - f.Counting.Reset(); - var snap = await f.Reader.GetAsync(TimeSpan.Zero); - - Assert.NotNull(snap); - Assert.Equal(40, snap!.ByRun.Count); - Assert.Equal(1, f.Counting.QueueScans); - - // The point of the change: aggregating the backlog reads nothing per run. - Assert.Equal(0, f.Counting.PointReads); - } - - [Fact] - public async Task ASlowScanDoesNotReturnAnAlreadyExpiredSnapshot() - { - var f = await NewFixtureAsync(); - await f.Queue.EnqueueBatchAsync("run", [("t0", 4)], At(0)); - - // A scan that takes far longer than the age the caller asked for. Stamped on entry, the - // returned snapshot would be born ~400ms old and instantly past a 100ms TTL, so the next poll - // starts another — the loop this guards against. - f.Counting.ScanDelay = TimeSpan.FromMilliseconds(400); - - var snap = await f.Reader.GetAsync(TimeSpan.FromMilliseconds(100)); - - Assert.NotNull(snap); - Assert.True(snap!.AgeSeconds < 0.2, - $"snapshot was born {snap.AgeSeconds:N3}s old — TakenUtc is being stamped before the scan"); - } - - [Fact] - public async Task TheRefreshIntervalBacksOffToWhatTheScanActuallyCosts() - { - var f = await NewFixtureAsync(); - await f.Queue.EnqueueBatchAsync("run", [("t0", 4)], At(0)); - - f.Counting.ScanDelay = TimeSpan.FromMilliseconds(400); - await f.Reader.GetAsync(TimeSpan.FromMilliseconds(50)); // records the cost - - f.Counting.Reset(); - - // Older than the 50ms the caller wants, well inside the 400ms the scan actually costs. Without - // the floor this re-scans; with it the cached snapshot stands. - await Task.Delay(120); - var again = await f.Reader.GetAsync(TimeSpan.FromMilliseconds(50)); - - Assert.NotNull(again); - Assert.Equal(0, f.Counting.QueueScans); - } - - [Fact] - public async Task AFastScanStillRefreshesOnTheOrdinaryTtl() - { - // The backoff must not become a permanent cache on a healthy instance, where the scan is - // milliseconds and the floor should collapse back to the caller's requested age. - var f = await NewFixtureAsync(); - await f.Queue.EnqueueBatchAsync("run", [("t0", 4)], At(0)); - - await f.Reader.GetAsync(TimeSpan.Zero); - f.Counting.Reset(); - - await Task.Delay(60); - await f.Reader.GetAsync(TimeSpan.FromMilliseconds(20)); - - Assert.Equal(1, f.Counting.QueueScans); - } - - [Fact] - public async Task RunSummariesStillSizeRunsFromTheirCounter() - { - // The counter reads moved out of the snapshot and into the one caller that reads them; this is - // the behaviour that must survive the move. - var f = await NewFixtureAsync(); - await f.Store.InitRemainingAsync("run", 10); - await f.Store.DecrementRemainingAsync("run", 1); - await f.Queue.EnqueueBatchAsync("run", - Enumerable.Range(0, 9).Select(i => ($"task-{i}", 4)).ToList(), At(0)); - - f.Counting.Reset(); - var summary = Assert.Single(await f.Reader.GetRunSummariesAsync(), s => s.Name == "run"); - - Assert.Equal(10, summary.Total); - Assert.Equal(1, summary.Completed); - - // Resolved here, and only here: one point read for the one active run. - Assert.Equal(1, f.Counting.PointReads); - } -} diff --git a/tests/Craft.Tests/JobQueueStoreTests.cs b/tests/Craft.Tests/JobQueueStoreTests.cs deleted file mode 100644 index 7ed2264..0000000 --- a/tests/Craft.Tests/JobQueueStoreTests.cs +++ /dev/null @@ -1,336 +0,0 @@ -using Craft.Configuration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// The durable job queue that lets the dispatch side hold a worker-pool-sized buffer instead of the -/// whole backlog, and makes the row in storage — not the in-memory copy — the thing that actually runs. -/// -/// Two properties carry the design: -/// -/// Ordering. Rows sort by partition key then row key, so a single read yields the highest-priority, -/// oldest-first work across every run. That is what the in-memory PriorityQueue gives today, and it -/// has to survive the move or a P0 audit-log task ends up behind thousands of P4 standards. -/// -/// Exclusivity. A claim is one conditional transaction, so two workers cannot take the same task — -/// the loser is rejected and retries rather than forcing the write. This is the property the -/// 28-hour deadlock's redrive heuristic was standing in for. -/// -public class JobQueueStoreTests -{ - private static readonly TimeSpan Lease = TimeSpan.FromMinutes(20); - - private static (JobQueueStore Queue, RunRemainingCounterTests.ConditionalStore Backing) NewQueue() - { - var backing = new RunRemainingCounterTests.ConditionalStore(); - var queue = new JobQueueStore(NullLogger.Instance, new CraftSettings(), backing); - return (queue, backing); - } - - private static DateTime At(int minute) => new(2026, 8, 9, 2, minute, 0, DateTimeKind.Utc); - - [Fact] - public async Task ClaimsHighestPriorityFirstAcrossRuns() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - - // Queued in the least helpful order: the low-priority bulk first, the urgent one last. - await queue.EnqueueBatchAsync("StandardsApply", - Enumerable.Range(0, 5).Select(i => ($"std-{i}", 4)).ToList(), At(0)); - await queue.EnqueueAsync("AuditLogIngest", "audit-0", 0, At(5)); - - var claimed = await queue.ClaimBatchAsync("worker-a", 2, Lease); - - Assert.Single(claimed); - Assert.Equal("audit-0", claimed[0].TaskId); - Assert.Equal("AuditLogIngest", claimed[0].RunName); - } - - [Fact] - public async Task ReEnqueuingATaskUpsertsOneRowRatherThanDuplicating() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - - // The re-dispatch case (crash recovery, orphan re-drive). Schema v2 keys deterministically per - // (run, task), so the second enqueue UPDATES the first row instead of adding a duplicate — the - // duplicate that used to get claimed and executed a second time. - await queue.EnqueueAsync("run", "task-0", 4, At(1)); - await queue.EnqueueAsync("run", "task-0", 4, At(10)); - - Assert.Single(await queue.GetQueuedTaskIdsAsync("run")); - - Assert.Equal("task-0", Assert.Single(await queue.ClaimBatchAsync("worker-a", 8, Lease)).TaskId); - // Only one row ever existed, so nothing is left to claim a second time. - Assert.Empty(await queue.ClaimBatchAsync("worker-b", 8, Lease)); - } - - /// - /// THE GUARANTEE. Two workers claiming at once must not both get the same task. The batch is one - /// conditional transaction, so the loser comes back empty and retries. - /// - [Fact] - public async Task TwoWorkersNeverClaimTheSameTask() - { - var (queue, backing) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueBatchAsync("run", Enumerable.Range(0, 4).Select(i => ($"task-{i}", 4)).ToList(), At(0)); - - // Land a competing claim in the window between our read and our write. - IReadOnlyList competitor = []; - backing.OnBeforeConditionalWrite = () => - { - backing.OnBeforeConditionalWrite = null; - competitor = queue.ClaimBatchAsync("worker-b", 4, Lease).GetAwaiter().GetResult(); - }; - - var mine = await queue.ClaimBatchAsync("worker-a", 4, Lease); - - Assert.Equal(4, competitor.Count); - Assert.Empty(mine); - - // And nothing is left claimable, rather than the rows being double-owned. - Assert.Empty(await queue.ClaimBatchAsync("worker-c", 4, Lease)); - } - - [Fact] - public async Task ClaimedWorkIsNotHandedOutAgainWhileTheLeaseHolds() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueBatchAsync("run", Enumerable.Range(0, 3).Select(i => ($"task-{i}", 4)).ToList(), At(0)); - - var first = await queue.ClaimBatchAsync("worker-a", 2, Lease); - var second = await queue.ClaimBatchAsync("worker-b", 2, Lease); - - Assert.Equal(2, first.Count); - Assert.Single(second); - Assert.Empty(first.Select(j => j.TaskId).Intersect(second.Select(j => j.TaskId))); - } - - /// - /// A worker that dies holding a claim must give the work back on its own. This is what replaces the - /// age-based re-drive: nothing has to notice the worker is gone. - /// - [Fact] - public async Task ExpiredLeaseMakesWorkClaimableAgain() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueAsync("run", "task-0", 4, At(0)); - - var held = await queue.ClaimBatchAsync("worker-a", 1, TimeSpan.FromMilliseconds(40)); - Assert.Single(held); - - Assert.Empty(await queue.ClaimBatchAsync("worker-b", 1, Lease)); - - await Task.Delay(120); - - var reclaimed = await queue.ClaimBatchAsync("worker-b", 1, Lease); - Assert.Equal("task-0", Assert.Single(reclaimed).TaskId); - } - - [Fact] - public async Task RenewingKeepsTheClaimAndFailsOnceItHasLapsed() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueAsync("run", "task-0", 4, At(0)); - - var held = await queue.ClaimBatchAsync("worker-a", 1, TimeSpan.FromMilliseconds(40)); - - Assert.True(await queue.RenewAsync(held, "worker-a", Lease)); - Assert.Empty(await queue.ClaimBatchAsync("worker-b", 1, Lease)); - - // Someone else now owns it — renewal must report that rather than taking it back. - await queue.RemoveAsync(held[0]); - await queue.EnqueueAsync("run", "task-0", 4, At(0)); - await queue.ClaimBatchAsync("worker-b", 1, Lease); - - Assert.False(await queue.RenewAsync(held, "worker-a", Lease)); - } - - [Fact] - public async Task RemovedWorkIsGoneFromTheQueue() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueBatchAsync("run", [("task-0", 4), ("task-1", 4)], At(0)); - - var claimed = await queue.ClaimBatchAsync("worker-a", 2, Lease); - foreach (var job in claimed) await queue.RemoveAsync(job); - - // Nothing is left, even once the leases would have lapsed. - Assert.Empty(await queue.ClaimBatchAsync("worker-b", 2, TimeSpan.Zero)); - } - - [Fact] - public async Task RemovingARunClearsOnlyThatRun() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueBatchAsync("doomed", [("a", 4), ("b", 4)], At(0)); - await queue.EnqueueAsync("survivor", "c", 4, At(1)); - - await queue.RemoveRunAsync("doomed"); - - var left = await queue.ClaimBatchAsync("worker-a", 10, Lease); - Assert.Equal("survivor", Assert.Single(left).RunName); - } - - [Fact] - public void PriorityBucketsSortNumericallyNotLexically() - { - // "P4" vs "P10" would order 10 before 4 as strings, quietly inverting priority. - Assert.Equal("P00", JobQueueStore.Bucket(0)); - Assert.Equal("P04", JobQueueStore.Bucket(4)); - Assert.Equal("P10", JobQueueStore.Bucket(10)); - Assert.True(string.CompareOrdinal(JobQueueStore.Bucket(4), JobQueueStore.Bucket(10)) < 0); - } - - [Fact] - public void RowKeyIsDeterministicPerRunAndTask() - { - // Schema v2: the key is a function of (run, task) only, so re-dispatching a task upserts its one - // row instead of writing a second, time-prefixed one — the duplicate-execution class. - Assert.Equal(JobQueueStore.BuildRowKey("r", "t"), JobQueueStore.BuildRowKey("r", "t")); - Assert.NotEqual(JobQueueStore.BuildRowKey("r", "t1"), JobQueueStore.BuildRowKey("r", "t2")); - Assert.NotEqual(JobQueueStore.BuildRowKey("r1", "t"), JobQueueStore.BuildRowKey("r2", "t")); - } - - [Fact] - public async Task EmptyQueueClaimsNothingRatherThanBlocking() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - - Assert.Empty(await queue.ClaimBatchAsync("worker-a", 8, Lease)); - } - - // ─── Status/maintenance surface (the table-backed worker-health view) ─── - - [Fact] - public void RowKeyEscapingKeepsDistinctPairsDistinctAndKeysLegal() - { - // The '|' separator and '%' escape are themselves escaped, so "a|b"+"c" and "a"+"b|c" cannot - // collide onto one row. - Assert.NotEqual(JobQueueStore.BuildRowKey("a|b", "c"), JobQueueStore.BuildRowKey("a", "b|c")); - - // An illegal character in a component is escaped away, keeping the key legal for Azure Table. - var key = JobQueueStore.BuildRowKey("run", "Owner/Repo - No tenant"); - Assert.DoesNotContain(key, c => c is '/' or '\\' or '#' or '?' || char.IsControl(c)); - } - - [Fact] - public async Task ClearAllEmptiesTheQueueButLeavesItUsable() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueBatchAsync("run-a", - Enumerable.Range(0, 3).Select(i => ($"task-{i}", 4)).ToList(), At(0)); - await queue.EnqueueAsync("run-b", "solo", 4, At(0)); - - var removed = await queue.ClearAllAsync(); - - Assert.Equal(4, removed); - Assert.Empty(await queue.ListQueuedAsync()); - Assert.Empty(await queue.GetQueuedTaskIdsAsync("run-a")); - - // The schema marker survives, so the queue keeps working — a fresh enqueue lands and is claimable. - await queue.EnqueueAsync("run-c", "again", 4, At(1)); - Assert.Equal("again", Assert.Single(await queue.ClaimBatchAsync("w", 8, Lease)).TaskId); - } - - [Fact] - public async Task ListQueuedReportsIdentityAgeAndClaimState() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueAsync("run", "waiting", 4, At(2)); - await queue.EnqueueAsync("run", "taken", 0, At(1)); - await queue.ClaimBatchAsync("worker-a", 1, Lease); - - var rows = await queue.ListQueuedAsync(); - - Assert.Equal(2, rows.Count); - - var taken = Assert.Single(rows, r => r.TaskId == "taken"); - Assert.True(taken.Claimed); - Assert.Equal("worker-a", taken.Owner); - Assert.Equal(0, taken.Priority); - Assert.Equal(At(1), taken.QueuedUtc); - - var waiting = Assert.Single(rows, r => r.TaskId == "waiting"); - Assert.False(waiting.Claimed); - Assert.Equal(At(2), waiting.QueuedUtc); - } - - [Fact] - public async Task ListQueuedTreatsALapsedLeaseAsUnclaimed() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueAsync("run", "task-0", 4, At(0)); - await queue.ClaimBatchAsync("worker-a", 1, TimeSpan.FromMilliseconds(40)); - - await Task.Delay(120); - - Assert.False(Assert.Single(await queue.ListQueuedAsync()).Claimed); - } - - [Fact] - public async Task RemoveTaskRemovesOnlyThatTasksRows() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueAsync("run", "doomed", 4, At(0)); - await queue.EnqueueAsync("run", "survivor", 4, At(0)); - await queue.EnqueueAsync("other-run", "doomed", 4, At(0)); - - Assert.Equal(1, await queue.RemoveTaskAsync("run", "doomed")); - - var left = await queue.ListQueuedAsync(); - Assert.Equal(2, left.Count); - Assert.DoesNotContain(left, r => r.RunName == "run" && r.TaskId == "doomed"); - } - - [Fact] - public async Task ReprioritizeMovesTheRowKeepingItsPlaceInLine() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueAsync("run", "older", 4, At(0)); - await queue.EnqueueAsync("run", "boosted", 4, At(1)); - await queue.EnqueueAsync("run", "urgent", 0, At(2)); - - Assert.Equal(1, await queue.ReprioritizeTaskAsync("run", "boosted", 0)); - - // The moved row keeps its enqueue time, so within P0 the earlier "boosted" outranks "urgent". - var claimed = await queue.ClaimBatchAsync("worker-a", 2, Lease); - Assert.Equal(["boosted", "urgent"], claimed.Select(j => j.TaskId).ToArray()); - Assert.All(claimed, j => Assert.Equal(0, j.Priority)); - } - - /// - /// A claimed row is already buffered inside some instance — re-adding it unclaimed at a new - /// priority would create a second runnable copy of the task, the exact failure this queue's - /// claim semantics exist to prevent. - /// - [Fact] - public async Task ReprioritizeLeavesClaimedRowsAlone() - { - var (queue, _) = NewQueue(); - await queue.InitializeAsync(); - await queue.EnqueueAsync("run", "task-0", 4, At(0)); - await queue.ClaimBatchAsync("worker-a", 1, Lease); - - Assert.Equal(0, await queue.ReprioritizeTaskAsync("run", "task-0", 0)); - - var row = Assert.Single(await queue.ListQueuedAsync()); - Assert.Equal(4, row.Priority); - Assert.True(row.Claimed); - } -} diff --git a/tests/Craft.Tests/MemoryTableStore.cs b/tests/Craft.Tests/MemoryTableStore.cs new file mode 100644 index 0000000..27caba3 --- /dev/null +++ b/tests/Craft.Tests/MemoryTableStore.cs @@ -0,0 +1,211 @@ +using Craft.Storage; + +namespace Craft.Tests; + +/// +/// In-memory with the backend properties the orchestration relies on: rows come +/// back ordered by PartitionKey then RowKey, every write stamps a new ETag, and +/// is all-or-nothing under one lock. Reads hand out copies, so mutating a row +/// read from here never writes it. Partition and row-key range reads are index seeks, as on Azure, so a test +/// with a large table pays for the rows it reads rather than for the whole table. +/// +internal sealed class MemoryTableStore : ICraftTableStore +{ + private readonly object _lock = new(); + private readonly Dictionary _tables = new(); + private long _etag; + + /// Awaited before a transaction is checked — lets a test interleave a competing write. + public Func? BeforeSubmit { get; set; } + + public int Submits { get; private set; } + + /// Tables deleted through , in order. + public List DroppedTables { get; } = []; + + public Task DeleteTableAsync(string table, CancellationToken ct = default) + { + lock (_lock) + { + DroppedTables.Add(table); + _tables.Remove(table); + } + return Task.CompletedTask; + } + + private static readonly Comparer<(string, string)> Order = Comparer<(string, string)>.Create((a, b) => + { + var c = string.CompareOrdinal(a.Item1, b.Item1); + return c != 0 ? c : string.CompareOrdinal(a.Item2, b.Item2); + }); + + private sealed class Table + { + public readonly Dictionary<(string, string), StoreRow> Rows = new(); + public readonly SortedSet<(string, string)> Keys = new(Order); + + public void Set((string, string) key, StoreRow row) + { + Rows[key] = row; + Keys.Add(key); + } + + public void Remove((string, string) key) + { + Rows.Remove(key); + Keys.Remove(key); + } + + /// Keys from (inclusive) to (exclusive), in order. + public IEnumerable<(string, string)> Between((string, string) from, (string, string) to) => + Order.Compare(from, to) >= 0 ? [] : Keys.GetViewBetween(from, to).Where(k => Order.Compare(k, to) < 0); + } + + private Table Of(string t) + { + if (!_tables.TryGetValue(t, out var rows)) _tables[t] = rows = new Table(); + return rows; + } + + private StoreRow Stamp(StoreRow r) => new(r.PartitionKey, r.RowKey) + { + ETag = $"W/\"{++_etag}\"", + Timestamp = DateTimeOffset.UtcNow, + Properties = new Dictionary(r.Properties), + }; + + private static StoreRow Copy(StoreRow r) => new(r.PartitionKey, r.RowKey) + { + ETag = r.ETag, + Timestamp = r.Timestamp, + Properties = new Dictionary(r.Properties), + }; + + public IReadOnlyList All(string table) + { + lock (_lock) + { + var t = Of(table); + return t.Keys.Select(k => Copy(t.Rows[k])).ToList(); + } + } + + public Task PingAsync(CancellationToken ct = default) => Task.CompletedTask; + + public Task EnsureTableAsync(string table, CancellationToken ct = default) + { + lock (_lock) Of(table); + return Task.CompletedTask; + } + + public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) + { + lock (_lock) Of(table).Set((row.PartitionKey, row.RowKey), Stamp(row)); + return Task.CompletedTask; + } + + public Task UpsertBatchAsync(string table, string partitionKey, IReadOnlyList rows, CancellationToken ct = default) + { + lock (_lock) foreach (var r in rows) Of(table).Set((r.PartitionKey, r.RowKey), Stamp(r)); + return Task.CompletedTask; + } + + public Task TryReplaceBatchAsync(string table, string partitionKey, IReadOnlyList rows, CancellationToken ct = default) => + TrySubmitAsync(table, partitionKey, rows.Select(StoreOp.Replace).ToList(), ct); + + public async Task TrySubmitAsync(string table, string partitionKey, IReadOnlyList ops, CancellationToken ct = default) + { + if (BeforeSubmit is { } before) await before(); + return Submit(table, ops); + } + + private bool Submit(string table, IReadOnlyList ops) + { + lock (_lock) + { + var t = Of(table); + foreach (var op in ops) + { + t.Rows.TryGetValue((op.Row.PartitionKey, op.Row.RowKey), out var cur); + var ok = op.Kind switch + { + StoreOpKind.Insert => cur == null, + StoreOpKind.Replace => cur != null && cur.ETag == op.Row.ETag, + StoreOpKind.Delete => op.Row.ETag == null || (cur != null && cur.ETag == op.Row.ETag), + _ => true, + }; + if (!ok) return false; + } + foreach (var op in ops) + { + var key = (op.Row.PartitionKey, op.Row.RowKey); + if (op.Kind == StoreOpKind.Delete) t.Remove(key); + else t.Set(key, Stamp(op.Row)); + } + Submits++; + return true; + } + } + + public Task GetAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) + { + lock (_lock) + return Task.FromResult(Of(table).Rows.TryGetValue((partitionKey, rowKey), out var r) ? Copy(r) : null); + } + + private List Snapshot(string table, Func> keys) + { + lock (_lock) + { + var t = Of(table); + return keys(t).Select(k => Copy(t.Rows[k])).ToList(); + } + } + + public async IAsyncEnumerable QueryPartitionAsync(string table, string partitionKey, + [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) + { + foreach (var r in Snapshot(table, t => t.Between((partitionKey, ""), (partitionKey + "\0", "")))) + { + yield return r; + await Task.Yield(); + } + } + + public async IAsyncEnumerable QueryRowKeyRangeAsync(string table, string partitionKey, string fromRowKey, + string toRowKey, IReadOnlyList? properties = null, int? maxPerPage = null, + [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) + { + foreach (var r in Snapshot(table, t => t.Between((partitionKey, fromRowKey), (partitionKey, toRowKey)))) + { + yield return r; + await Task.Yield(); + } + } + + public async IAsyncEnumerable QueryTableAsync(string table, + [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) + { + foreach (var r in Snapshot(table, t => t.Keys)) + { + yield return r; + await Task.Yield(); + } + } + + public Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) + { + lock (_lock) Of(table).Remove((partitionKey, rowKey)); + return Task.CompletedTask; + } + + public Task DeletePartitionAsync(string table, string partitionKey, CancellationToken ct = default) + { + lock (_lock) + { + var t = Of(table); + foreach (var k in t.Between((partitionKey, ""), (partitionKey + "\0", "")).ToList()) t.Remove(k); + } + return Task.CompletedTask; + } +} diff --git a/tests/Craft.Tests/OrchestrationAzuriteTests.cs b/tests/Craft.Tests/OrchestrationAzuriteTests.cs new file mode 100644 index 0000000..d5f41d2 --- /dev/null +++ b/tests/Craft.Tests/OrchestrationAzuriteTests.cs @@ -0,0 +1,266 @@ +using System.Text.Json; +using Craft.Configuration; +using Craft.Storage; + +namespace Craft.Tests; + +/// +/// The orchestration end to end on a real table backend: entity-group transactions, ETag guards, row-key range +/// queries and key escaping as Azure applies them, none of which the in-memory store can prove. Azurite by +/// default, a real account via CRAFT_TEST_TABLE_CONNECTION; each test is skipped, not failed, when neither is +/// reachable, and works in its own table prefix. +/// +[Collection(LargeAllocationSerialTests.Name)] +public class OrchestrationAzuriteTests +{ + private static async Task TryCreateAsync(int poolSize = 4, Func? wrap = null) + { + var settings = new CraftSettings(); + var connection = Environment.GetEnvironmentVariable("CRAFT_TEST_TABLE_CONNECTION"); + if (!string.IsNullOrWhiteSpace(connection)) settings.Auth.UserStorageConnection = connection; + else settings.Storage.AllowDevelopmentStorage = true; + + var tables = new AzureTableStore(settings); + try + { + using var cts = new CancellationTokenSource(TimeSpan.FromSeconds(3)); + await tables.PingAsync(cts.Token); + } + catch + { + return null; + } + + var prefix = "azo" + Guid.NewGuid().ToString("N")[..10]; + return await OrchestrationHarness.CreateAsync(poolSize, wrap?.Invoke(tables) ?? tables, s => s.Orchestrator.TablePrefix = prefix); + } + + private static string Batch(int n, string prefix) => OrchestrationHarness.Batch(n, prefix); + + [Fact] + public async Task AFanOutWithALargeAggregationPayload_RunsEveryTaskOnce_AndAggregatesThemAll() + { + await using var h = await TryCreateAsync(); + if (h == null) return; + var big = JsonSerializer.Serialize(new { blob = new string('x', 150_000) }); + Assert.True(await h.Start("Az Fan-Out #1", Batch(60, "t"), "Agg", big)); + + // A space and a '#' in the name: the run is keyed by its sanitized form. + Assert.True(await h.DriveUntilFinished(TableKeys.Sanitize("Az Fan-Out #1"), 60_000)); + var post = Assert.Single(h.Svc.PostExecs); + Assert.Equal(60, post.Lines.Length); + Assert.Equal(big, post.Parameters["ParametersJson"]); + Assert.Equal(60, h.Svc.Started.Distinct().Count()); + Assert.Equal(60, h.Svc.Started.Count); + } + + [Fact] + public async Task AChildHoldsItsParentsAggregation_UntilItFinishes() + { + await using var h = await TryCreateAsync(); + if (h == null) return; + var childGate = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + h.Svc.BeforeRun = async t => + { + var id = FakeOrchestrator.IdOf(t); + if (id == "p0") + { + var parent = (await h.Store.GetActiveRunsAsync("AzParent"))[0]; + var link = h.Svc.RegisterPendingChild(parent.RunKey, "AzChild")!.Value; + await h.Svc.StartFromBatchAsync("AzChild", Batch(2, "c"), 4, null, null, CancellationToken.None, + parentRunKey: link.ParentRunKey, childKey: link.ChildKey); + } + if (id.StartsWith('c')) await childGate.Task; + }; + Assert.True(await h.Start("AzParent", Batch(1, "p"), "Agg")); + + Assert.True(await h.DriveUntil(async () => await h.Store.GetRunByNameAsync("AzChild") != null && h.Svc.Started.Count >= 3, 30_000)); + await h.DriveUntil(() => Task.FromResult(false), 500); + Assert.Empty(h.Svc.PostExecs); + + childGate.SetResult(); + Assert.True(await h.DriveUntilFinished("AzParent", 30_000)); + Assert.Single(h.Svc.PostExecs); + } + + [Fact] + public async Task AConcurrencyLimit_HoldsOnARealBackend() + { + await using var h = await TryCreateAsync(poolSize: 6); + if (h == null) return; + h.Svc.HoldMs = 150; + Assert.True(await h.Start("AzCapped", Batch(12, "c"), maxConcurrency: 2)); + + Assert.True(await h.DriveUntilFinished("AzCapped", 60_000)); + Assert.Equal(2, h.Svc.MaxActive); + Assert.Equal(12, h.Svc.Started.Distinct().Count()); + } + + [Fact] + public async Task ASequentialStopOnFailureRun_StopsAndRecordsTheRestCancelled() + { + await using var h = await TryCreateAsync(); + if (h == null) return; + h.Svc.Body = t => FakeOrchestrator.IdOf(t) == "s1" ? throw new InvalidOperationException("down") : "{}"; + Assert.True(await h.Start("AzStop", Batch(4, "s"), "Agg", sequential: true, stopOnFailure: true)); + + Assert.True(await h.DriveUntilFinished("AzStop", 30_000)); + Assert.Equal(["s0", "s1"], h.Svc.Started); + var run = (await h.Store.GetRunByNameAsync("AzStop"))!; + Assert.Equal((1, 2, "CompletedWithErrors"), (run.Failed, run.Cancelled, run.Status)); + Assert.Single(h.Svc.PostExecs); + } + + [Fact] + public async Task StackedRunsOfOneName_AreFoundByName_AndCancelledTogether() + { + await using var h = await TryCreateAsync(); + if (h == null) return; + // A dash and a tilde-free suffix, so the by-name range has neighbours that share its prefix. + Assert.True(await h.Start("AzTwin", Batch(5, "a"))); + Assert.True(await h.Start("AzTwin", Batch(5, "b"))); + Assert.True(await h.Start("AzTwin-Other", Batch(5, "c"))); + Assert.False(await h.Start("AzTwin", Batch(1, "d"), allowCollision: false)); + + Assert.Equal(2, (await h.Store.GetActiveRunsAsync("AzTwin")).Count); + var (found, cancelled) = await h.Svc.CancelRunAsync("AzTwin"); + Assert.True(found); + Assert.Equal(10, cancelled); + Assert.Empty(await h.Store.GetActiveRunsAsync("AzTwin")); + Assert.Single(await h.Store.GetActiveRunsAsync("AzTwin-Other")); + } + + [Fact] + public async Task AClaimLeftByADeadProcess_IsTakenBackAfterItsLease() + { + await using var h = await TryCreateAsync(); + if (h == null) return; + Assert.True(await h.Start("AzOrphan", Batch(3, "o"))); + var run = (await h.Store.GetRunByNameAsync("AzOrphan"))!; + Assert.Equal(3, (await h.Store.ClaimAsync(run.RunKey, 3, "dead-host", TimeSpan.FromMilliseconds(1), false)).Count); + await Task.Delay(50); + + Assert.True(await h.DriveUntilFinished("AzOrphan", 30_000)); + Assert.All(await h.Store.GetTasksAsync(run.RunKey, 'D'), t => Assert.Equal(2, t.Attempt)); + } + + // ── the hardening guarantees, on a real backend ── + + private static readonly TimeSpan Lease = TimeSpan.FromMinutes(30); + + private static Task CreateRunAsync(WorkStore s, string name, int tasks, string? postExec = null) + { + var started = DateTime.UtcNow; + return s.CreateRunAsync(new RunHeader + { + RunKey = WorkStore.RunKeyFor(name, started), + Name = name, + StartedUtc = started, + TaskScriptName = "Invoke-CraftTask", + PostExecFunctionName = postExec, + }, Enumerable.Range(0, tasks).Select(i => new WorkStore.NewTask($"t{i}", [])).ToList()); + } + + [Fact] + public async Task AFinishOfFortyNineTasksThatReachesTheBarrier_IsExactlyOneHundredEntities_AndIsAccepted() + { + await using var h = await TryCreateAsync(); + if (h == null) return; + var run = await CreateRunAsync(h.Store, "AzHundred", 49, "Agg"); + var claims = await h.Store.ClaimAsync(run.RunKey, 49, "w", Lease, false); + Assert.Equal(49, claims.Count); + + // 49 x (delete running + insert done) + the aggregation insert + the header = 100, the documented maximum. + var outcome = await h.Store.FinishAsync(run.RunKey, claims.Select(c => new WorkStore.Finish(c.Seq, "Completed", Owner: "w")).ToList()); + + Assert.True(outcome!.ReachedBarrier); + Assert.Equal(RunPhase.Aggregate, (await h.Store.GetRunAsync(run.RunKey))!.Phase); + } + + [Fact] + public async Task ARangeReadsPageSize_IsAPageNotALimit() + { + await using var h = await TryCreateAsync(); + if (h == null) return; + var run = await CreateRunAsync(h.Store, "AzPaged", 23); + + var rows = 0; + await foreach (var _ in h.Tables.QueryRowKeyRangeAsync($"{h.Prefix}Work", run.RunKey, "P|", "P}", maxPerPage: 5)) rows++; + + Assert.Equal(23, rows); + } + + [Fact] + public async Task TheInstanceLock_HoldsOnARealBackend() + { + await using var h = await TryCreateAsync(); + if (h == null) return; + + var race = await Task.WhenAll(h.Store.TryHoldInstanceLockAsync("racer-a", TimeSpan.FromSeconds(2)), + h.Store.TryHoldInstanceLockAsync("racer-b", TimeSpan.FromSeconds(2))); + Assert.Single(race, r => r.Held); + var winner = race.Single(r => r.Held).Holder!.Owner; + var loser = winner == "racer-a" ? "racer-b" : "racer-a"; + Assert.False((await h.Store.TryHoldInstanceLockAsync(loser, Lease)).Held); + + await Task.Delay(TimeSpan.FromSeconds(2.5)); + Assert.True((await h.Store.TryHoldInstanceLockAsync(loser, Lease)).Held); + await h.Store.ReleaseInstanceLockAsync(loser); + Assert.Null(await h.Store.GetInstanceLockAsync()); + } + + [Fact] + public async Task AStoppedProcesssClaims_AreTakenBackByTheLockHolder_OnARealBackend() + { + await using var h = await TryCreateAsync(); + if (h == null) return; + Assert.True(await h.Start("AzInherited", Batch(3, "i"))); + var run = (await h.Store.GetRunByNameAsync("AzInherited"))!; + Assert.Equal(3, (await h.Store.ClaimAsync(run.RunKey, 3, "old-host/7/aaaa", TimeSpan.FromDays(1), false)).Count); + + await h.Pump.AcquireLockAsync(CancellationToken.None); + Assert.True(await h.DriveUntilFinished("AzInherited", 30_000)); + Assert.All(await h.Store.GetTasksAsync(run.RunKey, 'D'), t => Assert.Equal(2, t.Attempt)); + } + + [Fact] + public async Task CrashesAtWriteBoundaries_AreRepaired_OnARealBackend() + { + FaultyTableStore? faulty = null; + await using var h = await TryCreateAsync(wrap: t => faulty = new FaultyTableStore(t)); + if (h == null) return; + h.Store.IndexRetries = [TimeSpan.Zero]; + + faulty!.FailUpsert = (table, _, rk) => table == $"{h.Prefix}Work" && rk == WorkStore.HeaderKey; + await Assert.ThrowsAsync(() => CreateRunAsync(h.Store, "AzHalfMade", 3)); + faulty.FailUpsert = null; + + var done = await CreateRunAsync(h.Store, "AzUnretired", 1); + var claim = await h.Store.ClaimAsync(done.RunKey, 1, "w", Lease, false); + faulty.FailDelete = (table, _) => table == $"{h.Prefix}Ready" || table == $"{h.Prefix}Names"; + await h.Store.FinishAsync(done.RunKey, [new WorkStore.Finish(claim[0].Seq, "Completed", Owner: "w")]); + faulty.FailDelete = null; + + var r = await h.Store.RepairIndexesAsync(TimeSpan.Zero); + + Assert.Equal((1, 1), (r.Removed, r.Retired)); + var active = 0; + await foreach (var _ in h.Tables.QueryPartitionAsync($"{h.Prefix}Names", "A")) active++; + Assert.Equal(0, active); + Assert.Equal(1, await h.Store.SweepFinishedAsync(TimeSpan.Zero)); + } + + [Fact] + public async Task ATaskFinishedTwiceInOneBatch_IsAppliedOnce_OnARealBackend() + { + await using var h = await TryCreateAsync(); + if (h == null) return; + var run = await CreateRunAsync(h.Store, "AzTwice", 2); + var claim = (await h.Store.ClaimAsync(run.RunKey, 1, "w", Lease, false))[0]; + + var outcome = await h.Store.FinishAsync(run.RunKey, + [new WorkStore.Finish(claim.Seq, "Completed", Owner: "w"), new WorkStore.Finish(claim.Seq, "Cancelled", Owner: "w")]); + + Assert.Equal(1, outcome!.Applied); + } +} diff --git a/tests/Craft.Tests/OrchestrationContractTests.cs b/tests/Craft.Tests/OrchestrationContractTests.cs new file mode 100644 index 0000000..78e1e70 --- /dev/null +++ b/tests/Craft.Tests/OrchestrationContractTests.cs @@ -0,0 +1,311 @@ +using System.Text.Json; +using Craft.Configuration; +using Microsoft.Extensions.Configuration; + +namespace Craft.Tests; + +/// +/// The orchestration contract as CIPP sees it, end to end through the real store, pump and JobManager with +/// only the PowerShell calls faked: a batch becomes tasks invoked with TaskJson; their output reaches +/// the PostExecution script as one JSON line each, with FunctionName and ParametersJson; a run +/// queued by a task holds its parent until it finishes; runs of one name stack up unless the caller asks +/// for no collisions. +/// +public class OrchestrationContractTests +{ + private static Task NewAsync() => OrchestrationHarness.CreateAsync(); + + private static string Batch(int n, string prefix = "t") => OrchestrationHarness.Batch(n, prefix); + + private static Task Start(OrchestrationHarness h, string name, string batch, string? postExec = null, + string? postParams = null, bool sequential = false, int priority = 4, bool allowCollision = true) => + h.Start(name, batch, postExec, postParams, sequential, priority, allowCollision); + + [Fact] + public async Task EveryTaskRunsOnce_WithItsBatchItemAsTaskJson_AndTheRunCompletes() + { + await using var h = await NewAsync(); + Assert.True(await Start(h, "FanOut", Batch(10))); + + Assert.True(await h.DriveUntilFinished("FanOut")); + var run = (await h.Store.GetRunByNameAsync("FanOut"))!; + Assert.Equal("Completed", run.Status); + Assert.Equal(10, run.Done); + Assert.Equal(Enumerable.Range(0, 10).Select(i => $"t{i}").Order(), + h.Svc.Tasks.Select(t => t["TenantFilter"].ToString()!).Order()); + Assert.Empty(await ReadyNames(h)); + } + + [Fact] + public async Task PostExecution_RunsOnceAfterEveryTask_WithOneLinePerTaskOutput_AndItsParameters() + { + await using var h = await NewAsync(); + var bigParameters = JsonSerializer.Serialize(new { blob = new string('x', 100_000) }); + Assert.True(await Start(h, "WithPost", Batch(5), "AuditLogs", bigParameters)); + + Assert.True(await h.DriveUntilFinished("WithPost")); + var post = Assert.Single(h.Svc.PostExecs); + Assert.Equal("AuditLogs", post.Parameters["FunctionName"]); + Assert.Equal(bigParameters, post.Parameters["ParametersJson"]); + Assert.Equal(Enumerable.Range(0, 5).Select(i => $"t{i}").Order(), + post.Lines.Select(l => JsonDocument.Parse(l).RootElement.GetProperty("tenant").GetString()!).Order()); + Assert.Equal(5, h.Svc.Tasks.Count); + Assert.Equal("Completed", (await h.Store.GetRunByNameAsync("WithPost"))!.PostExecStatus); + } + + [Fact] + public async Task AFailingTask_IsRecorded_AndTheRunStillAggregatesAndFinishesWithErrors() + { + await using var h = await NewAsync(); + h.Svc.Body = t => t["TenantFilter"].ToString() == "t1" ? throw new InvalidOperationException("boom") : "{}"; + Assert.True(await Start(h, "Partial", Batch(3), "Agg")); + + Assert.True(await h.DriveUntilFinished("Partial")); + var run = (await h.Store.GetRunByNameAsync("Partial"))!; + Assert.Equal("CompletedWithErrors", run.Status); + Assert.Equal(1, run.Failed); + Assert.Single(h.Svc.PostExecs); + var failed = Assert.Single(await h.Store.GetTasksAsync(run.RunKey, 'D'), t => t.Status == "Failed"); + Assert.Equal("boom", failed.LastError); + } + + [Fact] + public async Task APostExecutionThatKeepsFailing_IsRetried_ThenTheRunFinishes() + { + await using var h = await NewAsync(); + h.Svc.PostExecBody = () => throw new InvalidOperationException("aggregate down"); + Assert.True(await Start(h, "PostFails", Batch(2), "Agg")); + + // A failed aggregation is handed back to pending, and the pump skips a run whose counts have not + // moved for a while, so the retries are spaced out; drive with that backoff out of the way. + Assert.True(await h.DriveUntil(async () => + { + h.Pump.ForgetBackoff(); + return await h.Store.GetRunByNameAsync("PostFails") is { IsFinished: true }; + })); + Assert.Equal(new CraftSettings().Orchestrator.MaxRetries, h.Svc.PostExecs.Count); + Assert.Equal("Failed", (await h.Store.GetRunByNameAsync("PostFails"))!.PostExecStatus); + } + + [Fact] + public async Task WithoutCollisions_ARunNameStillGoing_IsNotStartedAgain_AndItsBatchFileIsStillDeleted() + { + await using var h = await NewAsync(); + Assert.True(await Start(h, "Recurring", Batch(2), allowCollision: false)); + + var file = Path.Combine(Path.GetTempPath(), $"craft-test-{Guid.NewGuid():N}.jsonl"); + await File.WriteAllTextAsync(file, "{\"Name\":\"Job\",\"TenantFilter\":\"x\"}\n"); + Assert.False(await h.Svc.StartFromBatchAsync("Recurring", "", 4, null, null, CancellationToken.None, batchFilePath: file, + allowCollision: false)); + Assert.False(File.Exists(file)); + Assert.Single(await ReadyNames(h)); + + Assert.True(await h.DriveUntilFinished("Recurring")); + Assert.True(await Start(h, "Recurring", Batch(1), allowCollision: false)); + } + + /// + /// Callers append a queue id to a run's name so their queue page can find each outing. Without collisions, + /// Cache-{queueId} must still collide with yesterday's Cache-{otherQueueId} and with plain + /// Cache, or the flag never skips anything. + /// + [Fact] + public async Task WithoutCollisions_OutingsThatCarryAQueueId_CollideWithEachOther_AndWithThePlainName() + { + await using var h = await NewAsync(); + var first = $"Cache-{Guid.NewGuid()}"; + Assert.True(await Start(h, first, Batch(1), allowCollision: false)); + + Assert.False(await Start(h, $"Cache-{Guid.NewGuid()}", Batch(1), allowCollision: false)); + Assert.False(await Start(h, "Cache", Batch(1), allowCollision: false)); + Assert.True(await Start(h, $"CacheMore-{Guid.NewGuid()}", Batch(1), allowCollision: false)); + Assert.True(await Start(h, "Cache-weekly", Batch(1), allowCollision: false)); + Assert.True(await Start(h, $"Cache-{Guid.NewGuid()}", Batch(1))); + + Assert.True(await h.DriveUntilAllFinished()); + Assert.True(await Start(h, $"Cache-{Guid.NewGuid()}", Batch(1), allowCollision: false)); + } + + [Fact] + public async Task ANameSuffixThatIsNotAQueueId_IsItsOwnName_ForCollisions() + { + await using var h = await NewAsync(); + Assert.True(await Start(h, "Report-weekly", Batch(1), allowCollision: false)); + + Assert.True(await Start(h, "Report", Batch(1), allowCollision: false)); + Assert.False(await Start(h, $"Report-{Guid.NewGuid()}", Batch(1), allowCollision: false)); + } + + [Fact] + public async Task ByDefault_RunsOfOneNameStackUp_AndEachRunsEveryTask() + { + await using var h = await NewAsync(); + Assert.True(await Start(h, "Stacked", Batch(3), "Agg")); + Assert.True(await Start(h, "Stacked", Batch(3), "Agg")); + Assert.Equal(["Stacked", "Stacked"], await ReadyNames(h)); + + Assert.True(await h.DriveUntil(async () => (await ReadyNames(h)).Count == 0)); + Assert.Equal(6, h.Svc.Tasks.Count); + Assert.Equal(2, h.Svc.PostExecs.Count); + Assert.All(h.Svc.PostExecs, p => Assert.Equal(3, p.Lines.Length)); + // Same task ids in both runs, so the jobs share a display name; each must still be its own job, or + // the pump would track (and renew the lease of) only one of the two claims. + Assert.Equal(2, h.Jobs.GetJobs().Count(j => j.Name == "Stacked-Job_t0")); + } + + [Fact] + public async Task ACollisionFreeStart_IsSkippedWhileAStackedRunOfThatNameIsGoing() + { + await using var h = await NewAsync(); + Assert.True(await Start(h, "Mixed", Batch(1))); + Assert.True(await Start(h, "Mixed", Batch(1))); + Assert.False(await Start(h, "Mixed", Batch(1), allowCollision: false)); + } + + [Fact] + public async Task AChildFindsItsExactParent_ByRunKey_WhenRunsOfThatNameOverlap() + { + await using var h = await NewAsync(); + Assert.True(await Start(h, "Parent", Batch(1, "older"), "Agg")); + Assert.True(await Start(h, "Parent", Batch(1, "newer"), "Agg")); + var outings = await h.Store.GetActiveRunsAsync("Parent"); + var (older, newest) = (outings[0], outings[1]); + + var link = h.Svc.RegisterPendingChild(older.RunKey, "Child"); + Assert.Equal(older.RunKey, link!.Value.ParentRunKey); + Assert.Equal(2, (await h.Store.GetRunAsync(older.RunKey))!.Total); + Assert.Equal(1, (await h.Store.GetRunAsync(newest.RunKey))!.Total); + + // An older caller passing only the name gets the newest outing. + Assert.Equal(newest.RunKey, h.Svc.RegisterPendingChild("Parent", "Other")!.Value.ParentRunKey); + // A run re-queueing itself is never its own child, by key or by name. + Assert.Null(h.Svc.RegisterPendingChild(older.RunKey, "Parent")); + } + + [Fact] + public async Task CancellingByName_CancelsEveryRunOfThatName() + { + await using var h = await NewAsync(); + Assert.True(await Start(h, "Twin", Batch(4))); + Assert.True(await Start(h, "Twin", Batch(4))); + + var (found, cancelled) = await h.Svc.CancelRunAsync("Twin"); + + Assert.True(found); + Assert.Equal(8, cancelled); + Assert.Empty(await h.Store.GetActiveRunsAsync("Twin")); + } + + [Fact] + public async Task AChildQueuedByATask_HoldsTheParentsPostExecution_UntilTheChildFinishes() + { + await using var h = await NewAsync(); + var childGate = new TaskCompletionSource(); + h.Svc.Body = t => + { + if (t["TenantFilter"].ToString() == "p0") + { + var link = h.Svc.RegisterPendingChild("Parent", "Child"); + Assert.NotNull(link); + h.Svc.StartFromBatchAsync("Child", Batch(2, "c"), 4, null, null, CancellationToken.None, + parentRunKey: link!.Value.ParentRunKey, childKey: link.Value.ChildKey).GetAwaiter().GetResult(); + } + if (t["TenantFilter"].ToString()!.StartsWith('c')) childGate.Task.GetAwaiter().GetResult(); + return "{}"; + }; + Assert.True(await Start(h, "Parent", Batch(1, "p"), "Agg")); + + Assert.True(await h.DriveUntil(async () => await h.Store.GetRunByNameAsync("Child") != null && h.Svc.Tasks.Count >= 2)); + await h.DriveUntil(() => Task.FromResult(false), 300); + Assert.Empty(h.Svc.PostExecs); + Assert.False((await h.Store.GetRunByNameAsync("Parent"))!.IsFinished); + + childGate.SetResult(); + Assert.True(await h.DriveUntilFinished("Parent")); + Assert.True((await h.Store.GetRunByNameAsync("Child"))!.IsFinished); + Assert.Single(h.Svc.PostExecs); + } + + [Fact] + public async Task AChildThatNeverStarts_ReleasesItsParent() + { + await using var h = await NewAsync(); + h.Svc.Body = _ => + { + var link = h.Svc.RegisterPendingChild("Parent", "Stillborn")!.Value; + h.Svc.AbandonPendingChildAsync(link.ParentRunKey, link.ChildKey).GetAwaiter().GetResult(); + return "{}"; + }; + Assert.True(await Start(h, "Parent", Batch(1), "Agg")); + + Assert.True(await h.DriveUntilFinished("Parent")); + Assert.Single(h.Svc.PostExecs); + } + + [Fact] + public async Task ARunCannotBeItsOwnChild_NorAChildOfAFinishedRun() + { + await using var h = await NewAsync(); + Assert.True(await Start(h, "Solo", Batch(1))); + Assert.Null(h.Svc.RegisterPendingChild("Solo", "Solo")); + Assert.True(await h.DriveUntilFinished("Solo")); + Assert.Null(h.Svc.RegisterPendingChild("Solo", "FollowUp")); + } + + [Fact] + public async Task CancellingARun_CancelsWhatIsPending_AndTheRunStillFinalizes() + { + await using var h = await NewAsync(); + Assert.True(await Start(h, "Doomed", Batch(20))); + + var (found, cancelled) = await h.Svc.CancelRunAsync("Doomed"); + Assert.True(found); + Assert.Equal(20, cancelled); + var run = (await h.Store.GetRunByNameAsync("Doomed"))!; + Assert.True(run.IsFinished); + Assert.Equal(20, run.Cancelled); + Assert.Empty(h.Svc.Tasks); + } + + [Fact] + public async Task ASequentialRun_RunsEveryStepInOrder_OnOneWorker_PastAFailingStep() + { + await using var h = await NewAsync(); + h.Svc.Body = t => t["TenantFilter"].ToString() == "s2" ? throw new InvalidOperationException("step down") : "{}"; + Assert.True(await Start(h, "Steps", Batch(5, "s"), sequential: true)); + + Assert.True(await h.DriveUntilFinished("Steps")); + Assert.Equal(["s0", "s1", "s2", "s3", "s4"], h.Svc.Tasks.Select(t => t["TenantFilter"].ToString())); + Assert.Equal(1, h.Svc.Checkouts); + Assert.Equal(1, h.Svc.Reclaims); + Assert.Equal(1, (await h.Store.GetRunByNameAsync("Steps"))!.Failed); + } + + [Fact] + public async Task WorkClaimedByAProcessThatDied_IsTakenBackOnceItsLeaseLapses() + { + await using var h = await NewAsync(); + Assert.True(await Start(h, "Orphaned", Batch(3))); + var run = (await h.Store.GetRunByNameAsync("Orphaned"))!; + Assert.Equal(3, (await h.Store.ClaimAsync(run.RunKey, 3, "dead-host", TimeSpan.Zero, false)).Count); + + Assert.True(await h.DriveUntilFinished("Orphaned")); + Assert.Equal(3, h.Svc.Tasks.Count); + Assert.All(await h.Store.GetTasksAsync(run.RunKey, 'D'), t => Assert.Equal(2, t.Attempt)); + } + + [Fact] + public async Task ALowerBandRunsFirst_AndWithinABandTheOlderRun_NotTheAlphabeticallyFirst() + { + await using var h = await NewAsync(); + Assert.True(await Start(h, "Zulu", Batch(1, "z"), priority: 4)); + await Task.Delay(5); + Assert.True(await Start(h, "Alpha", Batch(1, "a"), priority: 4)); + Assert.True(await Start(h, "Background", Batch(1, "b"), priority: 9)); + Assert.True(await Start(h, "Urgent", Batch(1, "u"), priority: 1)); + + Assert.Equal(["Urgent", "Zulu", "Alpha", "Background"], await ReadyNames(h)); + } + + private static Task> ReadyNames(OrchestrationHarness h) => h.ReadyNamesAsync(); +} diff --git a/tests/Craft.Tests/OrchestrationCostTests.cs b/tests/Craft.Tests/OrchestrationCostTests.cs new file mode 100644 index 0000000..bc8fd4c --- /dev/null +++ b/tests/Craft.Tests/OrchestrationCostTests.cs @@ -0,0 +1,342 @@ +using Craft.Configuration; +using Craft.Orchestration; +using Craft.PowerShellHost; +using Craft.Storage; +using Microsoft.Extensions.Configuration; +using Microsoft.Extensions.Logging.Abstractions; +using Xunit.Abstractions; + +namespace Craft.Tests; + +/// +/// Storage cost pins. Each operation's table traffic is counted and held to a bound that does not grow with +/// the size of the queue, so a change that turns an O(batch) step into an O(queue) one fails here, long before +/// it shows up as a backlog on a large instance. Where a bound is a deliberate trade-off it says why. +/// +public class OrchestrationCostTests(ITestOutputHelper output) +{ + private const string Work = "OrchestratorWork", Ready = "OrchestratorReady", Names = "OrchestratorNames", + Finished = "OrchestratorFinished"; + + private static readonly TimeSpan Lease = TimeSpan.FromMinutes(30); + + private static (WorkStore Store, CountingTableStore Count) NewStore() + { + var count = new CountingTableStore(new MemoryTableStore()); + return (new WorkStore(NullLogger.Instance, new CraftSettings(), count), count); + } + + private static DateTime s_clock = new(2026, 10, 6, 0, 0, 0, DateTimeKind.Utc); + private static DateTime NextStart() => s_clock = s_clock.AddTicks(1); + + private static Task CreateAsync(WorkStore s, string name, int tasks, int maxConcurrency = 0, + string? postExec = null) + { + var started = NextStart(); + return s.CreateRunAsync(new RunHeader + { + RunKey = WorkStore.RunKeyFor(name, started), + Name = name, + StartedUtc = started, + TaskScriptName = "Invoke-CraftTask", + MaxConcurrency = maxConcurrency, + PostExecFunctionName = postExec, + }, Enumerable.Range(0, tasks).Select(i => new WorkStore.NewTask($"t{i}", new() { ["i"] = i })).ToList()); + } + + private static JobManager NewJobs(int poolSize = 4) + { + var settings = new CraftSettings(); + settings.Worker.BgPoolSize = poolSize; + var config = new ConfigurationBuilder().AddInMemoryCollection([]).Build(); + var repo = new ScriptRepository(NullLogger.Instance, settings); + var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); + var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); + return new JobManager(NullLogger.Instance, settings, limiter); + } + + private static WorkPump NewPump(WorkStore store, JobManager jobs, int batchSize = 4, int lowWater = 2) + { + var settings = new CraftSettings(); + settings.Worker.BgPoolSize = 4; + var config = new ConfigurationBuilder().AddInMemoryCollection(new Dictionary + { + ["JobQueueBatchSize"] = batchSize.ToString(System.Globalization.CultureInfo.InvariantCulture), + ["JobQueueLowWaterMark"] = lowWater.ToString(System.Globalization.CultureInfo.InvariantCulture), + }).Build(); + return new WorkPump(NullLogger.Instance, store, jobs, config, settings); + } + + [Theory] + [InlineData(10)] + [InlineData(5_000)] + public async Task CreatingARun_CostsAFixedNumberOfCalls_WhateverItsSize(int tasks) + { + var (s, c) = NewStore(); + await s.InitializeAsync(); + c.Reset(); + + await CreateAsync(s, "Sized", tasks); + + output.WriteLine($"work: {c.For(Work)} names: {c.For(Names)} ready: {c.For(Ready)}"); + Assert.Equal(2, c.For(Work).BatchUpserts); // payload rows, then pending rows (chunked by the backend) + Assert.Equal(1, c.For(Work).Upserts); // the header, last + Assert.Equal(1, c.For(Work).PointReads); // the created run read back + Assert.Equal(2, c.For(Names).Upserts); // latest-by-name and the active-outing row + Assert.Equal(1, c.For(Ready).Upserts); + } + + [Fact] + public async Task Claiming_ReadsOnlyTheRowsItClaims_FromAnyRunSize() + { + var (s, c) = NewStore(); + var run = await CreateAsync(s, "Big", 10_000); + c.Reset(); + + var claims = await s.ClaimAsync(run.RunKey, 49, "w", Lease, reclaimExpired: false); + + Assert.Equal(49, claims.Count); + var w = c.For(Work); + output.WriteLine($"plain claim: {w}"); + Assert.Equal((1, 49, 1, 0), (w.Queries, w.Rows, w.Submits, w.PointReads)); + + c.Reset(); + await s.ClaimAsync(run.RunKey, 49, "w", Lease, reclaimExpired: true); + w = c.For(Work); + output.WriteLine($"claim with lapsed-lease check: {w}"); + Assert.Equal(2, w.Queries); // running range, then pending range + Assert.Equal(49 + 49, w.Rows); // the 49 live claims are read to find lapsed ones + Assert.Equal(1, w.Submits); + } + + [Fact] + public async Task ARunAtItsConcurrencyLimit_CostsNoStorageReads_AndThePumpNeverHoldsMoreThanTheLimit() + { + var (s, c) = NewStore(); + await CreateAsync(s, "Capped", 10_000, maxConcurrency: 3); + var pump = NewPump(s, NewJobs(), batchSize: 49, lowWater: 1_000); + + Assert.Equal(3, await pump.RefillAsync(CancellationToken.None)); + c.Reset(); + for (var i = 0; i < 5; i++) Assert.Equal(0, await pump.RefillAsync(CancellationToken.None)); + + output.WriteLine($"5 refills at the limit: work {c.For(Work)} ready {c.For(Ready)}"); + Assert.Equal((0, 0, 0), (c.For(Work).PointReads, c.For(Work).Queries, c.For(Work).Submits)); + Assert.Equal(5, c.For(Ready).Queries); + } + + [Fact] + public async Task ASequentialStep_FinishesAndClaimsTheNext_InOneTransaction_FromAnyRunSize() + { + var (s, c) = NewStore(); + var started = NextStart(); + var run = await s.CreateRunAsync(new RunHeader + { + RunKey = WorkStore.RunKeyFor("Steps", started), + Name = "Steps", + StartedUtc = started, + TaskScriptName = "Invoke-CraftTask", + Sequential = true, + }, Enumerable.Range(0, 10_000).Select(i => new WorkStore.NewTask($"t{i}", new() { ["i"] = i })).ToList()); + var step = (await s.ClaimSequentialAsync(run.RunKey, "w", Lease))!; + c.Reset(); + + var r = await s.FinishStepAsync(run.RunKey, new WorkStore.Finish(step.Seq, "Completed", Owner: "w"), "w", Lease); + + Assert.NotNull(r.Payload); + var w = c.For(Work); + output.WriteLine($"step: work {w} ready {c.For(Ready)}"); + Assert.Equal(3, w.PointReads); // header and step (with the next-step range, read together), payload + Assert.Equal((1, 1, 0), (w.Queries, w.Rows, w.UnboundedRanges)); + Assert.Equal(1, w.Submits); + Assert.Equal(1, c.For(Ready).Upserts); + } + + [Fact] + public async Task FinishingABatch_IsOneTransaction_AndOneReadyUpdate() + { + var (s, c) = NewStore(); + var run = await CreateAsync(s, "Fin", 10_000); + var claims = await s.ClaimAsync(run.RunKey, 49, "w", Lease, false); + c.Reset(); + + await s.FinishAsync(run.RunKey, claims.Select(x => new WorkStore.Finish(x.Seq, "Completed", Owner: "w")).ToList()); + + var w = c.For(Work); + output.WriteLine($"finish 49: {w} ready: {c.For(Ready)}"); + Assert.Equal(1, w.Submits); + Assert.Equal(49 + 2, w.PointReads); // each claimed row, the header before and after + Assert.Equal(0, w.Queries); + Assert.Equal(1, c.For(Ready).Upserts); + } + + [Fact] + public async Task ConcurrentFinishesOfOneRun_AreCoalescedIntoFewTransactions() + { + var (s, c) = NewStore(); + var run = await CreateAsync(s, "Coalesce", 100); + var claims = await s.ClaimAsync(run.RunKey, 40, "w", Lease, false); + var batcher = new FinishBatcher(s, NullLogger.Instance, TimeSpan.FromMilliseconds(30)); + c.Reset(); + + await Task.WhenAll(claims.Select(x => Task.Run(() => batcher.FinishAsync(run.RunKey, new WorkStore.Finish(x.Seq, "Completed", Owner: "w"))))); + + output.WriteLine($"40 concurrent finishes: {c.For(Work)}"); + Assert.InRange(c.For(Work).Submits, 1, 3); + Assert.Equal(40, (await s.GetRunAsync(run.RunKey))!.Done); + } + + [Fact] + public async Task TheStatusSnapshot_ReadsTheReadyListOnce_AndOnlyTheHeadOfTheQueue() + { + var (s, c) = NewStore(); + for (var i = 0; i < 5_000; i++) await CreateAsync(s, $"Run{i}", 2); + var reader = new JobQueueStatusReader(NullLogger.Instance, NewJobs(), s); + c.Reset(); + + var snap = (await reader.GetAsync())!; + + Assert.Equal(10_000, snap.Total); + output.WriteLine($"snapshot over 5,000 runs: ready {c.For(Ready)} work {c.For(Work)}"); + Assert.Equal((1, 5_000, 5), (c.For(Ready).Queries, c.For(Ready).Rows, c.For(Ready).Pages)); + Assert.InRange(c.For(Work).Queries, 0, 50); // HeadRuns + Assert.InRange(c.For(Work).Rows, 0, 100); + Assert.Equal(0, c.For(Work).PointReads); + } + + [Theory] + [InlineData(10)] + [InlineData(3_000)] + public async Task ARefill_CostsTheSame_WhateverTheBacklogBehindTheHead(int runs) + { + var (s, c) = NewStore(); + for (var i = 0; i < runs; i++) await CreateAsync(s, $"Run{i}", 5); + var pump = NewPump(s, NewJobs()); + c.Reset(); + + var claimed = await pump.RefillAsync(CancellationToken.None); + + Assert.Equal(4, claimed); + output.WriteLine($"refill with {runs} runs queued: ready {c.For(Ready)} work {c.For(Work)}"); + Assert.Equal((1, 1), (c.For(Ready).Queries, c.For(Ready).Pages)); + Assert.Equal(1, c.For(Work).PointReads); // the head run's header + Assert.Equal(2, c.For(Work).Queries); // its running range (first visit) and pending range + Assert.Equal(1, c.For(Work).Submits); + } + + [Fact] + public async Task LookingUpActiveRunsByName_IsOneRangeQuery_PlusOneReadPerOuting() + { + var (s, c) = NewStore(); + for (var i = 0; i < 1_000; i++) await CreateAsync(s, $"Other{i}", 1); + for (var i = 0; i < 3; i++) await CreateAsync(s, "Wanted", 1); + c.Reset(); + + Assert.Equal(3, (await s.GetActiveRunsAsync("Wanted")).Count); + Assert.Equal((1, 3), (c.For(Names).Queries, c.For(Names).Rows)); + Assert.Equal(3, c.For(Work).PointReads); + } + + [Fact] + public async Task TheRetentionSweep_ReadsOnlyRunsPastTheCutoff() + { + var (s, c) = NewStore(); + var old = new List(); + for (var i = 0; i < 20; i++) old.Add(await CreateAsync(s, $"Done{i}", 1)); + foreach (var run in old) + { + var claim = await s.ClaimAsync(run.RunKey, 1, "w", Lease, false); + await s.FinishAsync(run.RunKey, [new WorkStore.Finish(claim[0].Seq, "Completed", Owner: "w")]); + } + c.Reset(); + + Assert.Equal(0, await s.SweepFinishedAsync(TimeSpan.FromHours(1))); + Assert.Equal((1, 0), (c.For(Finished).Queries, c.For(Finished).Rows)); + + Assert.Equal(20, await s.SweepFinishedAsync(TimeSpan.Zero)); + } + + [Fact] + public async Task AWholeFanOut_CostsAFewTableCallsPerTask_EndToEnd() + { + var count = new CountingTableStore(new MemoryTableStore()); + await using var h = await OrchestrationHarness.CreateAsync(poolSize: 8, tables: count); + count.Reset(); + Assert.True(await h.Start("Throughput", OrchestrationHarness.Batch(1_000, "t"), "Agg")); + + Assert.True(await h.DriveUntilFinished("Throughput", 60_000)); + var t = count.Total(); + output.WriteLine($"1,000-task fan-out end to end: {t} work {count.For(Work)}"); + // Measured: see the output line. Bounds leave headroom for scheduling noise, not for a new per-task call. + Assert.InRange(t.Submits / 1000.0, 0, PerTaskSubmits); + Assert.InRange(t.PointReads / 1000.0, 0, PerTaskPointReads); + Assert.InRange(t.Queries / 1000.0, 0, PerTaskQueries); + Assert.InRange((t.Upserts + t.BatchUpserts) / 1000.0, 0, PerTaskWrites); + } + + [Fact] + public async Task ManySmallRuns_CostAFewTableCallsPerRun_EndToEnd() + { + var count = new CountingTableStore(new MemoryTableStore()); + await using var h = await OrchestrationHarness.CreateAsync(poolSize: 8, tables: count); + count.Reset(); + for (var i = 0; i < 300; i++) Assert.True(await h.Start($"Small{i}", OrchestrationHarness.Batch(1, $"s{i}_"))); + + Assert.True(await h.DriveUntilAllFinished(60_000)); + var t = count.Total(); + output.WriteLine($"300 single-task runs end to end: {t}"); + Assert.InRange((t.Submits + t.PointReads + t.Queries + t.Upserts + t.BatchUpserts + t.Deletes) / 300.0, 0, PerSmallRunCalls); + } + + // Measured 2026-10-06 (three runs, steady): per task 0.25 transactions (claims and finishes batched), + // 3.66 point reads (header and payload at dispatch, the claimed row and header at finish), 0.39 queries and + // 1.13 writes (each task's result row for the aggregation, plus a Ready update per finish batch); per + // single-task run 21.5 calls in all. Bounds are about 1.5x. + private const double PerTaskSubmits = 0.4, PerTaskPointReads = 5.5, PerTaskQueries = 0.6, PerTaskWrites = 1.7, + PerSmallRunCalls = 32; + + /// + /// The case that wedges a big instance: thousands of runs that have nothing to claim (waiting on children, + /// or on claims held elsewhere) sit ahead of one run that does. The pump gives each run it looks at a + /// storage read from a small per-refill budget, so if the blocked runs keep spending that budget the run + /// at the back is starved. Simulates ten minutes at one refill a second and requires the run at the back to + /// be reached quickly and then claimed from on nearly every refill. + /// + [Fact] + public async Task TheRunAtTheBackOfAQueueOfBlockedRuns_IsReached_AndKeepsBeingClaimed() + { + var (s, c) = NewStore(); + for (var i = 0; i < 2_000; i++) + { + var blocked = await CreateAsync(s, $"Blocked{i}", 1); + await s.ClaimAsync(blocked.RunKey, 1, "elsewhere", TimeSpan.FromDays(1), false); + } + await CreateAsync(s, "Tail", 100_000); + + var jobs = NewJobs(); + jobs.SetWorkResolver((_, _) => Task.FromResult?>(null)); + _ = Task.Run(() => jobs.StartAsync(CancellationToken.None)); + var pump = NewPump(s, jobs); + var now = new DateTime(2026, 10, 6, 0, 0, 0, DateTimeKind.Utc); + pump.Clock = () => now; + + var first = -1; + var claimedRefills = new List(); + c.Reset(); + for (var second = 0; second < 600; second++) + { + now = now.AddSeconds(1); + var claimed = await pump.RefillAsync(CancellationToken.None); + if (claimed > 0 && first < 0) first = second; + claimedRefills.Add(claimed > 0); + for (var spin = 0; spin < 200 && jobs.QueuedCount > 0; spin++) await Task.Delay(1); + } + await jobs.StopAsync(CancellationToken.None); + + var lateShare = claimedRefills.Skip(300).Count(x => x) / 300.0; + output.WriteLine($"first claim at refill {first}; claimed in {lateShare:P0} of the last 300 refills; work {c.For(Work)} ready {c.For(Ready)}"); + Assert.InRange(first, 0, 70); + Assert.InRange(c.For(Ready).Pages, 0, 600 * 3); // the whole Ready list each refill, 1,000 rows a page + Assert.True(lateShare >= 0.9, $"the run at the back was claimed in only {lateShare:P0} of the last 300 refills"); + } +} diff --git a/tests/Craft.Tests/OrchestrationHarness.cs b/tests/Craft.Tests/OrchestrationHarness.cs new file mode 100644 index 0000000..c8a281a --- /dev/null +++ b/tests/Craft.Tests/OrchestrationHarness.cs @@ -0,0 +1,175 @@ +using System.Collections.Concurrent; +using System.Text.Json; +using Craft.Configuration; +using Craft.Hosting; +using Craft.Orchestration; +using Craft.PowerShellHost; +using Craft.Storage; +using Microsoft.Extensions.Configuration; +using Microsoft.Extensions.Logging; +using Microsoft.Extensions.Logging.Abstractions; + +namespace Craft.Tests; + +/// +/// The real orchestrator with only its PowerShell calls faked. Tasks are batch items; each records itself, can +/// hold or wait on a gate, and returns {"tenant": TenantFilter} unless says otherwise. +/// Tracks the start order and the peak number of tasks running at once. +/// +internal sealed class FakeOrchestrator(JobManager jobs, WorkStore store, ResultStore results, IConfiguration config, + CraftSettings settings, ILogger logger) + : OrchestratorService(logger, null!, null!, jobs, store, results, config, settings) +{ + public const string PostExecScript = "Invoke-CraftPostExecution"; + + public readonly ConcurrentQueue> Tasks = new(); + public readonly ConcurrentQueue Started = new(); + + /// The job name each task ran under, as worker stats show it. + public readonly ConcurrentQueue RanAs = new(); + public readonly ConcurrentQueue<(Dictionary Parameters, string[] Lines)> PostExecs = new(); + + /// The task's output; throw to fail it. + public Func, string>? Body; + + /// Awaited before the task does anything else (a gate, or a hold for one task). + public Func, Task>? BeforeRun; + + public Func? PostExecBody; + public int HoldMs; + public int Checkouts, Reclaims; + + private int _active, _maxActive; + public int MaxActive => Volatile.Read(ref _maxActive); + + public static string IdOf(Dictionary task) => task["TenantFilter"].ToString()!; + + internal override string? FindScript(string name) => name; + + internal override async Task RunScriptAsync(string path, Dictionary parameters, bool captureOutput, + PowerShellWorker? worker = null) + { + if (path == PostExecScript) + { + PostExecs.Enqueue((parameters, File.ReadAllLines((string)parameters["ResultsPath"]))); + if (PostExecBody != null) await PostExecBody(); + return string.Empty; + } + + var task = JsonSerializer.Deserialize>((string)parameters["TaskJson"])!; + Tasks.Enqueue(task); + Started.Enqueue(IdOf(task)); + RanAs.Enqueue(OperationContext.Current?.Function); + var now = Interlocked.Increment(ref _active); + for (var seen = _maxActive; now > seen; seen = _maxActive) + if (Interlocked.CompareExchange(ref _maxActive, now, seen) == seen) break; + try + { + if (BeforeRun != null) await BeforeRun(task); + if (HoldMs > 0) await Task.Delay(HoldMs); + var output = Body?.Invoke(task) ?? JsonSerializer.Serialize(new { tenant = IdOf(task) }); + return captureOutput ? output : string.Empty; + } + finally + { + Interlocked.Decrement(ref _active); + } + } + + internal override PowerShellWorker? CheckoutSequentialWorker(CancellationToken ct) { Interlocked.Increment(ref Checkouts); return null; } + internal override void ReclaimSequentialWorker(PowerShellWorker? worker, bool faulted) => Interlocked.Increment(ref Reclaims); +} + +/// Store, pump and JobManager wired as in the host, over an in-memory table store, driven by hand. +internal sealed class OrchestrationHarness : IAsyncDisposable +{ + public required FakeOrchestrator Svc { get; init; } + public required WorkStore Store { get; init; } + public required WorkPump Pump { get; init; } + public required JobManager Jobs { get; init; } + public required ICraftTableStore Tables { get; init; } + public required CapturingLogger Log { get; init; } + public required string Prefix { get; init; } + + public static async Task CreateAsync(int poolSize = 4, ICraftTableStore? tables = null, + Action? configure = null) + { + var settings = new CraftSettings(); + settings.Worker.BgPoolSize = poolSize; + configure?.Invoke(settings); + // The limiter's starting concurrency otherwise follows the CPU count, which differs between machines. + var config = new ConfigurationBuilder().AddInMemoryCollection(new Dictionary + { + ["BackgroundBaseConcurrency"] = poolSize.ToString(System.Globalization.CultureInfo.InvariantCulture), + }).Build(); + var repo = new ScriptRepository(NullLogger.Instance, settings); + var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); + var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); + var jobs = new JobManager(NullLogger.Instance, settings, limiter); + tables ??= new MemoryTableStore(); + var store = new WorkStore(NullLogger.Instance, settings, tables); + var results = new ResultStore(NullLogger.Instance, settings, tables); + var log = new CapturingLogger(); + var svc = new FakeOrchestrator(jobs, store, results, config, settings, log); + await svc.ResumeInterruptedRunsAsync(CancellationToken.None); + var pump = new WorkPump(NullLogger.Instance, store, jobs, config, settings, svc); + _ = Task.Run(() => jobs.StartAsync(CancellationToken.None)); + return new OrchestrationHarness + { + Svc = svc, + Store = store, + Pump = pump, + Jobs = jobs, + Tables = tables, + Log = log, + Prefix = settings.Orchestrator.TablePrefix, + }; + } + + public static string Batch(int n, string prefix = "t") => + JsonSerializer.Serialize(Enumerable.Range(0, n).Select(i => new { Name = "Job", TenantFilter = $"{prefix}{i}", N = i })); + + public Task Start(string name, string batch, string? postExec = null, string? postParams = null, + bool sequential = false, int priority = 4, bool allowCollision = true, int maxConcurrency = 0, bool stopOnFailure = false) => + Svc.StartFromBatchAsync(name, batch, priority, postExec, postParams, CancellationToken.None, sequential: sequential, + allowCollision: allowCollision, maxConcurrency: maxConcurrency, stopOnFailure: stopOnFailure); + + public async Task DriveUntil(Func> done, int timeoutMs = 10_000) + { + var deadline = Environment.TickCount64 + timeoutMs; + while (Environment.TickCount64 < deadline) + { + await Pump.RefillAsync(CancellationToken.None); + if (await done()) return true; + await Task.Delay(10); + } + return await done(); + } + + public Task DriveUntilFinished(string name, int timeoutMs = 10_000) => + DriveUntil(async () => await Store.GetRunByNameAsync(name) is { IsFinished: true }, timeoutMs); + + public Task DriveUntilAllFinished(int timeoutMs = 10_000) => + DriveUntil(async () => (await ReadyNamesAsync()).Count == 0, timeoutMs); + + public async Task> ReadyNamesAsync() + { + var names = new List(); + await foreach (var e in Store.ReadReadyAsync()) names.Add(e.Name); + return names; + } + + public async ValueTask DisposeAsync() => await Jobs.StopAsync(CancellationToken.None); +} + +/// Keeps every rendered log line (at any level) for tests that pin what operators and tooling read. +internal sealed class CapturingLogger : ILogger +{ + public readonly ConcurrentQueue<(LogLevel Level, string Message)> Lines = new(); + + public IDisposable? BeginScope(TState state) where TState : notnull => null; + public bool IsEnabled(LogLevel logLevel) => true; + + public void Log(LogLevel logLevel, EventId eventId, TState state, Exception? exception, + Func formatter) => Lines.Enqueue((logLevel, formatter(state, exception))); +} diff --git a/tests/Craft.Tests/OrchestrationLifecycleTests.cs b/tests/Craft.Tests/OrchestrationLifecycleTests.cs new file mode 100644 index 0000000..c7d8007 --- /dev/null +++ b/tests/Craft.Tests/OrchestrationLifecycleTests.cs @@ -0,0 +1,324 @@ +using System.Text.RegularExpressions; +using Craft.Configuration; +using Craft.Orchestration; +using Craft.PowerShellHost; +using Craft.Storage; +using Microsoft.Extensions.Configuration; +using Microsoft.Extensions.Logging; +using Microsoft.Extensions.Logging.Abstractions; + +namespace Craft.Tests; + +/// +/// Everything around a run that is not its scheduling mode: leases, shutdown, operator actions, startup cleanup, +/// result cleanup, and the log lines operators and the health tooling read. +/// +public class OrchestrationLifecycleTests +{ + private static string Batch(int n, string prefix) => OrchestrationHarness.Batch(n, prefix); + + private static JobManager NewJobs() + { + var settings = new CraftSettings(); + settings.Worker.BgPoolSize = 4; + var config = new ConfigurationBuilder().AddInMemoryCollection([]).Build(); + var repo = new ScriptRepository(NullLogger.Instance, settings); + var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); + var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); + return new JobManager(NullLogger.Instance, settings, limiter); + } + + private static (WorkStore Store, WorkPump Pump, JobManager Jobs) NewIdlePump(int leaseSeconds = 1800) + { + var settings = new CraftSettings(); + settings.Worker.BgPoolSize = 4; + var store = new WorkStore(NullLogger.Instance, settings, new MemoryTableStore()); + var jobs = NewJobs(); + var config = new ConfigurationBuilder().AddInMemoryCollection(new Dictionary + { + ["JobQueueLeaseSeconds"] = leaseSeconds.ToString(System.Globalization.CultureInfo.InvariantCulture), + }).Build(); + return (store, new WorkPump(NullLogger.Instance, store, jobs, config, settings), jobs); + } + + private static async Task CreateAsync(WorkStore s, string name, int tasks) + { + var started = DateTime.UtcNow; + return await s.CreateRunAsync(new RunHeader + { + RunKey = WorkStore.RunKeyFor(name, started), + Name = name, + StartedUtc = started, + TaskScriptName = "Invoke-CraftTask", + }, Enumerable.Range(0, tasks).Select(i => new WorkStore.NewTask($"t{i}", [])).ToList()); + } + + // ── leases and shutdown ── + + [Fact] + public async Task AClaimHeldPastTwoThirdsOfItsLease_IsRenewed_SoALongTaskNeverLosesIt() + { + var (store, pump, _) = NewIdlePump(leaseSeconds: 60); + var run = await CreateAsync(store, "Long", 2); + var now = DateTime.UtcNow; + pump.Clock = () => now; + Assert.Equal(2, await pump.RefillAsync(CancellationToken.None)); + var before = (await store.GetTasksAsync(run.RunKey, 'R')).Select(t => t.LeaseUntil!.Value).Min(); + + now = now.AddSeconds(30); // half way: not yet due + await Task.Delay(20); + await pump.RenewAsync(CancellationToken.None); + Assert.Equal(before, (await store.GetTasksAsync(run.RunKey, 'R')).Select(t => t.LeaseUntil!.Value).Min()); + + now = now.AddSeconds(15); // three quarters: renewed + await pump.RenewAsync(CancellationToken.None); + Assert.All(await store.GetTasksAsync(run.RunKey, 'R'), t => Assert.True(t.LeaseUntil > before)); + } + + [Fact] + public async Task OnShutdown_ClaimsThatNeverStarted_AreHandedBack_WithTheirAttemptRefunded() + { + var (store, pump, _) = NewIdlePump(); + var run = await CreateAsync(store, "Stopping", 6); + Assert.Equal(4, await pump.RefillAsync(CancellationToken.None)); + + await pump.StopAsync(CancellationToken.None); + + Assert.Empty(await store.GetTasksAsync(run.RunKey, 'R')); + var pending = await store.GetTasksAsync(run.RunKey, 'P'); + Assert.Equal(6, pending.Count); + Assert.All(pending, t => Assert.Equal(0, t.Attempt)); + } + + /// + /// The job manager stops after the pump, so it can still dispatch during shutdown. A claim handed back must + /// also be taken out of its queue, or this process runs the task while its successor runs it again. + /// + [Fact] + public async Task OnShutdown_AHandedBackClaim_IsNeverRunByThisProcess() + { + var (store, pump, jobs) = NewIdlePump(); + var ran = 0; + jobs.SetWorkResolver((_, _) => Task.FromResult?>(_ => + { + Interlocked.Increment(ref ran); + return Task.CompletedTask; + })); + await CreateAsync(store, "Handback", 4); + Assert.Equal(4, await pump.RefillAsync(CancellationToken.None)); + + await pump.StopAsync(CancellationToken.None); + _ = Task.Run(() => jobs.StartAsync(CancellationToken.None)); + await Task.Delay(500); + + Assert.Equal(0, ran); + await jobs.StopAsync(CancellationToken.None); + } + + /// + /// Depending on timing, the cancel lands while the job is still queued or just after the dispatcher has + /// dequeued it. The second case once dropped the state-writer notification: the claim was never finished, + /// lapsed half an hour later and ran after all. Either way the task must end up Cancelled. + /// + [Fact] + public async Task CancellingABufferedJob_RecordsItsTaskCancelled_SoTheClaimCannotLapseAndRunIt() + { + await using var h = await OrchestrationHarness.CreateAsync(poolSize: 1); + var gate = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + h.Svc.BeforeRun = t => FakeOrchestrator.IdOf(t) == "b0" ? gate.Task : Task.CompletedTask; + Assert.True(await h.Start("Buffered", Batch(3, "b"))); + var run = (await h.Store.GetRunByNameAsync("Buffered"))!; + + Assert.True(await h.DriveUntil(() => Task.FromResult(h.Jobs.GetJobs(status: "Queued").Count > 0)), "no job was ever buffered"); + var queued = h.Jobs.GetJobs(status: "Queued").First(); + Assert.True(h.Jobs.CancelJob(queued.Id)); + gate.SetResult(); + + var finished = await h.DriveUntilFinished("Buffered"); + var state = string.Join(",", (await h.Store.GetTasksAsync(run.RunKey)).Select(t => $"{t.TaskId}:{t.State}:{t.Status}:{t.Owner}")); + Assert.True(finished, $"run did not finish; started={string.Join(",", h.Svc.Started)} tasks={state} cancelled={queued.Name}"); + var done = await h.Store.GetTasksAsync(run.RunKey, 'D'); + Assert.Single(done, t => t.Status == "Cancelled"); + Assert.Equal(2, h.Svc.Started.Count); + } + + // ── operator actions ── + + [Fact] + public async Task Reprioritizing_MovesTheWholeRunToItsNewBand() + { + await using var h = await OrchestrationHarness.CreateAsync(); + Assert.True(await h.Start("First", Batch(2, "f"), priority: 4)); + Assert.True(await h.Start("Second", Batch(2, "s"), priority: 4)); + + Assert.True(await h.Svc.ReprioritizeRunAsync("Second", 1)); + + Assert.Equal(["Second", "First"], await h.ReadyNamesAsync()); + Assert.Equal(1, (await h.Store.GetRunByNameAsync("Second"))!.Priority); + } + + [Fact] + public async Task CancellingOneQueuedTaskByName_FindsItInWhicheverStackedRunHoldsIt() + { + await using var h = await OrchestrationHarness.CreateAsync(); + Assert.True(await h.Start("Twin", Batch(2, "a"))); + Assert.True(await h.Start("Twin", Batch(2, "b"))); + + Assert.True(await h.Svc.TryCancelQueuedTaskAsync("Twin", "Job_b1")); + Assert.False(await h.Svc.TryCancelQueuedTaskAsync("Twin", "Job_b1")); + Assert.False(await h.Svc.TryCancelQueuedTaskAsync("Twin", "Job_nope")); + } + + [Fact] + public async Task ClearingTheQueue_CancelsEveryPendingTaskOfEveryRun() + { + await using var h = await OrchestrationHarness.CreateAsync(); + Assert.True(await h.Start("A", Batch(3, "a"))); + Assert.True(await h.Start("B", Batch(4, "b"), priority: 9)); + + Assert.Equal(7, await h.Svc.ClearQueueAsync()); + Assert.Empty(await h.ReadyNamesAsync()); + } + + [Fact] + public async Task AReadyEntryWhoseRunIsGone_IsDroppedByThePump() + { + await using var h = await OrchestrationHarness.CreateAsync(); + Assert.True(await h.Start("Vanished", Batch(2, "v"))); + var run = (await h.Store.GetRunByNameAsync("Vanished"))!; + await h.Tables.DeletePartitionAsync("OrchestratorWork", run.RunKey); + + await h.Pump.RefillAsync(CancellationToken.None); + Assert.Empty(await h.ReadyNamesAsync()); + } + + // ── bulk cancel ── + + /// + /// Cancelling a big backlog while the pump is working. The pump once kept claiming the run's tasks during the + /// cancel (each cancelled page woke it), so the two fought over the same rows and the run header: the call took + /// tens of seconds, under-reported, and could give up on lost races. Now the pump leaves a run being + /// cancelled alone, and the cancel reuses the rows it has just read. + /// + [Fact] + public async Task CancellingABigBacklogWithThePumpRunning_CancelsEveryPendingTask_AndReportsTheTrueCount() + { + await using var h = await OrchestrationHarness.CreateAsync(poolSize: 4); + h.Svc.HoldMs = 200; + h.Svc.MarkRecoveryDone(); + await h.Pump.StartAsync(CancellationToken.None); + try + { + Assert.True(await h.Start("Backlog", Batch(5000, "b"))); + Assert.True(await h.DriveUntil(() => Task.FromResult(h.Svc.Started.Count >= 4))); + + var sw = System.Diagnostics.Stopwatch.StartNew(); + var (found, cancelled) = await h.Svc.CancelRunAsync("Backlog"); + sw.Stop(); + + var run = (await h.Store.GetRunByNameAsync("Backlog"))!; + Assert.True(found); + Assert.Equal(run.Cancelled, cancelled); + Assert.Empty(await h.Store.GetTasksAsync(run.RunKey, 'P')); + Assert.True(cancelled >= 5000 - h.Svc.Started.Count - 8, $"cancelled {cancelled}, started {h.Svc.Started.Count}"); + Assert.True(sw.Elapsed < TimeSpan.FromSeconds(20), $"cancelling 5,000 took {sw.Elapsed.TotalSeconds:F1}s"); + } + finally + { + await h.Pump.StopAsync(CancellationToken.None); + } + } + + [Fact] + public async Task ACancelInterruptedAfterItsFlag_IsFinishedByThePump() + { + await using var h = await OrchestrationHarness.CreateAsync(poolSize: 4); + Assert.True(await h.Start("Interrupted", Batch(300, "i"), "Agg")); + var run = (await h.Store.GetRunByNameAsync("Interrupted"))!; + await h.Store.RequestCancelAsync(run.RunKey); // the process died before cancelling anything + + Assert.True(await h.DriveUntilFinished("Interrupted", 30_000)); + var after = (await h.Store.GetRunByNameAsync("Interrupted"))!; + Assert.Equal(300, after.Cancelled); + Assert.Empty(h.Svc.Started); + Assert.Single(h.Svc.PostExecs); + } + + [Fact] + public async Task ABulkCancel_ReusesTheRowsItReads_RatherThanReadingEachAgain() + { + var count = new CountingTableStore(new MemoryTableStore()); + await using var h = await OrchestrationHarness.CreateAsync(tables: count); + Assert.True(await h.Start("Cheap", Batch(4900, "c"))); + count.Reset(); + + Assert.Equal(4900, (await h.Svc.CancelRunAsync("Cheap")).cancelledCount); + + // 100 pages: the header before and after each, never one read per cancelled task. + Assert.InRange(count.For("OrchestratorWork").PointReads, 0, 400); + } + + // ── startup and cleanup ── + + [Fact] + public async Task Startup_DropsThePreviousDesignsTables() + { + var tables = new MemoryTableStore(); + await using var h = await OrchestrationHarness.CreateAsync(tables: tables); + + Assert.Equal(["OrchestratorQueue", "OrchestratorQueueIndex", "OrchestratorTasks", "OrchestratorRuns", "OrchestratorResults"], + tables.DroppedTables); + } + + [Fact] + public async Task ACompletedRunsResults_AreDeleted_OnceItsAggregationHasReadThem() + { + var tables = new MemoryTableStore(); + await using var h = await OrchestrationHarness.CreateAsync(tables: tables); + Assert.True(await h.Start("Results", Batch(5, "r"), "Agg")); + + Assert.True(await h.DriveUntilFinished("Results")); + Assert.Equal(5, Assert.Single(h.Svc.PostExecs).Lines.Length); + Assert.Empty(tables.All("OrchestratorTaskResults")); + } + + // ── what operators and tooling read ── + + [Fact] + public async Task TheLogLines_HealthToolingParses_KeepTheirShape() + { + await using var h = await OrchestrationHarness.CreateAsync(poolSize: 1); + var gate = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + h.Svc.BeforeRun = t => FakeOrchestrator.IdOf(t) == "l0" ? gate.Task : Task.CompletedTask; + h.Svc.Body = t => FakeOrchestrator.IdOf(t) == "l2" ? throw new InvalidOperationException("boom") : "{}"; + Assert.True(await h.Start("Logged", Batch(3, "l"), "Agg", priority: 6)); + Assert.True(await h.DriveUntil(() => Task.FromResult(h.Svc.Started.Contains("l0")))); + await h.Svc.LogRunStatusAsync(CancellationToken.None); + gate.SetResult(); + Assert.True(await h.DriveUntilFinished("Logged")); + + var lines = h.Log.Lines.Select(l => l.Message).ToList(); + // The patterns Get-CraftInstanceStatus.ps1 matches; change them together or not at all. + Assert.Contains(lines, l => Regex.IsMatch(l, @"\] Run Logged created with 3 tasks at P6")); + Assert.Contains(lines, l => Regex.IsMatch(l, @"\] Run Logged T\+[\d.]+min: \d+/3 done 1 running 2 pending 0 failed")); + Assert.Contains(lines, l => l.Contains("Dispatching PostExecution") && Regex.IsMatch(l, @"for run Logged\b")); + Assert.Contains(lines, l => Regex.IsMatch(l, @"\] Run Logged finalized: CompletedWithErrors \(2/1/0/3\)")); + Assert.Contains(h.Log.Lines, l => l.Level == LogLevel.Debug && l.Message.StartsWith("[Scheduler] Task completed: ", StringComparison.Ordinal)); + } + + [Fact] + public async Task APostExecutionThatGivesUp_SaysSoInTheLineTheToolingMatches() + { + await using var h = await OrchestrationHarness.CreateAsync(); + h.Svc.PostExecBody = () => throw new InvalidOperationException("aggregate down"); + Assert.True(await h.Start("GivesUp", Batch(1, "g"), "Agg")); + + Assert.True(await h.DriveUntil(async () => + { + h.Pump.ForgetBackoff(); + return await h.Store.GetRunByNameAsync("GivesUp") is { IsFinished: true }; + })); + Assert.Contains(h.Log.Lines, l => l.Message.Contains("giving up and cleaning up results") + && Regex.IsMatch(l.Message, @"PostExecution for GivesUp\b")); + } +} diff --git a/tests/Craft.Tests/OrchestrationModeTests.cs b/tests/Craft.Tests/OrchestrationModeTests.cs new file mode 100644 index 0000000..afc6f8a --- /dev/null +++ b/tests/Craft.Tests/OrchestrationModeTests.cs @@ -0,0 +1,320 @@ +using Craft.Storage; + +namespace Craft.Tests; + +/// +/// How a run's tasks are scheduled, end to end: fan-out (all at once), a concurrency limit (at most N at once, +/// workers released between tasks so other work interleaves), and sequential (one at a time on one pinned +/// worker, optionally stopping at the first failure). +/// +public class OrchestrationModeTests +{ + private static string Batch(int n, string prefix) => OrchestrationHarness.Batch(n, prefix); + + // ── concurrency limit ── + + [Fact] + public async Task AConcurrencyLimit_IsNeverExceeded_AndIsUsedInFull() + { + await using var h = await OrchestrationHarness.CreateAsync(poolSize: 6); + h.Svc.HoldMs = 40; + Assert.True(await h.Start("Capped", Batch(24, "c"), maxConcurrency: 2)); + + Assert.True(await h.DriveUntilFinished("Capped", 20_000)); + Assert.Equal(2, h.Svc.MaxActive); + Assert.Equal(24, h.Svc.Started.Distinct().Count()); + Assert.Equal(24, h.Svc.Started.Count); + } + + [Fact] + public async Task NoLimit_UsesTheWholePool() + { + await using var h = await OrchestrationHarness.CreateAsync(poolSize: 4); + h.Svc.HoldMs = 60; + Assert.True(await h.Start("Open", Batch(16, "o"))); + + Assert.True(await h.DriveUntilFinished("Open", 20_000)); + Assert.Equal(4, h.Svc.MaxActive); + } + + [Fact] + public async Task ALimitAboveThePoolSize_IsBoundedByThePool() + { + await using var h = await OrchestrationHarness.CreateAsync(poolSize: 3); + h.Svc.HoldMs = 40; + Assert.True(await h.Start("Roomy", Batch(12, "r"), maxConcurrency: 10)); + + Assert.True(await h.DriveUntilFinished("Roomy", 20_000)); + Assert.Equal(3, h.Svc.MaxActive); + } + + [Fact] + public async Task ALimitOfOne_RunsTheTasksInPayloadOrder() + { + await using var h = await OrchestrationHarness.CreateAsync(poolSize: 4); + h.Svc.HoldMs = 10; + Assert.True(await h.Start("OneByOne", Batch(6, "s"), maxConcurrency: 1)); + + Assert.True(await h.DriveUntilFinished("OneByOne")); + Assert.Equal(["s0", "s1", "s2", "s3", "s4", "s5"], h.Svc.Started); + Assert.Equal(1, h.Svc.MaxActive); + } + + [Fact] + public async Task ALimitOfOne_ReleasesTheWorkerBetweenTasks_SoUrgentWorkRunsInBetween() + { + Assert.Equal(["s0", "u0", "s1", "s2"], await InterleaveAsync(sequential: false)); + } + + [Fact] + public async Task ASequentialRun_HoldsItsWorker_SoUrgentWorkWaitsForTheWholeRun() + { + Assert.Equal(["s0", "s1", "s2", "u0"], await InterleaveAsync(sequential: true)); + } + + /// One worker. A three-task run (limit 1, or sequential) is held on its first task while an urgent + /// run is queued; the order everything then runs in shows whether the worker went back to the pool. + private static async Task> InterleaveAsync(bool sequential) + { + await using var h = await OrchestrationHarness.CreateAsync(poolSize: 1); + var gate = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + h.Svc.BeforeRun = t => FakeOrchestrator.IdOf(t) == "s0" ? gate.Task : Task.CompletedTask; + Assert.True(await h.Start("Slow", Batch(3, "s"), priority: 5, sequential: sequential, + maxConcurrency: sequential ? 0 : 1)); + + Assert.True(await h.DriveUntil(() => Task.FromResult(h.Svc.Started.Contains("s0")))); + Assert.True(await h.Start("Urgent", Batch(1, "u"), priority: 1)); + await h.DriveUntil(() => Task.FromResult(false), 200); + gate.SetResult(); + + Assert.True(await h.DriveUntilAllFinished()); + return [.. h.Svc.Started]; + } + + [Fact] + public async Task ALimitedRun_StillRunsItsPostExecutionOnce_AfterEveryTask() + { + await using var h = await OrchestrationHarness.CreateAsync(); + Assert.True(await h.Start("CappedAgg", Batch(5, "a"), "Agg", maxConcurrency: 1)); + + Assert.True(await h.DriveUntilFinished("CappedAgg")); + var post = Assert.Single(h.Svc.PostExecs); + Assert.Equal(5, post.Lines.Length); + } + + [Fact] + public async Task ALimitedRunWithFailures_StaysWithinItsLimit_AndFinishesWithErrors() + { + await using var h = await OrchestrationHarness.CreateAsync(poolSize: 6); + h.Svc.HoldMs = 20; + h.Svc.Body = t => int.Parse(FakeOrchestrator.IdOf(t)[1..], System.Globalization.CultureInfo.InvariantCulture) % 3 == 0 ? throw new InvalidOperationException("boom") : "{}"; + Assert.True(await h.Start("CappedFailing", Batch(12, "f"), maxConcurrency: 3)); + + Assert.True(await h.DriveUntilFinished("CappedFailing", 20_000)); + var run = (await h.Store.GetRunByNameAsync("CappedFailing"))!; + Assert.Equal(3, h.Svc.MaxActive); + Assert.Equal("CompletedWithErrors", run.Status); + Assert.Equal(4, run.Failed); + } + + [Fact] + public async Task TheLimitIsStored_SoAnotherProcessPickingUpTheRunHonoursIt() + { + var tables = new MemoryTableStore(); + await using (var first = await OrchestrationHarness.CreateAsync(tables: tables)) + Assert.True(await first.Start("Handover", Batch(10, "h"), maxConcurrency: 2)); + + await using var second = await OrchestrationHarness.CreateAsync(poolSize: 6, tables: tables); + second.Svc.HoldMs = 40; + Assert.True(await second.DriveUntilFinished("Handover", 20_000)); + Assert.Equal(2, second.Svc.MaxActive); + Assert.Equal(2, (await second.Store.GetRunByNameAsync("Handover"))!.MaxConcurrency); + } + + [Fact] + public async Task ALimitOnASequentialRun_IsDropped_AndANegativeLimitMeansNone() + { + await using var h = await OrchestrationHarness.CreateAsync(); + Assert.True(await h.Start("SeqCapped", Batch(2, "q"), sequential: true, maxConcurrency: 3)); + Assert.True(await h.Start("Negative", Batch(2, "n"), maxConcurrency: -5)); + + var seq = (await h.Store.GetRunByNameAsync("SeqCapped"))!; + Assert.True(seq.Sequential); + Assert.Equal(0, seq.MaxConcurrency); + Assert.Equal(0, (await h.Store.GetRunByNameAsync("Negative"))!.MaxConcurrency); + } + + // ── sequential: carry on (default) or stop on failure ── + + [Fact] + public async Task ASequentialRun_CarriesOnPastAFailedStep_ByDefault() + { + await using var h = await OrchestrationHarness.CreateAsync(); + h.Svc.Body = t => FakeOrchestrator.IdOf(t) == "s1" ? throw new InvalidOperationException("step down") : "{}"; + Assert.True(await h.Start("CarryOn", Batch(4, "s"), "Agg", sequential: true)); + + Assert.True(await h.DriveUntilFinished("CarryOn")); + Assert.Equal(["s0", "s1", "s2", "s3"], h.Svc.Started); + var run = (await h.Store.GetRunByNameAsync("CarryOn"))!; + Assert.Equal((1, 0, "CompletedWithErrors"), (run.Failed, run.Cancelled, run.Status)); + Assert.Equal(3, Assert.Single(h.Svc.PostExecs).Lines.Length); + } + + /// + /// A sequential run's steps all run inside the one job that drives them. Each step must still show up as + /// a job of its own: under its own name on the worker, and as its own record in job lists and run + /// summaries, or the queue page reports one task for the whole run. + /// + [Fact] + public async Task EveryStepOfASequentialRun_ShowsUpAsItsOwnJob() + { + await using var h = await OrchestrationHarness.CreateAsync(); + h.Svc.Body = t => FakeOrchestrator.IdOf(t) == "s1" ? throw new InvalidOperationException("step down") : "{}"; + Assert.True(await h.Start("SeqJobs", Batch(4, "s"), sequential: true)); + + Assert.True(await h.DriveUntilFinished("SeqJobs")); + Assert.Equal(["SeqJobs-Job_s0", "SeqJobs-Job_s1", "SeqJobs-Job_s2", "SeqJobs-Job_s3"], h.Svc.RanAs); + Assert.True(await h.DriveUntil(() => Task.FromResult(h.Jobs.GetJobs("SeqJobs").All(j => j.Status is not ("Queued" or "Running"))))); + var jobs = h.Jobs.GetJobs("SeqJobs").ToDictionary(j => j.Name, j => j.Status); + Assert.Equal(new Dictionary + { + ["SeqJobs-Job_s0"] = "Completed", + ["SeqJobs-Job_s1"] = "Failed", + ["SeqJobs-Job_s2"] = "Completed", + ["SeqJobs-Job_s3"] = "Completed", + }, jobs); + var summary = Assert.Single(h.Jobs.GetRunSummaries(), s => s.Name == "SeqJobs"); + Assert.Equal((4, 3, 1), (summary.Total, summary.Completed, summary.Failed)); + } + + [Fact] + public async Task ARunsSummaryAndTaskList_CoverItsLatestOuting_NotEveryRunOfThatName() + { + await using var h = await OrchestrationHarness.CreateAsync(); + for (var outing = 0; outing < 2; outing++) + { + Assert.True(await h.Start("SeqTwice", Batch(3, "s"), sequential: true)); + Assert.True(await h.DriveUntilFinished("SeqTwice")); + Assert.True(await h.DriveUntil(() => Task.FromResult(h.Jobs.GetJobs("SeqTwice").All(j => j.Status is not ("Queued" or "Running"))))); + } + + Assert.Equal(6, h.Jobs.GetJobs("SeqTwice").Count); + Assert.Equal(3, h.Jobs.GetRunJobs("SeqTwice", 100).Count); + var summary = Assert.Single(h.Jobs.GetRunSummaries(), s => s.Name == "SeqTwice"); + Assert.Equal((3, 3), (summary.Total, summary.Completed)); + } + + [Fact] + public async Task ASequentialRunsLastStep_DecidesItsJobsOutcome_AndItsAggregationIsAJobToo() + { + await using var h = await OrchestrationHarness.CreateAsync(); + h.Svc.Body = t => FakeOrchestrator.IdOf(t) == "s1" ? throw new InvalidOperationException("last down") : "{}"; + Assert.True(await h.Start("SeqLast", Batch(2, "s"), "Agg", sequential: true)); + + Assert.True(await h.DriveUntilFinished("SeqLast")); + Assert.True(await h.DriveUntil(() => Task.FromResult(h.Jobs.GetJobs("SeqLast").All(j => j.Status is not ("Queued" or "Running"))))); + var jobs = h.Jobs.GetJobs("SeqLast").ToDictionary(j => j.Name, j => j.Status); + Assert.Equal("Completed", jobs["SeqLast-Job_s0"]); + Assert.Equal("Failed", jobs["SeqLast-Job_s1"]); + Assert.Equal("Completed", jobs["SeqLast-PostExec"]); + Assert.Equal(3, jobs.Count); + } + + [Fact] + public async Task CancellingASequentialRun_StopsItAfterTheStepInHand() + { + await using var h = await OrchestrationHarness.CreateAsync(); + h.Svc.BeforeRun = async t => + { + if (FakeOrchestrator.IdOf(t) == "s1") await h.Svc.CancelRunAsync("SeqCancel"); + }; + Assert.True(await h.Start("SeqCancel", Batch(5, "s"), "Agg", sequential: true)); + + Assert.True(await h.DriveUntilFinished("SeqCancel")); + Assert.Equal(["s0", "s1"], h.Svc.Started); + var run = (await h.Store.GetRunByNameAsync("SeqCancel"))!; + Assert.Equal((0, 3, "CompletedWithErrors"), (run.Failed, run.Cancelled, run.Status)); + } + + /// + /// A step's finish and the claim of the next share one transaction. If that cannot be written the finish + /// goes the ordinary way (retried in the background) and the next step is claimed on its own. + /// + [Fact] + public async Task ASequentialStepWhoseCombinedWriteFails_IsRecordedOnItsOwn_AndTheRunCarriesOn() + { + var faulty = new FaultyTableStore(new MemoryTableStore()); + var failures = 0; + faulty.FailSubmit = (_, ops) => + ops.Count(o => o.Row.RowKey.StartsWith("R|", StringComparison.Ordinal)) == 2 + && Interlocked.CompareExchange(ref failures, 1, 0) == 0; + await using var h = await OrchestrationHarness.CreateAsync(tables: faulty); + Assert.True(await h.Start("SeqFallback", Batch(4, "s"), "Agg", sequential: true)); + + Assert.True(await h.DriveUntilFinished("SeqFallback")); + Assert.Equal(1, failures); + Assert.Equal(["s0", "s1", "s2", "s3"], h.Svc.Started); + var run = (await h.Store.GetRunByNameAsync("SeqFallback"))!; + Assert.Equal(("Completed", 4), (run.Status, run.Done)); + Assert.Contains(h.Log.Lines, l => l.Message.Contains("with its successor; recording it on its own", StringComparison.Ordinal)); + } + + [Fact] + public async Task StopOnFailure_CancelsTheStepsAfterTheFirstFailure_AndStillAggregates() + { + await using var h = await OrchestrationHarness.CreateAsync(); + h.Svc.Body = t => FakeOrchestrator.IdOf(t) == "s2" ? throw new InvalidOperationException("step down") : "{}"; + Assert.True(await h.Start("Stopper", Batch(5, "s"), "Agg", sequential: true, stopOnFailure: true)); + + Assert.True(await h.DriveUntilFinished("Stopper")); + Assert.Equal(["s0", "s1", "s2"], h.Svc.Started); + var run = (await h.Store.GetRunByNameAsync("Stopper"))!; + Assert.Equal((1, 2, "CompletedWithErrors"), (run.Failed, run.Cancelled, run.Status)); + + var done = (await h.Store.GetTasksAsync(run.RunKey, 'D')).OrderBy(t => t.Seq).ToList(); + Assert.Equal(["Completed", "Completed", "Failed", "Cancelled", "Cancelled"], done.Select(t => t.Status)); + Assert.All(done.Skip(3), t => Assert.Equal(WorkStore.StoppedReason("Job_s2"), t.LastError)); + Assert.Equal(2, Assert.Single(h.Svc.PostExecs).Lines.Length); + Assert.Equal(1, h.Svc.Checkouts); + Assert.Equal(1, h.Svc.Reclaims); + } + + [Fact] + public async Task StopOnFailure_WhenTheLastStepFails_HasNothingToCancel() + { + await using var h = await OrchestrationHarness.CreateAsync(); + h.Svc.Body = t => FakeOrchestrator.IdOf(t) == "s2" ? throw new InvalidOperationException("last") : "{}"; + Assert.True(await h.Start("StopLast", Batch(3, "s"), sequential: true, stopOnFailure: true)); + + Assert.True(await h.DriveUntilFinished("StopLast")); + var run = (await h.Store.GetRunByNameAsync("StopLast"))!; + Assert.Equal((1, 0), (run.Failed, run.Cancelled)); + } + + [Fact] + public async Task StopOnFailure_IsIgnoredForAFanOutRun_WhichAlwaysCarriesOn() + { + await using var h = await OrchestrationHarness.CreateAsync(); + h.Svc.Body = t => FakeOrchestrator.IdOf(t) == "f0" ? throw new InvalidOperationException("boom") : "{}"; + Assert.True(await h.Start("FanStop", Batch(4, "f"), stopOnFailure: true)); + + Assert.True(await h.DriveUntilFinished("FanStop")); + var run = (await h.Store.GetRunByNameAsync("FanStop"))!; + Assert.False(run.StopOnFailure); + Assert.Equal(4, h.Svc.Started.Count); + Assert.Equal((1, 0), (run.Failed, run.Cancelled)); + } + + [Fact] + public async Task TheRunsModeIsStoredOnItsHeader() + { + await using var h = await OrchestrationHarness.CreateAsync(); + Assert.True(await h.Start("ModeSeq", Batch(1, "a"), sequential: true, stopOnFailure: true)); + Assert.True(await h.Start("ModeCap", Batch(1, "b"), maxConcurrency: 7)); + + var seq = (await h.Store.GetRunByNameAsync("ModeSeq"))!; + var cap = (await h.Store.GetRunByNameAsync("ModeCap"))!; + Assert.Equal((true, 0, true), (seq.Sequential, seq.MaxConcurrency, seq.StopOnFailure)); + Assert.Equal((false, 7, false), (cap.Sequential, cap.MaxConcurrency, cap.StopOnFailure)); + } +} diff --git a/tests/Craft.Tests/OrchestrationRecoveryTests.cs b/tests/Craft.Tests/OrchestrationRecoveryTests.cs new file mode 100644 index 0000000..de750c6 --- /dev/null +++ b/tests/Craft.Tests/OrchestrationRecoveryTests.cs @@ -0,0 +1,365 @@ +using Craft.Configuration; +using Craft.Orchestration; +using Craft.PowerShellHost; +using Craft.Storage; +using Microsoft.Extensions.Configuration; +using Microsoft.Extensions.Logging.Abstractions; + +namespace Craft.Tests; + +/// +/// The guarantees that replace the old recovery machinery, each driven to its failure point: one process works +/// the queue (the instance lock), a stopped process's claims are taken back on sight, a failed index write never +/// strands a run, a crash at any write boundary is repaired from the active-run list, a finish that cannot be +/// written is retried rather than lost, and explains a stuck run. +/// +public class OrchestrationRecoveryTests +{ + private static readonly TimeSpan Lease = TimeSpan.FromMinutes(30); + + private static WorkStore NewStore(ICraftTableStore tables) => + new(NullLogger.Instance, new CraftSettings(), tables) { IndexRetries = [TimeSpan.Zero] }; + + private static JobManager NewJobs() + { + var settings = new CraftSettings(); + settings.Worker.BgPoolSize = 4; + var config = new ConfigurationBuilder().AddInMemoryCollection([]).Build(); + var repo = new ScriptRepository(NullLogger.Instance, settings); + var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); + var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); + return new JobManager(NullLogger.Instance, settings, limiter); + } + + private static WorkPump NewPump(WorkStore store, int lockSeconds = 5) + { + var settings = new CraftSettings(); + settings.Worker.BgPoolSize = 4; + var config = new ConfigurationBuilder().AddInMemoryCollection(new Dictionary + { + ["InstanceLockSeconds"] = lockSeconds.ToString(System.Globalization.CultureInfo.InvariantCulture), + }).Build(); + return new WorkPump(NullLogger.Instance, store, NewJobs(), config, settings); + } + + private static Task CreateAsync(WorkStore s, string name, int tasks, bool sequential = false, string? postExec = null) + { + var started = DateTime.UtcNow; + return s.CreateRunAsync(new RunHeader + { + RunKey = WorkStore.RunKeyFor(name, started), + Name = name, + StartedUtc = started, + TaskScriptName = "Invoke-CraftTask", + Sequential = sequential, + PostExecFunctionName = postExec, + }, Enumerable.Range(0, tasks).Select(i => new WorkStore.NewTask($"t{i}", [])).ToList()); + } + + // ── the instance lock ── + + [Fact] + public async Task OneProcessHoldsTheLock_TheNextGetsItOnceReleased() + { + var s = NewStore(new MemoryTableStore()); + + Assert.True((await s.TryHoldInstanceLockAsync("a", Lease)).Held); + var (held, holder) = await s.TryHoldInstanceLockAsync("b", Lease); + Assert.False(held); + Assert.Equal("a", holder!.Owner); + Assert.True((await s.TryHoldInstanceLockAsync("a", Lease)).Held); // renewal + + await s.ReleaseInstanceLockAsync("b"); // not the holder: no effect + Assert.Equal("a", (await s.GetInstanceLockAsync())!.Owner); + await s.ReleaseInstanceLockAsync("a"); + Assert.True((await s.TryHoldInstanceLockAsync("b", Lease)).Held); + } + + [Fact] + public async Task ALapsedLock_IsTakenOver() + { + var s = NewStore(new MemoryTableStore()); + Assert.True((await s.TryHoldInstanceLockAsync("crashed", TimeSpan.FromMilliseconds(30))).Held); + await Task.Delay(60); + + Assert.True((await s.TryHoldInstanceLockAsync("successor", Lease)).Held); + } + + [Fact] + public async Task TwoProcessesRacingForAFreeLock_OnlyOneGetsIt() + { + var mem = new MemoryTableStore(); + var s = NewStore(mem); + await s.InitializeAsync(); + var raced = false; + (bool Held, WorkStore.InstanceLock? Holder) rival = default; + mem.BeforeSubmit = async () => + { + if (raced) return; + raced = true; + mem.BeforeSubmit = null; + rival = await s.TryHoldInstanceLockAsync("rival", Lease); + }; + + var mine = await s.TryHoldInstanceLockAsync("me", Lease); + + Assert.True(rival.Held); + Assert.False(mine.Held); + Assert.Equal("rival", (await s.GetInstanceLockAsync())!.Owner); + } + + [Fact] + public async Task APumpWaitsForTheLock_AndStartsOnceItsPredecessorShutsDown() + { + var s = NewStore(new MemoryTableStore()); + var first = NewPump(s); + var second = NewPump(s); + await first.AcquireLockAsync(CancellationToken.None); + + var waiting = second.AcquireLockAsync(CancellationToken.None); + await Task.Delay(300); + Assert.False(waiting.IsCompleted); + Assert.False(second.HoldsLock); + + await first.StopAsync(CancellationToken.None); + await waiting.WaitAsync(TimeSpan.FromSeconds(5)); + Assert.True(second.HoldsLock); + } + + [Fact] + public async Task APumpTakesTheLock_FromAPredecessorThatCrashed_OnceItsLeaseRunsOut() + { + var s = NewStore(new MemoryTableStore()); + await s.TryHoldInstanceLockAsync("crashed/1/abc", TimeSpan.FromSeconds(2)); + var pump = NewPump(s); + + var started = DateTime.UtcNow; + await pump.AcquireLockAsync(CancellationToken.None).WaitAsync(TimeSpan.FromSeconds(10)); + + Assert.True(DateTime.UtcNow - started >= TimeSpan.FromSeconds(1.5)); + Assert.True(pump.HoldsLock); + } + + [Fact] + public async Task APumpThatLosesTheLock_StopsClaiming() + { + var s = NewStore(new MemoryTableStore()); + var pump = NewPump(s, lockSeconds: 6); + await pump.AcquireLockAsync(CancellationToken.None); + await s.ReleaseInstanceLockAsync((await s.GetInstanceLockAsync())!.Owner); + Assert.True((await s.TryHoldInstanceLockAsync("usurper", Lease)).Held); + + await Task.Delay(TimeSpan.FromSeconds(2.1)); // past a renewal interval (lease / 3) + Assert.False(await pump.KeepLockAsync(CancellationToken.None)); + Assert.False(pump.HoldsLock); + } + + // ── a stopped process's claims ── + + [Fact] + public async Task HoldingTheLock_AStoppedProcesssLiveClaimsAreTakenBackAtOnce() + { + var s = NewStore(new MemoryTableStore()); + var run = await CreateAsync(s, "Inherited", 3); + Assert.Equal(3, (await s.ClaimAsync(run.RunKey, 3, "old-host/7/aaaa", Lease, false)).Count); + var pump = NewPump(s); + + Assert.Equal(0, await pump.RefillAsync(CancellationToken.None)); // not holding the lock: left alone + pump.ForgetBackoff(); + await pump.AcquireLockAsync(CancellationToken.None); + Assert.Equal(3, await pump.RefillAsync(CancellationToken.None)); + Assert.All(await s.GetTasksAsync(run.RunKey, 'R'), t => Assert.Equal(2, t.Attempt)); + } + + [Fact] + public async Task HoldingTheLock_AStoppedProcesssSequentialDriver_IsTakenOverAtOnce() + { + var s = NewStore(new MemoryTableStore()); + var run = await CreateAsync(s, "SeqInherited", 3, sequential: true); + Assert.NotNull(await s.ClaimSequentialAsync(run.RunKey, "old-host/7/aaaa", Lease)); + var pump = NewPump(s); + await pump.AcquireLockAsync(CancellationToken.None); + + Assert.Equal(1, await pump.RefillAsync(CancellationToken.None)); + var step = Assert.Single(await s.GetTasksAsync(run.RunKey, 'R')); + Assert.Equal((0, 2), (step.Seq, step.Attempt)); + } + + // ── index writes that fail ── + + [Fact] + public async Task AFailedReadyUpdate_DoesNotStrandARun_ThisProcessIsWorking() + { + var faulty = new FaultyTableStore(new MemoryTableStore()); + await using var h = await OrchestrationHarness.CreateAsync(tables: faulty); + h.Store.IndexRetries = [TimeSpan.Zero]; + Assert.True(await h.Start("Stranded", OrchestrationHarness.Batch(2, "s"), "Agg")); + faulty.FailUpsert = (table, _, _) => table == "OrchestratorReady"; // every later Ready update fails + + Assert.True(await h.DriveUntilFinished("Stranded")); + Assert.Single(h.Svc.PostExecs); + } + + [Fact] + public async Task ARunWhoseReadyEntryNeverLanded_IsRelistedByTheRepairTheFailureTriggers() + { + var faulty = new FaultyTableStore(new MemoryTableStore()); + await using var h = await OrchestrationHarness.CreateAsync(tables: faulty); + h.Store.IndexRetries = [TimeSpan.Zero]; + faulty.FailUpsert = (table, _, _) => table == "OrchestratorReady"; + Assert.True(await h.Start("Unlisted", OrchestrationHarness.Batch(2, "u"))); + faulty.FailUpsert = null; + + Assert.Empty(await h.ReadyNamesAsync()); + Assert.Contains((await h.Svc.InspectRunAsync("Unlisted")).Runs[0].Diagnosis, d => d.StartsWith("Not on the Ready list", StringComparison.Ordinal)); + + await h.Pump.RefillAsync(CancellationToken.None); // sees the failure, repairs + await h.Pump.LastRepair!; + Assert.True(await h.DriveUntilFinished("Unlisted")); + } + + // ── a crash at a write boundary, repaired from the active-run list ── + + [Fact] + public async Task ACreationThatDiedBeforeItsHeader_IsRemoved_OnceItIsOldEnough() + { + var faulty = new FaultyTableStore(new MemoryTableStore()); + var s = NewStore(faulty); + faulty.FailUpsert = (table, _, rk) => table == "OrchestratorWork" && rk == WorkStore.HeaderKey; + await Assert.ThrowsAsync(() => CreateAsync(s, "HalfMade", 3)); + faulty.FailUpsert = null; + var runKey = Assert.Single(await ActiveKeysAsync(faulty)); + + var young = await s.RepairIndexesAsync(TimeSpan.FromHours(1)); + Assert.Equal((1, 0), (young.Young, young.Removed)); + + var old = await s.RepairIndexesAsync(TimeSpan.Zero); + Assert.Equal(1, old.Removed); + Assert.Empty(await ActiveKeysAsync(faulty)); + Assert.Empty(await s.GetTasksAsync(runKey)); + } + + [Fact] + public async Task ARunThatFinishedButWasNotRetired_IsRetiredByTheRepair() + { + var faulty = new FaultyTableStore(new MemoryTableStore()); + var s = NewStore(faulty); + var run = await CreateAsync(s, "Unretired", 1); + var claim = await s.ClaimAsync(run.RunKey, 1, "w", Lease, false); + faulty.FailDelete = (table, _) => table is "OrchestratorReady" or "OrchestratorNames"; + await s.FinishAsync(run.RunKey, [new WorkStore.Finish(claim[0].Seq, "Completed", Owner: "w")]); + faulty.FailDelete = null; + Assert.Single(await ActiveKeysAsync(faulty)); + Assert.True(s.IndexFailures > 0); + + var r = await s.RepairIndexesAsync(TimeSpan.FromMinutes(10)); + + Assert.Equal(1, r.Retired); + Assert.Empty(await ActiveKeysAsync(faulty)); + Assert.Empty(await NamesAsync(s)); + Assert.Equal(1, await s.SweepFinishedAsync(TimeSpan.Zero)); // retention still finds it + } + + internal static async Task> ActiveKeysAsync(FaultyTableStore t) + { + var keys = new List(); + await foreach (var r in t.QueryPartitionAsync("OrchestratorNames", "A")) keys.Add(r.RowKey); + return keys; + } + + private static async Task> NamesAsync(WorkStore s) + { + var names = new List(); + await foreach (var e in s.ReadReadyAsync()) names.Add(e.Name); + return names; + } + + // ── finishes that cannot be written ── + + [Fact] + public async Task AFinishThatCannotBeWritten_IsRetried_AndTheTaskIsNotRunAgain() + { + var faulty = new FaultyTableStore(new MemoryTableStore()); + await using var h = await OrchestrationHarness.CreateAsync(tables: faulty); + var failures = 0; + faulty.FailSubmit = (table, ops) => table == "OrchestratorWork" && ops.Any(o => o.Row.RowKey.StartsWith("D|", StringComparison.Ordinal)) + && Interlocked.Increment(ref failures) <= 3; + Assert.True(await h.Start("Flaky", OrchestrationHarness.Batch(4, "f"))); + + Assert.True(await h.DriveUntilFinished("Flaky", 30_000)); + Assert.True(failures >= 3); + Assert.Equal(4, h.Svc.Started.Count); + } + + [Fact] + public async Task ATaskFinishedTwiceInOneBatch_IsAppliedOnce() + { + var s = NewStore(new MemoryTableStore()); + var run = await CreateAsync(s, "Twice", 2); + var claim = (await s.ClaimAsync(run.RunKey, 1, "w", Lease, false))[0]; + + var outcome = await s.FinishAsync(run.RunKey, + [new WorkStore.Finish(claim.Seq, "Completed", Owner: "w"), new WorkStore.Finish(claim.Seq, "Cancelled", Owner: "w")]); + + Assert.Equal(1, outcome!.Applied); + Assert.Equal("Completed", Assert.Single(await s.GetTasksAsync(run.RunKey, 'D')).Status); + } + + // ── explaining a stuck run ── + + [Fact] + public async Task Inspection_SaysWhyARunIsNotMoving() + { + await using var h = await OrchestrationHarness.CreateAsync(); + + Assert.True(await h.Start("Capped", OrchestrationHarness.Batch(5, "c"), maxConcurrency: 2)); + var capped = (await h.Store.GetRunByNameAsync("Capped"))!; + Assert.Equal(2, (await h.Store.ClaimAsync(capped.RunKey, 2, h.Svc.Owner, Lease, false)).Count); + var cappedView = (await h.Svc.InspectRunAsync("Capped")).Runs.Single(); + Assert.Equal(2, cappedView.Running.Count(r => r.HeldHere)); + Assert.Contains(cappedView.Diagnosis, d => d.StartsWith("At its concurrency limit of 2", StringComparison.Ordinal)); + Assert.Equal("3", cappedView.Pending); + + Assert.True(await h.Start("Orphaned", OrchestrationHarness.Batch(1, "o"))); + var orphaned = (await h.Store.GetRunByNameAsync("Orphaned"))!; + await h.Store.ClaimAsync(orphaned.RunKey, 1, "gone-host/9/beef", Lease, false); + await h.Store.TryHoldInstanceLockAsync(h.Svc.Owner, Lease); + Assert.Contains((await h.Svc.InspectRunAsync("Orphaned")).Runs.Single().Diagnosis, + d => d.Contains("held by a process that no longer works the queue (gone-host/9/beef)")); + + Assert.True(await h.Start("Parent", OrchestrationHarness.Batch(1, "p"))); + var parent = (await h.Store.GetRunByNameAsync("Parent"))!; + Assert.NotNull(h.Svc.RegisterPendingChild(parent.RunKey, "Kid")); + Assert.Contains((await h.Svc.InspectRunAsync("Parent")).Runs.Single().Diagnosis, d => d.StartsWith("Waiting for 1 child run(s): Kid", StringComparison.Ordinal)); + + Assert.True(await h.Start("Quick", OrchestrationHarness.Batch(1, "q"))); + Assert.True(await h.DriveUntilFinished("Quick")); + Assert.Contains((await h.Svc.InspectRunAsync("Quick")).Runs.Single().Diagnosis, d => d.StartsWith("Finished Completed", StringComparison.Ordinal)); + } + + // ── bounded reads ── + + [Fact] + public async Task OnAHugeRun_StatusClaimAndInspection_ReadOnlyWhatTheyNeed() + { + var count = new CountingTableStore(new MemoryTableStore()); + await using var h = await OrchestrationHarness.CreateAsync(tables: count); + Assert.True(await h.Start("Huge", OrchestrationHarness.Batch(20_000, "h"))); + var reader = new JobQueueStatusReader(NullLogger.Instance, h.Jobs, h.Store); + count.Reset(); + + await reader.GetAsync(); + Assert.InRange(count.For("OrchestratorWork").Rows, 0, JobQueueStatusReader.HeadRows); + Assert.Equal(0, count.For("OrchestratorWork").UnboundedRanges); + + count.Reset(); + var run = (await h.Store.GetRunByNameAsync("Huge"))!; + await h.Store.ClaimAsync(run.RunKey, 49, "w", Lease, reclaimExpired: true, new WorkStore.ClaimProbe()); + Assert.Equal(0, count.For("OrchestratorWork").UnboundedRanges); + + count.Reset(); + await h.Svc.InspectRunAsync("Huge"); + Assert.Equal(0, count.For("OrchestratorWork").UnboundedRanges); + Assert.InRange(count.For("OrchestratorWork").Rows, 0, 1001 + 500); + } +} diff --git a/tests/Craft.Tests/OrchestrationThroughputTests.cs b/tests/Craft.Tests/OrchestrationThroughputTests.cs new file mode 100644 index 0000000..f022a26 --- /dev/null +++ b/tests/Craft.Tests/OrchestrationThroughputTests.cs @@ -0,0 +1,73 @@ +using System.Diagnostics; + +namespace Craft.Tests; + +/// +/// The pump's real loop, not single refills: it must refill as soon as the JobManager's buffer drains and as soon +/// as a run appears, not on its poll. On the poll alone, short tasks ran at about one batch a second (pool-size +/// tasks per second) and an idle instance took up to the 10 s idle poll to notice a new run. +/// +public class OrchestrationThroughputTests +{ + private static async Task StartedAsync() + { + var h = await OrchestrationHarness.CreateAsync(poolSize: 4); + h.Svc.MarkRecoveryDone(); + await h.Pump.StartAsync(CancellationToken.None); + var deadline = Environment.TickCount64 + 10_000; + while (!h.Pump.HoldsLock && Environment.TickCount64 < deadline) await Task.Delay(20); + Assert.True(h.Pump.HoldsLock); + return h; + } + + private static async Task WaitUntil(Func> done, int timeoutMs) + { + var deadline = Environment.TickCount64 + timeoutMs; + while (Environment.TickCount64 < deadline) + { + if (await done()) return true; + await Task.Delay(10); + } + return await done(); + } + + [Fact] + public async Task ShortTasks_AreNotHeldToOneBatchASecond() + { + await using var h = await StartedAsync(); + try + { + var sw = Stopwatch.StartNew(); + Assert.True(await h.Start("Quick", OrchestrationHarness.Batch(200, "q"))); + + Assert.True(await WaitUntil(async () => await h.Store.GetRunByNameAsync("Quick") is { IsFinished: true }, 30_000)); + sw.Stop(); + // On the poll alone this took about 50 s (four tasks a second). + Assert.True(sw.Elapsed < TimeSpan.FromSeconds(10), $"200 short tasks took {sw.Elapsed.TotalSeconds:F1}s"); + } + finally + { + await h.Pump.StopAsync(CancellationToken.None); + } + } + + [Fact] + public async Task AnIdleInstance_ClaimsANewRunAtOnce() + { + await using var h = await StartedAsync(); + try + { + await Task.Delay(TimeSpan.FromSeconds(4)); // let the idle poll back off + var sw = Stopwatch.StartNew(); + Assert.True(await h.Start("Fresh", OrchestrationHarness.Batch(1, "f"))); + + Assert.True(await WaitUntil(() => Task.FromResult(h.Svc.Started.Contains("f0")), 15_000)); + sw.Stop(); + Assert.True(sw.Elapsed < TimeSpan.FromSeconds(1), $"the first task of a new run started after {sw.Elapsed.TotalMilliseconds:F0}ms"); + } + finally + { + await h.Pump.StopAsync(CancellationToken.None); + } + } +} diff --git a/tests/Craft.Tests/OrchestratorBatchStreamingTests.cs b/tests/Craft.Tests/OrchestratorBatchStreamingTests.cs index dd8de03..b5d3965 100644 --- a/tests/Craft.Tests/OrchestratorBatchStreamingTests.cs +++ b/tests/Craft.Tests/OrchestratorBatchStreamingTests.cs @@ -175,8 +175,8 @@ public void MissingFile_YieldsNoTasks() /// /// Cleanup is the caller's, on every path — including the ones that never parse. /// - /// StartFromBatchAsync returns early when a run of the same name is already in progress or already - /// active, and neither return looks at the batch. Those are the common outcome for a duplicate + /// StartFromBatchAsync returns early when collisions are off and a run of the same name is already in + /// progress or active, and neither return looks at the batch. Those are the common outcome for a duplicate /// enqueue, so cleanup living at the parse site would leave the container's temp directory /// accumulating the batches of every run that was skipped rather than started. This pins the /// deletion to the outer method by driving it through those early returns. @@ -201,7 +201,7 @@ public async Task BatchFileIsDeleted_EvenWhenTheRunIsSkippedWithoutParsing() var path = WriteLines(["""{"FunctionName":"A","TenantFilter":"a.com"}"""]); await svc.StartFromBatchAsync("busy-run", string.Empty, 4, null, null, - CancellationToken.None, null, null, path); + CancellationToken.None, null, null, path, allowCollision: false); Assert.False(File.Exists(path), "the batch file outlived a skipped run — every enqueue that is skipped now leaks a temp file"); diff --git a/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs b/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs index 4b56760..46064b6 100644 --- a/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs +++ b/tests/Craft.Tests/OrchestratorBridgeLineageTests.cs @@ -1,12 +1,13 @@ using System.Collections.Concurrent; +using System.Collections.ObjectModel; using System.Management.Automation; using System.Management.Automation.Runspaces; using System.Reflection; -using System.Runtime.CompilerServices; using Craft.Hosting; using Craft.Orchestration; using Craft.PowerShellHost; using Craft.Services; +using Microsoft.Extensions.Configuration; using Microsoft.Extensions.Logging.Abstractions; namespace Craft.Tests; @@ -172,24 +173,98 @@ public void SelfParent_IsDroppedAtEnqueue() Assert.Null(pending!.ParentRunName); } + /// Run the real Start-CraftOrchestrator inside a production-configured worker, under a caller + /// context that the worker stamps into the runspace, and return what reached the bridge and the result. + private static async Task<(OrchestratorBridge.PendingOrchestration? Pending, string Result)> RunWrapperAsync( + string name, string inputObject, OperationContext.Invocation? caller = null) + { + var script = Path.Combine(AppContext.BaseDirectory, "Runtime", "CraftRuntime", "Start-CraftOrchestrator.ps1"); + var worker = await NewPinnedWorkerAsync(); + try + { + Collection output; + using (OperationContext.Set(caller ?? new OperationContext.Invocation("Push-Task"))) + output = await worker.InvokeScriptAsync(ScriptBlock.Create( + $"try {{ . '{script}'; Start-CraftOrchestrator -InputObject {inputObject} }} catch {{ \"ERROR: $_\" }}")); + var pending = TakePending(name); + if (pending?.BatchFilePath is { } path && File.Exists(path)) File.Delete(path); + return (pending, string.Join(",", output.Select(o => o?.ToString()))); + } + finally + { + worker.Dispose(); + } + } + [Fact] - public void Drain_ReleasesTheGate_WhenStartFails() - { - // The deadlock-avoidance guarantee: a child whose start attempt throws must stop gating - // its parent. The service here is deliberately missing its storage fields, so - // StartFromBatchAsync fails immediately — the finally in DrainPending must still release. - var svc = (OrchestratorService)RuntimeHelpers.GetUninitializedObject(typeof(OrchestratorService)); - void Set(string field, object value) => - typeof(OrchestratorService).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)! - .SetValue(svc, value); - Set("_logger", NullLogger.Instance); - var activeRuns = new ConcurrentDictionary(); - Set("_activeRuns", activeRuns); - Set("_childRuns", new ConcurrentDictionary>()); - Set("_recoveringChildren", new ConcurrentDictionary()); - var pendingChildRuns = new ConcurrentDictionary(); - Set("_pendingChildRuns", pendingChildRuns); - activeRuns.TryAdd("LineageDrainParent", new OrchestratorRun { Name = "LineageDrainParent", Status = "Running" }); + public async Task Wrapper_AChildInheritsOnlyItsParentsPriority_AndNamesItsExactParentRun() + { + var parent = new OperationContext.Invocation("Push-Task") { RunName = "WrapParent", RunKey = "WrapParent~8de0a1b2c3d4e5f", Priority = 7 }; + + var (pending, result) = await RunWrapperAsync("WrapChild", + "@{ OrchestratorName = 'WrapChild'; Batch = @(@{ FunctionName = 'X'; TenantFilter = 'a.com' }) }", parent); + + Assert.Equal("Craft-WrapChild", result); + Assert.NotNull(pending); + Assert.Equal(7, pending!.Priority); + Assert.Equal("WrapParent~8de0a1b2c3d4e5f", pending.ParentRunName); + Assert.Equal((false, true, 0, false), (pending.Sequential, pending.AllowCollision, pending.MaxConcurrency, pending.StopOnFailure)); + } + + [Fact] + public async Task Wrapper_AChildsOwnPriorityAndModeWin_OverItsParent() + { + var parent = new OperationContext.Invocation("Push-Task") { RunName = "WrapSeqParent", Priority = 7 }; + + var (pending, _) = await RunWrapperAsync("WrapOwnMode", + "@{ OrchestratorName = 'WrapOwnMode'; Priority = 2; Sequential = $true; StopOnFailure = $true; MaxConcurrency = 3; AllowCollision = $false; Batch = @(@{ FunctionName = 'X' }) }", + parent); + + Assert.NotNull(pending); + Assert.Equal(2, pending!.Priority); + Assert.Equal((true, false, 3, true), (pending.Sequential, pending.AllowCollision, pending.MaxConcurrency, pending.StopOnFailure)); + } + + [Fact] + public async Task Wrapper_WithoutAParent_UsesTheDefaultBand_AndNoLineage() + { + var (pending, _) = await RunWrapperAsync("WrapTopLevel", "@{ OrchestratorName = 'WrapTopLevel'; Batch = @(@{ FunctionName = 'X' }) }"); + + Assert.NotNull(pending); + Assert.Equal(4, pending!.Priority); + Assert.Null(pending.ParentRunName); + } + + [Fact] + public async Task Wrapper_WithoutCollisions_SkipsAndSaysSo_WhileARunOfThatNameIsActive() + { + // Against a real active run rather than a queued entry: the bridge queue is process-wide and any + // test's PostExecution drains it. + var (svc, store) = NewStoreBackedService(); + await CreateRunAsync(store, "WrapBusy"); + var previousService = s_serviceField.GetValue(null); + try + { + OrchestratorBridge.Initialize(svc); + var (pending, result) = await RunWrapperAsync("WrapBusy", + "@{ OrchestratorName = 'WrapBusy'; AllowCollision = $false; Batch = @(@{ FunctionName = 'X' }) }"); + + Assert.Equal("Craft-WrapBusy-Skipped", result); + Assert.Null(pending); + } + finally + { + s_serviceField.SetValue(null, previousService); + } + } + + [Fact] + public async Task Drain_ReleasesTheParent_WhenTheChildIsNeverCreated() + { + // The deadlock-avoidance guarantee: a child registered at enqueue that then fails to start (here an + // empty batch) must stop holding its parent, or the parent never reaches its barrier. + var (svc, store) = NewStoreBackedService(); + var parent = await CreateRunAsync(store, "LineageDrainParent"); var previousService = s_serviceField.GetValue(null); try @@ -198,18 +273,78 @@ void Set(string field, object value) => OrchestratorBridge.QueueOrchestration("LineageDrainChild", "[]", 4, null, null, null, parentRunName: "LineageDrainParent"); - Assert.True(pendingChildRuns.ContainsKey("LineageDrainChild")); + Assert.Equal(2, (await store.GetRunAsync(parent.RunKey))!.Total); OrchestratorBridge.DrainPending(); - Assert.False(pendingChildRuns.ContainsKey("LineageDrainChild")); - Assert.False(activeRuns.ContainsKey("LineageDrainChild")); + var after = (await store.GetRunAsync(parent.RunKey))!; + Assert.Equal(1, after.Done); + Assert.Equal(2, after.Total); + Assert.Null(await store.GetRunByNameAsync("LineageDrainChild")); } finally { // The bridge service is static process state — put back whatever was there so this - // test cannot redirect other tests' drains into the crippled service. + // test cannot redirect other tests' drains into this one. + s_serviceField.SetValue(null, previousService); + } + } + + [Fact] + public async Task IsRunActive_SeesUnfinishedRunsAndRunsQueuedToStart_SoACallerCanSkipAndSaySo() + { + var (svc, store) = NewStoreBackedService(); + await CreateRunAsync(store, "LineageActiveRun"); + + var previousService = s_serviceField.GetValue(null); + try + { + OrchestratorBridge.Initialize(svc); + Assert.True(OrchestratorBridge.IsRunActive("LineageActiveRun")); + Assert.False(OrchestratorBridge.IsRunActive("LineageNoSuchRun")); + + OrchestratorBridge.QueueOrchestration("LineageQueuedRun", "[]", 4); + Assert.True(OrchestratorBridge.IsRunActive("LineageQueuedRun")); + Assert.NotNull(TakePending("LineageQueuedRun")); + + await CreateRunAsync(store, $"LineageFamily-{Guid.NewGuid()}"); + Assert.True(OrchestratorBridge.IsRunActive($"LineageFamily-{Guid.NewGuid()}")); + Assert.True(OrchestratorBridge.IsRunActive("LineageFamily")); + var queued = $"LineageQueuedFamily-{Guid.NewGuid()}"; + OrchestratorBridge.QueueOrchestration(queued, "[]", 4); + Assert.True(OrchestratorBridge.IsRunActive($"LineageQueuedFamily-{Guid.NewGuid()}")); + Assert.NotNull(TakePending(queued)); + } + finally + { s_serviceField.SetValue(null, previousService); } } + + private static (OrchestratorService Service, Craft.Storage.WorkStore Store) NewStoreBackedService() + { + var settings = new Craft.Configuration.CraftSettings(); + var config = new ConfigurationBuilder().AddInMemoryCollection([]).Build(); + var repo = new ScriptRepository(NullLogger.Instance, settings); + var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); + var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); + var jobs = new Craft.Orchestration.JobManager(NullLogger.Instance, settings, limiter); + var mem = new MemoryTableStore(); + var store = new Craft.Storage.WorkStore(NullLogger.Instance, settings, mem); + var svc = new OrchestratorService(NullLogger.Instance, null!, limiter, jobs, store, + new Craft.Storage.ResultStore(NullLogger.Instance, settings, mem), config, settings); + return (svc, store); + } + + private static Task CreateRunAsync(Craft.Storage.WorkStore store, string name) + { + var started = DateTime.UtcNow; + return store.CreateRunAsync(new Craft.Storage.RunHeader + { + RunKey = Craft.Storage.WorkStore.RunKeyFor(name, started), + Name = name, + StartedUtc = started, + TaskScriptName = "Invoke-CraftTask", + }, [new Craft.Storage.WorkStore.NewTask("t0", new())]); + } } diff --git a/tests/Craft.Tests/OrchestratorChildRunGuardTests.cs b/tests/Craft.Tests/OrchestratorChildRunGuardTests.cs deleted file mode 100644 index f4a7343..0000000 --- a/tests/Craft.Tests/OrchestratorChildRunGuardTests.cs +++ /dev/null @@ -1,147 +0,0 @@ -using System.Collections.Concurrent; -using System.Reflection; -using System.Runtime.CompilerServices; -using Craft.Orchestration; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// A run must never wait on itself to finalize — and must genuinely wait on its children. -/// -/// The bridge registers the queued run under the ambient-or-explicit parent, so a run re-queued -/// from inside its own context (the recurring-run pattern) used to arrive as its own child. The -/// child-run guard then blocked finalization until the "child" left _activeRuns — which only happens -/// at finalization. Observed live as seven runs stuck "Running" for days, every task terminal, -/// Remaining=0, and their PostExecutions (audit log processing) never dispatched. -/// -/// The guard has to stay specific: a REAL child (different name) must block its parent from the -/// moment it is REGISTERED — which happens at enqueue time, while the child exists nowhere but the -/// bridge queue. Registration takes a pending gate that only lifts, once the start attempt has either put -/// the child into _activeRuns (which takes over the blocking) or failed (so the parent must not -/// wait forever). All directions are asserted here. -/// -public class OrchestratorChildRunGuardTests -{ - /// An OrchestratorService with only the fields the child-run guard touches. - private static OrchestratorService NewService() - { - var svc = (OrchestratorService)RuntimeHelpers.GetUninitializedObject(typeof(OrchestratorService)); - Set(svc, "_logger", NullLogger.Instance); - Set(svc, "_activeRuns", new ConcurrentDictionary()); - Set(svc, "_childRuns", new ConcurrentDictionary>()); - Set(svc, "_recoveringChildren", new ConcurrentDictionary()); - Set(svc, "_pendingChildRuns", new ConcurrentDictionary()); - return svc; - } - - private static void Set(object target, string field, object value) => - typeof(OrchestratorService).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)! - .SetValue(target, value); - - private static T Get(object target, string field) => - (T)typeof(OrchestratorService).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)! - .GetValue(target)!; - - private static void Activate(OrchestratorService svc, string name) => - Get>(svc, "_activeRuns") - .TryAdd(name, new OrchestratorRun { Name = name, Status = "Running" }); - - private static bool AllChildRunsComplete(OrchestratorService svc, string runName) => - (bool)typeof(OrchestratorService).GetMethod("AllChildRunsComplete", - BindingFlags.NonPublic | BindingFlags.Instance)!.Invoke(svc, [runName])!; - - [Fact] - public void SelfRegistration_IsRefused() - { - var svc = NewService(); - Activate(svc, "StandardsApply"); - - Assert.False(svc.TryRegisterPendingChildRun("StandardsApply", "StandardsApply")); - - Assert.False(Get>>(svc, "_childRuns") - .ContainsKey("StandardsApply")); - Assert.True(AllChildRunsComplete(svc, "StandardsApply")); - } - - [Fact] - public void InactiveParent_IsRefused() - { - // Runs queued from PostExecution land here: the spawning run has already finalized, so the - // new run keeps its lineage on the row but must not gate anything. - var svc = NewService(); - - Assert.False(svc.TryRegisterPendingChildRun("FinalizedParent", "FollowUpRun")); - Assert.True(AllChildRunsComplete(svc, "FinalizedParent")); - } - - [Fact] - public void PendingChild_BlocksParent_ThroughItsWholeLifecycle() - { - var svc = NewService(); - Activate(svc, "Parent"); - - // Registered at enqueue time — the child exists nowhere but the bridge queue, and the - // parent must already be blocked, because its own last task is what queued the child. - Assert.True(svc.TryRegisterPendingChildRun("Parent", "Child")); - Assert.False(AllChildRunsComplete(svc, "Parent")); - - // The drain starts the child (it enters the live graph) and then lifts the gate: still - // blocked, the live graph has taken over. - Activate(svc, "Child"); - svc.ReleasePendingChildRun("Child"); - Assert.False(AllChildRunsComplete(svc, "Parent")); - - // Child finalizes and is evicted from the live graph — the parent unblocks. - Get>(svc, "_activeRuns").TryRemove("Child", out _); - Assert.True(AllChildRunsComplete(svc, "Parent")); - } - - [Fact] - public void FailedStart_ReleasesTheGate() - { - // A child that never starts (0 tasks, missing task function, storage failure) must stop - // blocking once the start attempt is over — a leaked gate would defer the parent's - // finalize for the process lifetime. - var svc = NewService(); - Activate(svc, "Parent"); - - Assert.True(svc.TryRegisterPendingChildRun("Parent", "StillbornChild")); - Assert.False(AllChildRunsComplete(svc, "Parent")); - - svc.ReleasePendingChildRun("StillbornChild"); - Assert.True(AllChildRunsComplete(svc, "Parent")); - } - - [Fact] - public void DoubleQueuedChild_NeedsBothReleases() - { - // Two queue entries under the same child name (e.g. two parent tasks each queueing the - // same fan-out) hold independent gates: the first release must not lift the second's. - var svc = NewService(); - Activate(svc, "Parent"); - - Assert.True(svc.TryRegisterPendingChildRun("Parent", "SharedChild")); - Assert.True(svc.TryRegisterPendingChildRun("Parent", "SharedChild")); - - svc.ReleasePendingChildRun("SharedChild"); - Assert.False(AllChildRunsComplete(svc, "Parent")); - - svc.ReleasePendingChildRun("SharedChild"); - Assert.True(AllChildRunsComplete(svc, "Parent")); - } - - [Fact] - public void StaleSelfLink_DoesNotBlockFinalize() - { - // A self-link registered before the guard existed (or rebuilt from a pre-fix storage row) - // can still be sitting in the bag. It must not count as an outstanding child. - var svc = NewService(); - Activate(svc, "AuditLogSearchCreationV2"); - Get>>(svc, "_childRuns") - .TryAdd("AuditLogSearchCreationV2", ["AuditLogSearchCreationV2"]); - - Assert.True(AllChildRunsComplete(svc, "AuditLogSearchCreationV2")); - } -} diff --git a/tests/Craft.Tests/OrchestratorFinalizedRunTests.cs b/tests/Craft.Tests/OrchestratorFinalizedRunTests.cs deleted file mode 100644 index f519476..0000000 --- a/tests/Craft.Tests/OrchestratorFinalizedRunTests.cs +++ /dev/null @@ -1,346 +0,0 @@ -using System.Collections.Concurrent; -using System.Reflection; -using System.Runtime.CompilerServices; -using Craft.Configuration; -using Craft.Orchestration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// A finalized run must stay finalized. -/// -/// It did not. A run's queue rows were only dropped at finalize when it had NO post-execution, so a run -/// with one kept its rows from finalize until the post-execution succeeded. The pump re-claimed them in -/// that window, and the resolver — finding the run absent from _activeRuns because it had FINISHED — -/// rehydrated it from storage and put it back in the live graph. From there the next completion -/// re-finalized it and dispatched the aggregation again. -/// -/// Measured on a live 16-tenant instance before the fix, for one 13-task run: -/// 7x "Run MailboxRules_7ngn50... finalized: Completed (13/0/0/13)" -/// 7x "Dispatching PostExecution Push-StoreMailboxRules" -/// and individual tasks re-executed up to 4 times each, because a coalesced terminal write had not -/// landed yet and the rehydrated task still read Pending. -/// -/// Idempotent Push-* consumers absorbed this silently. Push-ScheduledTaskPostExecution does not — it -/// advances a recurring task by ScheduledTime + recurrence and writes it back, so every extra -/// invocation pushes the next run out by another interval. -/// -/// This pins the resolver half: a descriptor belonging to an already-finished run is dropped rather -/// than rehydrated. The guard has to be specific — an unfinished run absent from memory must still be -/// rehydrated, which is what crash recovery depends on — so both directions are asserted. -/// -public class OrchestratorFinalizedRunTests -{ - private sealed class FakeStore : ICraftTableStore - { - private readonly Dictionary> _tables = new(); - - public Task TryReplaceBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) => throw new NotSupportedException(); - - public Task PingAsync(CancellationToken ct = default) => Task.CompletedTask; - - public Task EnsureTableAsync(string table, CancellationToken ct = default) - { - if (!_tables.ContainsKey(table)) _tables[table] = new(); - return Task.CompletedTask; - } - - public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) - { - _tables[table][(row.PartitionKey, row.RowKey)] = row; - return Task.CompletedTask; - } - - public Task UpsertBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) - { - foreach (var r in rows) _tables[table][(r.PartitionKey, r.RowKey)] = r; - return Task.CompletedTask; - } - - public Task GetAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) - => Task.FromResult(_tables[table].TryGetValue((partitionKey, rowKey), out var r) ? r : null); - - public async IAsyncEnumerable QueryPartitionAsync(string table, string partitionKey, - [EnumeratorCancellation] CancellationToken ct = default) - { - foreach (var kv in _tables[table].Where(k => k.Key.Item1 == partitionKey).ToList()) - { - yield return kv.Value; - await Task.Yield(); - } - } - - public async IAsyncEnumerable QueryTableAsync(string table, - [EnumeratorCancellation] CancellationToken ct = default) - { - foreach (var kv in _tables[table].ToList()) - { - yield return kv.Value; - await Task.Yield(); - } - } - - public Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) - { - _tables[table].Remove((partitionKey, rowKey)); - return Task.CompletedTask; - } - - public Task DeletePartitionAsync(string table, string partitionKey, CancellationToken ct = default) - { - foreach (var k in _tables[table].Keys.Where(k => k.Item1 == partitionKey).ToList()) - _tables[table].Remove(k); - return Task.CompletedTask; - } - } - - /// - /// An OrchestratorService with only the fields the resolver reads. Building the real one would drag - /// in the PowerShell runner and the job manager for what is, on this path, a storage read plus a - /// status check. - /// - private static (OrchestratorService Svc, OrchestratorTableStore Store) NewService() - { - var settings = new CraftSettings { Orchestrator = { TablePrefix = "fin" + Guid.NewGuid().ToString("N")[..6] } }; - var store = new OrchestratorTableStore(NullLogger.Instance, settings, new FakeStore()); - - var svc = (OrchestratorService)RuntimeHelpers.GetUninitializedObject(typeof(OrchestratorService)); - Set(svc, "_logger", NullLogger.Instance); - Set(svc, "_store", store); - Set(svc, "_activeRuns", new ConcurrentDictionary()); - Set(svc, "_taskScriptPaths", new ConcurrentDictionary()); - // Field initializers do not run on an uninitialized object; the resolver locks this to read - // task status once it gets past the finished-run guard. - Set(svc, "_lock", new object()); - return (svc, store); - } - - private static void Set(object target, string field, object value) => - typeof(OrchestratorService).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)! - .SetValue(target, value); - - private static async Task ResolveAsync(OrchestratorService svc, string runName, string taskId) - { - var mi = typeof(OrchestratorService).GetMethod("ResolveTaskWorkAsync", - BindingFlags.NonPublic | BindingFlags.Instance)!; - try - { - var task = (Task)mi.Invoke(svc, [new JobDescriptor(runName, taskId, 4), CancellationToken.None])!; - await task; - return task.GetType().GetProperty("Result")!.GetValue(task); - } - catch (TargetInvocationException ex) - { - throw ex.InnerException ?? ex; - } - } - - private static async Task SeedRunAsync(OrchestratorTableStore store, string name, string status) - { - await store.InitializeAsync(); - await store.UpsertRunAsync(new OrchestratorRun - { - Name = name, - Status = status, - Priority = 4, - StartedUtc = DateTime.UtcNow, - // Pending on purpose: this is the state a coalesced terminal write has not caught up with, - // and the state that had the resolver hand back work for a task that had already run. - Tasks = [new OrchestratorTaskItem { Id = "task-0", Status = "Pending" }] - }); - await store.UpsertTaskAsync(name, new OrchestratorTaskItem { Id = "task-0", Status = "Pending" }); - } - - [Theory] - [InlineData("Completed")] - [InlineData("CompletedWithErrors")] - public async Task DescriptorForAFinishedRun_IsDropped_AndTheRunIsNotPutBackInTheLiveGraph(string status) - { - var (svc, store) = NewService(); - await SeedRunAsync(store, "finished-run", status); - ((ConcurrentDictionary)typeof(OrchestratorService) - .GetField("_taskScriptPaths", BindingFlags.NonPublic | BindingFlags.Instance)! - .GetValue(svc)!)["finished-run"] = "Invoke-CraftTask"; - - var work = await ResolveAsync(svc, "finished-run", "task-0"); - - Assert.Null(work); - - // The resurrection is the actual defect: once a finalized run is back in _activeRuns, the next - // completion re-finalizes it and dispatches its post-execution again. - var active = (ConcurrentDictionary)typeof(OrchestratorService) - .GetField("_activeRuns", BindingFlags.NonPublic | BindingFlags.Instance)!.GetValue(svc)!; - Assert.False(active.ContainsKey("finished-run"), - "a finalized run was rehydrated back into the live graph — it will finalize again"); - } - - // ── The finalize-once claim, and the paths that must not strand a run ────────────────────────── - - private static ConcurrentDictionary Claims(OrchestratorService svc) => - (ConcurrentDictionary)typeof(OrchestratorService) - .GetField("_finalizingRuns", BindingFlags.NonPublic | BindingFlags.Instance)!.GetValue(svc)!; - - /// - /// A second finalize is refused. This is the guard that keeps a run's Push-* aggregation to one - /// invocation; without it the observed run ran its aggregation seven times. - /// - [Fact] - public async Task FinalizeRun_RefusesASecondEntry() - { - var (svc, _) = NewService(); - Set(svc, "_finalizingRuns", new ConcurrentDictionary()); - - // Pre-claim, as a first finalize would. The second entry must return before touching anything — - // this instance has no writer or queue, so getting past the guard would throw rather than pass. - Claims(svc)["run"] = true; - - var run = new OrchestratorRun { Name = "run", Status = "Running", StartedUtc = DateTime.UtcNow }; - await InvokeFinalizeAsync(svc, run); - - // Untouched: a refused finalize must not restamp the run's status or completion time. - Assert.Equal("Running", run.Status); - Assert.Null(run.CompletedUtc); - } - - /// - /// The stranding risk the guard introduces: if a claim outlived a failed finalize, nothing would - /// ever finalize that run again — worse than the duplicate it prevents. A throw must release it. - /// - [Fact] - public async Task FinalizeRun_ReleasesTheClaim_WhenItThrows() - { - var (svc, _) = NewService(); - Set(svc, "_finalizingRuns", new ConcurrentDictionary()); - // _writer is left null, so the core finalize throws once past the guard. - - var run = new OrchestratorRun { Name = "run", Status = "Running", StartedUtc = DateTime.UtcNow }; - - await Assert.ThrowsAnyAsync(() => InvokeFinalizeAsync(svc, run)); - - Assert.False(Claims(svc).ContainsKey("run"), - "a failed finalize kept its claim — this run can never finalize again"); - } - - /// - /// Run names recur within one process (CIPPDBCacheOrchestrator, ProcessDeltaQueries fire on a - /// timer). Dispatching a run again is what makes it finalizable again. - /// - [Fact] - public async Task DispatchingARunAgain_ClearsAPreviousFinalizeClaim() - { - var (svc, _) = NewService(); - Set(svc, "_finalizingRuns", new ConcurrentDictionary()); - // No _runStatusTimers any more: the per-run status timer was replaced by a single sweep loop - // (RunStatusSweepLoopAsync), so DispatchPendingTasksAsync no longer creates or tracks a timer. - Claims(svc)["recurring-run"] = true; - - var run = new OrchestratorRun { Name = "recurring-run", Status = "Running", StartedUtc = DateTime.UtcNow }; - - // Only the bookkeeping prologue is exercised; the enqueue that follows needs a live queue. - try - { - var mi = typeof(OrchestratorService).GetMethod("DispatchPendingTasksAsync", - BindingFlags.NonPublic | BindingFlags.Instance)!; - await (Task)mi.Invoke(svc, [run, "Invoke-CraftTask", 4, CancellationToken.None, false])!; - } - catch { /* expected: no queue on this instance */ } - - Assert.False(Claims(svc).ContainsKey("recurring-run"), - "a recurring run kept its previous claim — its next outing would never finalize"); - } - - private static async Task InvokeFinalizeAsync(OrchestratorService svc, OrchestratorRun run) - { - var mi = typeof(OrchestratorService).GetMethod("FinalizeRunAsync", - BindingFlags.NonPublic | BindingFlags.Instance)!; - try - { - await (Task)mi.Invoke(svc, [run])!; - } - catch (TargetInvocationException ex) - { - throw ex.InnerException ?? ex; - } - } - - // ── A task already executing must not be started a second time ──────────────────────────────── - - /// - /// The crash-recovery race, at the resolver. - /// - /// A queue RowKey embeds the enqueue timestamp, so recovery re-dispatching a Pending task writes a - /// SECOND row rather than updating the surviving one. Most duplicates are harmless — by the time the - /// extra row is claimed the task has finished and it is dropped as terminal. But a claim that lands - /// while the task is still RUNNING used to pass the guard and start another copy: on a killed - /// 140-task fanout, a five-minute Intune collection was re-claimed four minutes in and ran twice. - /// - [Fact] - public async Task DescriptorForATaskThatIsAlreadyRunning_IsDropped() - { - var (svc, store) = NewService(); - await store.InitializeAsync(); - - var run = new OrchestratorRun - { - Name = "live-run", - Status = "Running", - Priority = 4, - StartedUtc = DateTime.UtcNow, - // OwnedHere: a worker in THIS process is executing it — the state dispatch leaves behind. - Tasks = [new OrchestratorTaskItem { Id = "task-0", Status = "Running", OwnedHere = true }] - }; - await store.UpsertRunAsync(run); - await store.UpsertTaskAsync("live-run", run.Tasks[0]); - - // In the live graph, mid-flight — the state a duplicate row races. - ((ConcurrentDictionary)typeof(OrchestratorService) - .GetField("_activeRuns", BindingFlags.NonPublic | BindingFlags.Instance)!.GetValue(svc)!)["live-run"] = run; - ((ConcurrentDictionary)typeof(OrchestratorService) - .GetField("_taskScriptPaths", BindingFlags.NonPublic | BindingFlags.Instance)! - .GetValue(svc)!)["live-run"] = "Invoke-CraftTask"; - - Assert.Null(await ResolveAsync(svc, "live-run", "task-0")); - } - - /// - /// The guard must not block genuine recovery. ResumeInterruptedRunsAsync flips interrupted tasks - /// from Running back to Pending before re-dispatching, so a task that really does need re-running - /// arrives here as Pending — and must resolve to work. - /// - [Fact] - public async Task ATaskResetToPendingByRecovery_StillResolvesToWork() - { - var (svc, store) = NewService(); - await SeedRunAsync(store, "recovered-run", "Running"); - ((ConcurrentDictionary)typeof(OrchestratorService) - .GetField("_taskScriptPaths", BindingFlags.NonPublic | BindingFlags.Instance)! - .GetValue(svc)!)["recovered-run"] = "Invoke-CraftTask"; - - Assert.NotNull(await ResolveAsync(svc, "recovered-run", "task-0")); - } - - [Fact] - public async Task DescriptorForAnUnfinishedRun_IsStillRehydrated() - { - // The guard must not swallow crash recovery: a Running run absent from memory is exactly what - // the rehydrate path exists for. - var (svc, store) = NewService(); - await SeedRunAsync(store, "live-run", "Running"); - ((ConcurrentDictionary)typeof(OrchestratorService) - .GetField("_taskScriptPaths", BindingFlags.NonPublic | BindingFlags.Instance)! - .GetValue(svc)!)["live-run"] = "Invoke-CraftTask"; - - var work = await ResolveAsync(svc, "live-run", "task-0"); - - Assert.NotNull(work); - - var active = (ConcurrentDictionary)typeof(OrchestratorService) - .GetField("_activeRuns", BindingFlags.NonPublic | BindingFlags.Instance)!.GetValue(svc)!; - Assert.True(active.ContainsKey("live-run"), - "an unfinished run was not re-established in the live graph — sibling completion tracking breaks"); - } -} diff --git a/tests/Craft.Tests/OrchestratorResultStreamingTests.cs b/tests/Craft.Tests/OrchestratorResultStreamingTests.cs index 59848e8..c129b0a 100644 --- a/tests/Craft.Tests/OrchestratorResultStreamingTests.cs +++ b/tests/Craft.Tests/OrchestratorResultStreamingTests.cs @@ -98,14 +98,21 @@ public Task DeletePartitionAsync(string table, string partitionKey, Cancellation } } - private static (OrchestratorTableStore Store, LazyProbeStore Backing) NewStore() + private static (ResultStore Store, LazyProbeStore Backing) NewStore() { var backing = new LazyProbeStore(); - var store = new OrchestratorTableStore( - NullLogger.Instance, new CraftSettings(), backing); + var store = new ResultStore( + NullLogger.Instance, new CraftSettings(), backing); return (store, backing); } + private static async Task ReadAllAsync(ResultStore store, string run) + { + var all = new List(); + await foreach (var r in store.StreamResultsAsync(run)) all.Add(r); + return all.ToArray(); + } + /// A result whose JSON exceeds the per-property limit and so gets chunked. private static string BigJson(int chars) => "\"" + new string('x', chars - 2) + "\""; @@ -211,7 +218,7 @@ public async Task SingleProperty_MultiChunkSingleRow_AndMultiRowSpill_AllRoundTr await store.StoreResultAsync("run", "b-chunked", chunked); await store.StoreResultAsync("run", "c-spilled", spilled); - var results = await store.GetResultsAsync("run"); + var results = await ReadAllAsync(store, "run"); Assert.Equal(3, results.Length); Assert.Contains(small, results); @@ -233,7 +240,7 @@ public async Task SpilledResult_Reassembles_WhenRowsArriveOutOfOrder() // still complete by chunk count rather than by arrival order. backing.Order = rows => rows.Reverse(); - var results = await store.GetResultsAsync("run"); + var results = await ReadAllAsync(store, "run"); Assert.Equal(2, results.Length); Assert.Contains(spilled, results); @@ -346,7 +353,7 @@ public async Task EmptyRun_WritesAnEmptyFile() // lines yields zero results without needing to special-case it. Assert.Equal(0, count); Assert.Equal("", await File.ReadAllTextAsync(path)); - Assert.Empty(await store.GetResultsAsync("run")); + Assert.Empty(await ReadAllAsync(store, "run")); } finally { diff --git a/tests/Craft.Tests/OrchestratorResultsAzuriteTests.cs b/tests/Craft.Tests/OrchestratorResultsAzuriteTests.cs index a8b5a0c..0174075 100644 --- a/tests/Craft.Tests/OrchestratorResultsAzuriteTests.cs +++ b/tests/Craft.Tests/OrchestratorResultsAzuriteTests.cs @@ -26,7 +26,7 @@ namespace Craft.Tests; /// public class OrchestratorResultsAzuriteTests { - private static async Task TryConnectAsync() + private static async Task TryConnectAsync() { var settings = new CraftSettings(); @@ -54,8 +54,8 @@ public class OrchestratorResultsAzuriteTests return null; } - var store = new OrchestratorTableStore( - NullLogger.Instance, settings, backing); + var store = new ResultStore( + NullLogger.Instance, settings, backing); await store.InitializeAsync(); return store; } @@ -99,7 +99,7 @@ public async Task EveryStorageShape_RoundTrips_AsExactlyOneLine() finally { if (File.Exists(path)) File.Delete(path); - await store.CleanupRunAsync("run"); + await store.DeleteRunAsync("run"); } } @@ -134,7 +134,7 @@ public async Task ManyResults_StreamOneLineEach_WithSpillsInterleaved() finally { if (File.Exists(path)) File.Delete(path); - await store.CleanupRunAsync("run"); + await store.DeleteRunAsync("run"); } } } diff --git a/tests/Craft.Tests/OrchestratorRetentionAzuriteTests.cs b/tests/Craft.Tests/OrchestratorRetentionAzuriteTests.cs deleted file mode 100644 index a86369f..0000000 --- a/tests/Craft.Tests/OrchestratorRetentionAzuriteTests.cs +++ /dev/null @@ -1,162 +0,0 @@ -using Azure; -using Azure.Data.Tables; -using Craft.Configuration; -using Craft.Orchestration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// The retention sweep against real tables (Azurite, or a storage account via -/// CRAFT_TEST_TABLE_CONNECTION). Two things only the real backend can prove: the orphan scan's -/// $select projection is accepted and still yields PartitionKey and Timestamp, and -/// DeletePartitionAsync really empties a partition written through the normal paths. Skipped, not -/// failed, when no backend is reachable — with the same caveat as -/// : a skip looks like a pass. -/// -public class OrchestratorRetentionAzuriteTests -{ - private sealed class Fixture : IAsyncDisposable - { - public required OrchestratorTableStore Store { get; init; } - public required AzureTableStore Backing { get; init; } - public required CraftSettings Settings { get; init; } - public required string Connection { get; init; } - - public static async Task TryConnectAsync() - { - var settings = new CraftSettings(); - var connection = Environment.GetEnvironmentVariable("CRAFT_TEST_TABLE_CONNECTION"); - if (!string.IsNullOrWhiteSpace(connection)) - settings.Auth.UserStorageConnection = connection; - else - { - settings.Storage.AllowDevelopmentStorage = true; - connection = "UseDevelopmentStorage=true"; - } - - // Unique per run so repeated runs cannot see each other's rows. - settings.Orchestrator.TablePrefix = "azrt" + Guid.NewGuid().ToString("N")[..8]; - - var backing = new AzureTableStore(settings); - try - { - using var cts = new CancellationTokenSource(TimeSpan.FromSeconds(3)); - await backing.PingAsync(cts.Token); - } - catch - { - return null; - } - - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); - await store.InitializeAsync(); - return new Fixture { Store = store, Backing = backing, Settings = settings, Connection = connection }; - } - - public string Table(string suffix) => Settings.Orchestrator.TablePrefix + suffix; - - public async Task CountAsync(string suffix, string partition) - { - var n = 0; - await foreach (var _ in Backing.QueryPartitionAsync(Table(suffix), partition)) n++; - return n; - } - - public Task AddRunAsync(string name, string status, DateTime started, DateTime? completed) => - Store.UpsertRunAsync(new OrchestratorRun { Name = name, Status = status, StartedUtc = started, CompletedUtc = completed }); - - public Task AddTasksAsync(string run, int count) => - Store.UpsertTaskBatchAsync(run, Enumerable.Range(1, count) - .Select(i => new OrchestratorTaskItem { Id = $"task-{i}", Status = "Completed" }).ToList()); - - public async ValueTask DisposeAsync() - { - // Drop the per-run tables so repeated local runs do not pile fixtures into the emulator. - var service = new TableServiceClient(Connection); - foreach (var suffix in new[] { "Runs", "Tasks", "Results" }) - { - try { await service.DeleteTableAsync(Table(suffix)); } - catch (RequestFailedException) { /* already gone */ } - } - } - } - - [Fact] - public async Task Sweep_RemovesFinishedRunsPastRetention_AndKeepsEverythingStillWanted() - { - await using var fx = await Fixture.TryConnectAsync(); - if (fx == null) return; - - var old = DateTime.UtcNow.AddDays(-3); - var recent = DateTime.UtcNow.AddHours(-1); - - await fx.AddRunAsync("done-old", "Completed", old, old); - await fx.AddTasksAsync("done-old", 3); - await fx.Store.StoreResultAsync("done-old", "task-1", "{\"ok\":true}"); - - await fx.AddRunAsync("cancelled-old", "Cancelled", old, old); - await fx.AddTasksAsync("cancelled-old", 1); - - await fx.AddRunAsync("failed-recent", "Failed", recent, recent); - await fx.AddTasksAsync("failed-recent", 2); - - // Started long ago, nobody in this process drives it, but its counter row was written just - // now — the heartbeat that says another process (or this one, moments ago) is still at it. - await fx.AddRunAsync("running", "Running", DateTime.UtcNow.AddDays(-10), null); - await fx.AddTasksAsync("running", 2); - await fx.Store.InitRemainingAsync("running", 2); - - // No Run row, but written moments ago: an orphan, just not an old one. - await fx.AddTasksAsync("ghost", 2); - - var result = await fx.Store.CleanupOldRunsAsync(TimeSpan.FromHours(48), new HashSet()); - - Assert.Collection(result.ExpiredRuns.OrderBy(n => n, StringComparer.Ordinal), - n => Assert.Equal("cancelled-old", n), - n => Assert.Equal("done-old", n)); - Assert.Empty(result.AbandonedRuns); - Assert.Equal(0, result.OrphanPartitionsRemoved); - Assert.Equal(4, result.RunsExamined); - - Assert.Null(await fx.Store.GetRunAsync("done-old")); - Assert.Null(await fx.Store.GetRunAsync("cancelled-old")); - Assert.Equal(0, await fx.CountAsync("Tasks", "done-old")); - Assert.Equal(0, await fx.CountAsync("Results", "done-old")); - Assert.Equal(0, await fx.CountAsync("Tasks", "cancelled-old")); - - Assert.NotNull(await fx.Store.GetRunAsync("failed-recent")); - Assert.Equal(2, await fx.CountAsync("Tasks", "failed-recent")); - Assert.Equal(2, (await fx.Store.GetRunAsync("running"))!.Tasks.Count); - Assert.Equal(2, await fx.CountAsync("Tasks", "ghost")); - } - - [Fact] - public async Task Sweep_RemovesOrphanedPartitions_OnceNothingInThemIsRecent() - { - await using var fx = await Fixture.TryConnectAsync(); - if (fx == null) return; - - await fx.AddTasksAsync("ghost", 3); - await fx.Store.StoreResultAsync("ghost", "task-1", "{\"ok\":true}"); - - // A run this process is driving: exempt from the abandoned rule, and its partition is never - // an orphan because its Run row is there. - await fx.AddRunAsync("kept", "Running", DateTime.UtcNow, null); - await fx.AddTasksAsync("kept", 2); - - // Every row was stamped by the service a moment ago. A cutoff in the future is the only way - // to make them "old" without waiting, and it also absorbs any skew between the emulator's - // clock and this process's. - var result = await fx.Store.CleanupOldRunsAsync(TimeSpan.FromMinutes(-5), new HashSet { "kept" }); - - Assert.Equal(2, result.OrphanPartitionsRemoved); - Assert.Empty(result.ExpiredRuns); - Assert.Empty(result.AbandonedRuns); - Assert.Equal(0, await fx.CountAsync("Tasks", "ghost")); - Assert.Equal(0, await fx.CountAsync("Results", "ghost")); - Assert.Equal(2, await fx.CountAsync("Tasks", "kept")); - Assert.NotNull(await fx.Store.GetRunAsync("kept")); - } -} diff --git a/tests/Craft.Tests/OrchestratorRetentionTests.cs b/tests/Craft.Tests/OrchestratorRetentionTests.cs deleted file mode 100644 index 4f833a2..0000000 --- a/tests/Craft.Tests/OrchestratorRetentionTests.cs +++ /dev/null @@ -1,314 +0,0 @@ -using Craft.Configuration; -using Craft.Orchestration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// The retention sweep is the only thing that bounds the orchestrator tables on a host that is not -/// restarted, and every rule in it is a rule about what NOT to delete while Craft is live. These pin -/// those rules against an in-memory backend; proves -/// the same sweep against real tables, where the projection and the partition deletes are real. -/// -public class OrchestratorRetentionTests -{ - private sealed class FakeStore : ICraftTableStore - { - private readonly Dictionary> _tables = new(); - - public Task TryReplaceBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) => throw new NotSupportedException(); - - public Task PingAsync(CancellationToken ct = default) => Task.CompletedTask; - - public Task EnsureTableAsync(string table, CancellationToken ct = default) - { - if (!_tables.ContainsKey(table)) _tables[table] = new(); - return Task.CompletedTask; - } - - public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) - { - _tables[table][(row.PartitionKey, row.RowKey)] = row; - return Task.CompletedTask; - } - - public Task UpsertBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) - { - foreach (var r in rows) _tables[table][(r.PartitionKey, r.RowKey)] = r; - return Task.CompletedTask; - } - - public Task GetAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) - => Task.FromResult(_tables[table].TryGetValue((partitionKey, rowKey), out var r) ? r : null); - - public async IAsyncEnumerable QueryPartitionAsync(string table, string partitionKey, - [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) - { - foreach (var kv in _tables[table].Where(k => k.Key.Item1 == partitionKey).ToList()) - { - yield return kv.Value; - await Task.Yield(); - } - } - - public async IAsyncEnumerable QueryTableAsync(string table, - [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) - { - foreach (var kv in _tables[table].ToList()) - { - yield return kv.Value; - await Task.Yield(); - } - } - - public Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) - { - _tables[table].Remove((partitionKey, rowKey)); - return Task.CompletedTask; - } - - public Task DeletePartitionAsync(string table, string partitionKey, CancellationToken ct = default) - { - foreach (var k in _tables[table].Keys.Where(k => k.Item1 == partitionKey).ToList()) - _tables[table].Remove(k); - return Task.CompletedTask; - } - } - - private static readonly TimeSpan Retention = TimeSpan.FromHours(48); - private static readonly HashSet NothingActive = new(StringComparer.Ordinal); - private static DateTime Old => DateTime.UtcNow.AddDays(-3); - private static DateTime Recent => DateTime.UtcNow.AddHours(-1); - - private sealed record Harness(OrchestratorTableStore Store, FakeStore Backing, CraftSettings Settings) - { - public string Runs => Settings.Orchestrator.TablePrefix + "Runs"; - public string Tasks => Settings.Orchestrator.TablePrefix + "Tasks"; - public string Results => Settings.Orchestrator.TablePrefix + "Results"; - - public Task AddRunAsync(string name, string status, DateTime started, DateTime? completed) => - Store.UpsertRunAsync(new OrchestratorRun { Name = name, Status = status, StartedUtc = started, CompletedUtc = completed }); - - public Task AddTasksAsync(string run, int count) => - Store.UpsertTaskBatchAsync(run, Enumerable.Range(1, count) - .Select(i => new OrchestratorTaskItem { Id = $"task-{i}", Status = "Completed" }).ToList()); - - /// A raw row carrying a Timestamp, the way a real backend returns every row. - public Task AddStampedRowAsync(string table, string partition, string rowKey, DateTime stamp) => - Backing.UpsertAsync(table, new StoreRow(partition, rowKey) { Timestamp = new DateTimeOffset(stamp, TimeSpan.Zero) }); - - public async Task CountAsync(string table, string partition) - { - var n = 0; - await foreach (var _ in Backing.QueryPartitionAsync(table, partition)) n++; - return n; - } - } - - private static async Task NewHarnessAsync() - { - var settings = new CraftSettings(); - var backing = new FakeStore(); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); - await store.InitializeAsync(); - return new Harness(store, backing, settings); - } - - [Fact] - public async Task CancelledRun_PastRetention_IsRemovedWithItsPartitions() - { - var h = await NewHarnessAsync(); - await h.AddRunAsync("cancelled", "Cancelled", Old, Old); - await h.AddTasksAsync("cancelled", 3); - await h.Store.StoreResultAsync("cancelled", "task-1", "{\"ok\":true}"); - - var result = await h.Store.CleanupOldRunsAsync(Retention); - - Assert.Equal("cancelled", Assert.Single(result.ExpiredRuns)); - Assert.Null(await h.Store.GetRunAsync("cancelled")); - Assert.Equal(0, await h.CountAsync(h.Tasks, "cancelled")); - Assert.Equal(0, await h.CountAsync(h.Results, "cancelled")); - } - - [Theory] - [InlineData("Completed")] - [InlineData("CompletedWithErrors")] - [InlineData("Failed")] - [InlineData("Cancelled")] - public async Task EveryTerminalStatus_PastRetention_IsRemoved(string status) - { - var h = await NewHarnessAsync(); - await h.AddRunAsync("run", status, Old, Old); - - var result = await h.Store.CleanupOldRunsAsync(Retention); - - Assert.Equal("run", Assert.Single(result.ExpiredRuns)); - Assert.Equal(1, result.RunsExamined); - Assert.Null(await h.Store.GetRunAsync("run")); - } - - [Fact] - public async Task FinishedRun_InsideRetention_IsKept_WithItsRows() - { - var h = await NewHarnessAsync(); - await h.AddRunAsync("recent", "Completed", Old, Recent); - await h.AddTasksAsync("recent", 2); - - var result = await h.Store.CleanupOldRunsAsync(Retention); - - Assert.Empty(result.ExpiredRuns); - Assert.NotNull(await h.Store.GetRunAsync("recent")); - Assert.Equal(2, await h.CountAsync(h.Tasks, "recent")); - } - - [Fact] - public async Task FinishedRun_WithoutCompletedUtc_IsJudgedByStartedUtc() - { - var h = await NewHarnessAsync(); - await h.AddRunAsync("failed-old", "Failed", Old, null); - await h.AddRunAsync("failed-recent", "Failed", Recent, null); - - var result = await h.Store.CleanupOldRunsAsync(Retention); - - Assert.Equal("failed-old", Assert.Single(result.ExpiredRuns)); - Assert.NotNull(await h.Store.GetRunAsync("failed-recent")); - } - - [Fact] - public async Task ActiveRun_IsKept_HoweverLongAgoItStarted() - { - var h = await NewHarnessAsync(); - await h.AddRunAsync("long", "Running", DateTime.UtcNow.AddDays(-10), null); - await h.AddTasksAsync("long", 2); - - var result = await h.Store.CleanupOldRunsAsync(Retention, new HashSet { "long" }); - - Assert.Empty(result.AbandonedRuns); - Assert.NotNull(await h.Store.GetRunAsync("long")); - Assert.Equal(2, await h.CountAsync(h.Tasks, "long")); - } - - [Fact] - public async Task RunNobodyIsDriving_WithNoWritesWithinRetention_IsAbandoned() - { - var h = await NewHarnessAsync(); - await h.AddRunAsync("stuck", "Pending", Old, null); - await h.AddTasksAsync("stuck", 2); - - var result = await h.Store.CleanupOldRunsAsync(Retention, NothingActive); - - Assert.Equal("stuck", Assert.Single(result.AbandonedRuns)); - Assert.Empty(result.ExpiredRuns); - Assert.Null(await h.Store.GetRunAsync("stuck")); - Assert.Equal(0, await h.CountAsync(h.Tasks, "stuck")); - } - - [Fact] - public async Task RunNobodyIsDriving_WithAFreshHeartbeat_IsKept() - { - // Started long ago, but a task completed an hour ago: the counter row is the heartbeat. - var h = await NewHarnessAsync(); - await h.AddRunAsync("slow", "Running", DateTime.UtcNow.AddDays(-10), null); - await h.AddTasksAsync("slow", 2); - await h.AddStampedRowAsync(h.Tasks, "slow", "!!run-counter", Recent); - - var result = await h.Store.CleanupOldRunsAsync(Retention, NothingActive); - - Assert.Empty(result.AbandonedRuns); - Assert.NotNull(await h.Store.GetRunAsync("slow")); - Assert.Equal(3, await h.CountAsync(h.Tasks, "slow")); - } - - [Fact] - public async Task UnknownStatus_IsTreatedAsNotFinished() - { - var h = await NewHarnessAsync(); - await h.AddRunAsync("odd-stale", "Suspended", Old, null); - await h.AddRunAsync("odd-fresh", "Suspended", Old, null); - await h.AddStampedRowAsync(h.Tasks, "odd-fresh", "!!run-counter", Recent); - - var result = await h.Store.CleanupOldRunsAsync(Retention, NothingActive); - - Assert.Equal("odd-stale", Assert.Single(result.AbandonedRuns)); - Assert.Empty(result.ExpiredRuns); - Assert.NotNull(await h.Store.GetRunAsync("odd-fresh")); - } - - [Fact] - public async Task OrphanedPartitions_WithOnlyOldRows_AreRemoved() - { - var h = await NewHarnessAsync(); - await h.AddStampedRowAsync(h.Tasks, "ghost", "task-1", Old); - await h.AddStampedRowAsync(h.Tasks, "ghost", "task-2", Old); - await h.AddStampedRowAsync(h.Results, "ghost", "task-1", Old); - - var result = await h.Store.CleanupOldRunsAsync(Retention); - - Assert.Equal(2, result.OrphanPartitionsRemoved); - Assert.Equal(0, await h.CountAsync(h.Tasks, "ghost")); - Assert.Equal(0, await h.CountAsync(h.Results, "ghost")); - } - - [Fact] - public async Task OrphanedPartition_WithAFreshRow_IsKeptWhole() - { - var h = await NewHarnessAsync(); - await h.AddStampedRowAsync(h.Tasks, "ghost", "task-1", Old); - await h.AddStampedRowAsync(h.Tasks, "ghost", "task-2", Old); - await h.AddStampedRowAsync(h.Tasks, "ghost", "task-3", Recent); - - var result = await h.Store.CleanupOldRunsAsync(Retention); - - Assert.Equal(0, result.OrphanPartitionsRemoved); - Assert.Equal(3, await h.CountAsync(h.Tasks, "ghost")); - } - - [Fact] - public async Task PartitionOfARetainedRun_IsNotAnOrphan_EvenWhenItsRowsAreOld() - { - var h = await NewHarnessAsync(); - await h.AddRunAsync("resumed", "Completed", Old, Recent); - await h.AddStampedRowAsync(h.Tasks, "resumed", "task-1", Old); - await h.AddStampedRowAsync(h.Results, "resumed", "task-1", Old); - - var result = await h.Store.CleanupOldRunsAsync(Retention); - - Assert.Equal(0, result.OrphanPartitionsRemoved); - Assert.Equal(1, await h.CountAsync(h.Tasks, "resumed")); - Assert.Equal(1, await h.CountAsync(h.Results, "resumed")); - } - - [Fact] - public async Task RowsWithoutATimestamp_AreNeverJudgedOrphans() - { - // A backend that cannot say how old a row is gets the benefit of the doubt. - var h = await NewHarnessAsync(); - await h.AddTasksAsync("ghost", 2); - - var result = await h.Store.CleanupOldRunsAsync(Retention); - - Assert.Equal(0, result.OrphanPartitionsRemoved); - Assert.Equal(2, await h.CountAsync(h.Tasks, "ghost")); - } - - [Fact] - public async Task Sweep_ReportsEverythingItExamined() - { - var h = await NewHarnessAsync(); - await h.AddRunAsync("done", "Completed", Old, Old); - await h.AddRunAsync("live", "Running", Recent, null); - await h.AddRunAsync("stuck", "Pending", Old, null); - await h.AddStampedRowAsync(h.Results, "ghost", "r", Old); - - var result = await h.Store.CleanupOldRunsAsync(Retention, new HashSet { "live" }); - - Assert.Equal(3, result.RunsExamined); - Assert.Equal("done", Assert.Single(result.ExpiredRuns)); - Assert.Equal("stuck", Assert.Single(result.AbandonedRuns)); - Assert.Equal(1, result.OrphanPartitionsRemoved); - Assert.NotNull(await h.Store.GetRunAsync("live")); - } -} diff --git a/tests/Craft.Tests/OrchestratorRunPersistenceTests.cs b/tests/Craft.Tests/OrchestratorRunPersistenceTests.cs deleted file mode 100644 index f3ca354..0000000 --- a/tests/Craft.Tests/OrchestratorRunPersistenceTests.cs +++ /dev/null @@ -1,249 +0,0 @@ -using System.Runtime.CompilerServices; -using Craft.Configuration; -using Craft.Orchestration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// The run row is what a restart rebuilds a run's identity from. Anything on -/// that is not written here is silently null after recovery, and the -/// feature that depends on it stops working without any error. -/// -/// Reference and ParentRunName were exactly that: present on the model, never persisted. A resumed run -/// came back unreferenceable (QueueStatusBridge could not look it up) and orphaned (its finalize never -/// re-checked the parent, and the parent had no record of it). -/// -public class OrchestratorRunPersistenceTests -{ - private sealed class FakeStore : ICraftTableStore - { - - // Claims are not exercised by this fake. Fail loudly rather than pretend the guard held — - // a silent 'true' here would look exactly like a successful claim. - public Task TryReplaceBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) => throw new NotSupportedException(); - private readonly Dictionary> _tables = new(); - - public Task PingAsync(CancellationToken ct = default) => Task.CompletedTask; - - public Task EnsureTableAsync(string table, CancellationToken ct = default) - { - if (!_tables.ContainsKey(table)) _tables[table] = new(); - return Task.CompletedTask; - } - - public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) - { - _tables[table][(row.PartitionKey, row.RowKey)] = row; - return Task.CompletedTask; - } - - public Task UpsertBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) - { - foreach (var r in rows) _tables[table][(r.PartitionKey, r.RowKey)] = r; - return Task.CompletedTask; - } - - public Task GetAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) - => Task.FromResult(_tables[table].TryGetValue((partitionKey, rowKey), out var r) ? r : null); - - public async IAsyncEnumerable QueryPartitionAsync(string table, string partitionKey, - [EnumeratorCancellation] CancellationToken ct = default) - { - foreach (var kv in _tables[table].Where(k => k.Key.Item1 == partitionKey).ToList()) - { - yield return kv.Value; - await Task.Yield(); - } - } - - public async IAsyncEnumerable QueryTableAsync(string table, - [EnumeratorCancellation] CancellationToken ct = default) - { - foreach (var kv in _tables[table].ToList()) - { - yield return kv.Value; - await Task.Yield(); - } - } - - public Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) - { - _tables[table].Remove((partitionKey, rowKey)); - return Task.CompletedTask; - } - - public Task DeletePartitionAsync(string table, string partitionKey, CancellationToken ct = default) - { - foreach (var k in _tables[table].Keys.Where(k => k.Item1 == partitionKey).ToList()) - _tables[table].Remove(k); - return Task.CompletedTask; - } - } - - private static OrchestratorTableStore NewStore() => - new(NullLogger.Instance, new CraftSettings(), new FakeStore()); - - [Fact] - public async Task Reference_And_ParentRunName_SurviveARestart() - { - var store = NewStore(); - await store.InitializeAsync(); - - await store.UpsertRunAsync(new OrchestratorRun - { - Name = "ChildRun", - Reference = "user-request-4711", - ParentRunName = "ParentRun", - Status = "Running", - Priority = 3, - StartedUtc = DateTime.UtcNow, - }); - - var recovered = await store.GetRunAsync("ChildRun"); - - Assert.NotNull(recovered); - Assert.Equal("user-request-4711", recovered!.Reference); - Assert.Equal("ParentRun", recovered.ParentRunName); - } - - /// - /// PostExecAttemptCount is what bounds the retry of a failed post-execution across restarts. If it - /// did not survive the restart it would read back as 0 every time, and the bound would never be - /// reached — a post-execution that crashes the host would be retried on every start forever. - /// - [Fact] - public async Task PostExecAttemptCount_SurvivesARestart() - { - var store = NewStore(); - await store.InitializeAsync(); - - await store.UpsertRunAsync(new OrchestratorRun - { - Name = "AggregatedRun", - Status = "Completed", - PostExecFunctionName = "StoreMailboxPermissions", - PostExecStatus = "Failed", - PostExecAttemptCount = 2, - StartedUtc = DateTime.UtcNow, - }); - - var recovered = await store.GetRunAsync("AggregatedRun"); - - Assert.Equal("Failed", recovered!.PostExecStatus); - Assert.Equal(2, recovered.PostExecAttemptCount); - } - - /// - /// Rows written before the counter existed have no such property. They must read as 0 — an - /// already-exhausted reading would abandon in-flight aggregations on the upgrade restart. - /// - [Fact] - public async Task RunWrittenWithoutTheCounter_ReadsAsZeroAttempts() - { - var store = NewStore(); - await store.InitializeAsync(); - - await store.UpsertRunAsync(new OrchestratorRun - { - Name = "Legacy", - Status = "Completed", - PostExecStatus = "Pending", - StartedUtc = DateTime.UtcNow, - }); - - Assert.Equal(0, (await store.GetRunAsync("Legacy"))!.PostExecAttemptCount); - } - - /// Runs without either field keep round-tripping as null — no backfill required. - [Fact] - public async Task RunsWithoutReferenceOrParent_RoundTripAsNull() - { - var store = NewStore(); - await store.InitializeAsync(); - - await store.UpsertRunAsync(new OrchestratorRun - { - Name = "Plain", - Status = "Running", - Priority = 5, - StartedUtc = DateTime.UtcNow, - }); - - var recovered = await store.GetRunAsync("Plain"); - - Assert.Null(recovered!.Reference); - Assert.Null(recovered.ParentRunName); - } - - /// - /// The summary scan is what startup uses to rebuild parent→child links. It must report parentage - /// without loading task rows — using GetRunAsync per run would pull every task of every run. - /// - [Fact] - public async Task ListRunSummaries_ReportsParentage_WithoutLoadingTasks() - { - var store = NewStore(); - await store.InitializeAsync(); - - var childTasks = Enumerable.Range(0, 50) - .Select(i => new OrchestratorTaskItem { Id = $"t{i}", Status = "Pending" }).ToList(); - - await store.UpsertRunAsync(new OrchestratorRun - { - Name = "Parent", - Status = "Running", - StartedUtc = DateTime.UtcNow, - }); - await store.UpsertRunAsync(new OrchestratorRun - { - Name = "Child", - ParentRunName = "Parent", - Status = "Running", - StartedUtc = DateTime.UtcNow, - Tasks = childTasks, - }); - await store.UpsertRunAsync(new OrchestratorRun - { - Name = "DoneChild", - ParentRunName = "Parent", - Status = "Completed", - StartedUtc = DateTime.UtcNow, - }); - await store.UpsertTaskBatchAsync("Child", childTasks); - - var summaries = await store.ListRunSummariesAsync(); - - Assert.Equal(3, summaries.Count); - Assert.Null(summaries.Single(s => s.Name == "Parent").ParentRunName); - Assert.Equal("Parent", summaries.Single(s => s.Name == "Child").ParentRunName); - Assert.Equal("Running", summaries.Single(s => s.Name == "Child").Status); - - // Terminal children are filtered out of the rebuild by status — they cannot block a parent. - Assert.Equal("Completed", summaries.Single(s => s.Name == "DoneChild").Status); - } - - /// - /// Reference is looked up case-insensitively by QueueStatusBridge, so it has to come back with its - /// original casing intact rather than being normalised on the way through storage. - /// - [Fact] - public async Task Reference_PreservesCasing() - { - var store = NewStore(); - await store.InitializeAsync(); - - await store.UpsertRunAsync(new OrchestratorRun - { - Name = "Run", - Reference = "Tenant-ABC_Sync", - Status = "Running", - StartedUtc = DateTime.UtcNow, - }); - - Assert.Equal("Tenant-ABC_Sync", (await store.GetRunAsync("Run"))!.Reference); - } -} diff --git a/tests/Craft.Tests/OrchestratorSequentialTests.cs b/tests/Craft.Tests/OrchestratorSequentialTests.cs deleted file mode 100644 index f5824e2..0000000 --- a/tests/Craft.Tests/OrchestratorSequentialTests.cs +++ /dev/null @@ -1,527 +0,0 @@ -using System.Collections.Concurrent; -using System.Reflection; -using Craft.Configuration; -using Craft.Orchestration; -using Craft.PowerShellHost; -using Craft.Services; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// Pins SEQUENTIAL orchestrator mode. -/// -/// A run marked runs its tasks ONE AT A TIME, in ascending -/// (payload) order, ON A SINGLE PINNED WORKER: one entry row -/// is dispatched, that one claim drives the WHOLE run inline (checkout one worker → run every step on it → -/// reclaim once), and the not-yet-reached steps never get their own queue row. The default (false) is the -/// existing fan-out: every task enqueued up front and drained in parallel by the pool. -/// -/// These fix the contract at every seam it touches: -/// - persistence — the flag and the per-task order survive a round trip AND a status-write rewrite -/// (Replace mode erases any column the write omits), so a resumed run keeps its order; -/// - dispatch — a sequential run enqueues only its single entry row; fan-out enqueues all; -/// - driver — every step runs, in Sequence order, on ONE worker; a failing step does not strand the -/// rest (best-effort); a cancelled run marks the remaining steps Cancelled; a duplicate -/// entry row is a no-op while a driver is already active; -/// - re-drive — the watchdog leaves a run alone while its driver is active (or its entry job is still -/// queued/running), yet still restarts the driver if the entry row is lost. -/// -/// The driver's loop logic is exercised through , a subclass that overrides the -/// three PowerShell seams (checkout / run-step / reclaim), so these tests need no worker pool. The actual -/// pinning to a live worker is proven separately by live validation against the dev backend. -/// -public class OrchestratorSequentialTests -{ - private const string TaskFunc = "Invoke-CraftTask"; - - // ─── seam subclass ────────────────────────────────────────────────────────────────────────────── - // Overrides only the three PowerShell interactions. Everything else — ordering, best-effort, marker - // writes, completion, cancellation — runs the real OrchestratorService code. - - private sealed class SeqDriver : OrchestratorService - { - // OrchestratorService has only a parameterized constructor; a subclass must chain to it to compile. - // Never actually invoked — the harness builds this via GetUninitializedObject — so the nulls are safe. - private SeqDriver() : base(null!, null!, null!, null!, null!, null!, null!, null!) { } - - public List Order = new(); // step ids in the order the driver ran them - public HashSet FailIds = new(); // ids whose step throws (best-effort test) - public int Checkouts; // must be exactly 1 for a run that starts - public int Reclaims; // must pair 1:1 with Checkouts - public string StepOutput = string.Empty; // what a step "returns" (PostExecution capture test) - public Action? OnStep; // test hook, runs synchronously inside a step - - internal override PowerShellWorker? CheckoutSequentialWorker(CancellationToken ct) - { - Checkouts++; - return null; // step execution is faked, so the worker handle is never dereferenced - } - - internal override void ReclaimSequentialWorker(PowerShellWorker? worker, bool faulted) => Reclaims++; - - internal override Task RunSequentialStepAsync( - OrchestratorRun run, OrchestratorTaskItem task, string taskPath, PowerShellWorker? worker) - { - Order.Add(task.Id); - OnStep?.Invoke(task); - if (FailIds.Contains(task.Id)) - throw new InvalidOperationException("boom " + task.Id); - return Task.FromResult(StepOutput); - } - } - - // ─── harness ──────────────────────────────────────────────────────────────────────────────────── - - private sealed record Harness(SeqDriver Svc, OrchestratorTableStore Store, JobQueueStore Queue, JobManager Jobs); - - private static async Task NewHarnessAsync() - { - var settings = new CraftSettings - { - Orchestrator = { TablePrefix = "seq" + Guid.NewGuid().ToString("N")[..8] } - }; - // Direct, synchronous status writes — no batching barrier or drain loop to stand up in a unit test. - settings.Orchestrator.BatchStatusWrites = false; - - var backing = new RunRemainingCounterTests.ConditionalStore(); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); - var queue = new JobQueueStore(NullLogger.Instance, settings, backing); - await store.InitializeAsync(); - await queue.InitializeAsync(); - var writer = new OrchestratorStatusWriter(store, NullLogger.Instance, settings); - - // Build the service without its constructor (which drags in the PowerShell runner and the worker - // pool) and set only the fields the dispatch / driver / re-drive paths read — the same shape the - // other orchestrator unit tests use. - var svc = (SeqDriver)System.Runtime.CompilerServices.RuntimeHelpers.GetUninitializedObject(typeof(SeqDriver)); - svc.Order = new(); - svc.FailIds = new(); - Set(svc, "_logger", NullLogger.Instance); - Set(svc, "_store", store); - Set(svc, "_queue", queue); - Set(svc, "_writer", writer); - Set(svc, "_lock", new object()); - Set(svc, "_activeRuns", new ConcurrentDictionary()); - Set(svc, "_taskScriptPaths", new ConcurrentDictionary()); - Set(svc, "_finalizingRuns", new ConcurrentDictionary()); - Set(svc, "_finalizeDeferrals", new ConcurrentDictionary()); - Set(svc, "_childRuns", new ConcurrentDictionary>()); - Set(svc, "_cancelledRuns", new ConcurrentDictionary()); - Set(svc, "_activeSequentialDrivers", new ConcurrentDictionary()); - Set(svc, "_requeueFailures", new ConcurrentDictionary()); - Set(svc, "_deferrals", NewFieldDict(svc, "_deferrals")); - Set(svc, "_redriveBackoff", NewFieldDict(svc, "_redriveBackoff")); - Set(svc, "_shedParameters", false); - Set(svc, "_redriveBackoffEnabled", false); // pin the sequential logic, not the backoff timing - Set(svc, "_redriveBase", TimeSpan.FromSeconds(60)); - Set(svc, "_settings", settings); - - // A JobManager with an empty job map, so IsQueuedOrRunning answers false (nothing dispatched here). - var jm = (JobManager)System.Runtime.CompilerServices.RuntimeHelpers.GetUninitializedObject(typeof(JobManager)); - var jobsField = typeof(JobManager).GetField("_jobs", BindingFlags.NonPublic | BindingFlags.Instance)!; - jobsField.SetValue(jm, Activator.CreateInstance(jobsField.FieldType)); - Set(svc, "_jobManager", jm); - - return new Harness(svc, store, queue, jm); - } - - private static object NewFieldDict(object svc, string field) - { - var t = typeof(OrchestratorService).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)!.FieldType; - return Activator.CreateInstance(t)!; - } - - private static void Set(object target, string field, object? value) => - typeof(OrchestratorService).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)! - .SetValue(target, value); - - private static T Get(object target, string field) => - (T)typeof(OrchestratorService).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)! - .GetValue(target)!; - - private static Task Invoke(OrchestratorService svc, string method, params object[] args) => - (Task)typeof(OrchestratorService).GetMethod(method, BindingFlags.NonPublic | BindingFlags.Instance)! - .Invoke(svc, args)!; - - private static Task Dispatch(Harness h, OrchestratorRun run) => - Invoke(h.Svc, "DispatchPendingTasksAsync", run, TaskFunc, run.Priority, CancellationToken.None, false); - private static Task Redrive(Harness h, OrchestratorRun run) => Invoke(h.Svc, "RedrivePendingTasksAsync", run); - - /// Run the sequential driver to completion. Pre-seeds _finalizingRuns so the background - /// finalize the last step schedules cannot mutate the run under the test's assertions. - private static async Task DriveAsync(Harness h, OrchestratorRun run, string taskPath = TaskFunc, - CancellationToken ct = default) - { - Get>(h.Svc, "_finalizingRuns")[run.Name] = true; - var mi = typeof(OrchestratorService).GetMethod("BuildSequentialRunWork", - BindingFlags.NonPublic | BindingFlags.Instance)!; - var work = (Func)mi.Invoke(h.Svc, [run, taskPath])!; - try { await work(ct); } - catch (TargetInvocationException ex) { throw ex.InnerException ?? ex; } - } - - private static OrchestratorRun MakeRun(string name, int count, bool sequential) - { - var run = new OrchestratorRun - { - Name = name, - Status = "Running", - Priority = 4, - Sequential = sequential, - StartedUtc = DateTime.UtcNow, - TaskScriptName = TaskFunc - }; - for (var i = 0; i < count; i++) - run.Tasks.Add(new OrchestratorTaskItem - { - Id = $"{name}_t{i}", - Status = "Pending", - Sequence = i, - Parameters = new Dictionary { ["FunctionName"] = "Push-Noop", ["idx"] = i } - }); - return run; - } - - private static async Task> QueuedAsync(Harness h, string run) => - (await h.Queue.GetQueuedTaskIdsAsync(run)).OrderBy(x => x, StringComparer.Ordinal).ToList(); - - /// Re-drive re-enqueues via a fire-and-forget Task.Run, so poll for the row to land. - private static async Task> WaitQueuedAsync(Harness h, string run, int expected) - { - for (var i = 0; i < 100; i++) - { - var ids = await h.Queue.GetQueuedTaskIdsAsync(run); - if (ids.Count >= expected) return ids.OrderBy(x => x, StringComparer.Ordinal).ToList(); - await Task.Delay(20); - } - return await QueuedAsync(h, run); - } - - private static string St(OrchestratorRun run, int i) => run.Tasks.Single(t => t.Sequence == i).Status; - - // ─── persistence ──────────────────────────────────────────────────────────────────────────────── - - [Fact] - public async Task Sequential_And_Sequence_SurviveCreationRoundTrip() - { - var h = await NewHarnessAsync(); - var run = MakeRun("persist-seq", 3, sequential: true); - await h.Store.UpsertRunAsync(run); - await h.Store.UpsertTaskBatchAsync(run.Name, run.Tasks); - - var loaded = await h.Store.GetRunAsync(run.Name); - - Assert.NotNull(loaded); - Assert.True(loaded!.Sequential); - Assert.Equal([0, 1, 2], - loaded.Tasks.OrderBy(t => t.Sequence).Select(t => t.Sequence).ToArray()); - } - - [Fact] - public async Task FanOutRun_RoundTripsSequentialFalse() - { - var h = await NewHarnessAsync(); - var run = MakeRun("persist-fanout", 2, sequential: false); - await h.Store.UpsertRunAsync(run); - await h.Store.UpsertTaskBatchAsync(run.Name, run.Tasks); - - var loaded = await h.Store.GetRunAsync(run.Name); - - Assert.False(loaded!.Sequential); - } - - [Fact] - public async Task Sequence_SurvivesStatusWriteRewrite() - { - // The status writer rewrites a task row on every transition with Replace semantics, so Sequence - // must be part of that write or a resumed run would read every task as Sequence 0 and lose order. - var h = await NewHarnessAsync(); - var run = MakeRun("persist-statuswrite", 2, sequential: true); - await h.Store.UpsertRunAsync(run); - await h.Store.UpsertTaskBatchAsync(run.Name, run.Tasks); - - // Simulate a terminal status write (as the coalescing writer does) for the second task. - var t1 = run.Tasks[1]; - await h.Store.WriteTaskStatusBatchAsync( - [ - new TaskStatusWrite(run.Name, t1.Id, "Completed", "{}", 0, null, DateTime.UtcNow, null, t1.Sequence) - ]); - - var loaded = await h.Store.GetRunAsync(run.Name); - var reloaded = loaded!.Tasks.Single(t => t.Id == t1.Id); - Assert.Equal(1, reloaded.Sequence); // not erased to 0 by the Replace write - Assert.Equal("Completed", reloaded.Status); - } - - // ─── dispatch gate ────────────────────────────────────────────────────────────────────────────── - - [Fact] - public async Task Sequential_Dispatch_EnqueuesOnlyTheEntryRow() - { - var h = await NewHarnessAsync(); - var run = MakeRun("disp-seq", 5, sequential: true); - - await Dispatch(h, run); - - var queued = await QueuedAsync(h, run.Name); - Assert.Equal(["disp-seq_t0"], queued); // only Sequence 0, despite five pending tasks - } - - [Fact] - public async Task FanOut_Dispatch_EnqueuesEveryTask() - { - var h = await NewHarnessAsync(); - var run = MakeRun("disp-fanout", 5, sequential: false); - - await Dispatch(h, run); - - var queued = await QueuedAsync(h, run.Name); - Assert.Equal(5, queued.Count); - } - - [Fact] - public async Task Sequential_Dispatch_OnResume_EnqueuesEntryOnly_NeverASecondRow() - { - // Resume shape: tasks 0,1 done; task 2 is the reached task and its queue row SURVIVED the restart; - // task 3 has not been reached. Dispatch must not enqueue task 3 (a second row would let a second - // driver start) — it must leave the already-queued entry alone. - var h = await NewHarnessAsync(); - var run = MakeRun("disp-resume", 4, sequential: true); - run.Tasks[0].Status = "Completed"; - run.Tasks[1].Status = "Completed"; - await h.Queue.EnqueueBatchAsync(run.Name, [(run.Tasks[2].Id, 4)], DateTime.UtcNow); - - await Dispatch(h, run); - - var queued = await QueuedAsync(h, run.Name); - Assert.Equal(["disp-resume_t2"], queued); // task 3 NOT enqueued - } - - [Fact] - public async Task Sequential_Dispatch_OnResume_ReEnqueuesCurrentStep_WhenItsRowIsGone() - { - // The reached step's queue row was lost. Dispatch re-enqueues exactly it, to restart the driver. - var h = await NewHarnessAsync(); - var run = MakeRun("disp-resume-gone", 3, sequential: true); - run.Tasks[0].Status = "Completed"; - - await Dispatch(h, run); - - var queued = await QueuedAsync(h, run.Name); - Assert.Equal(["disp-resume-gone_t1"], queued); - } - - // ─── driver ───────────────────────────────────────────────────────────────────────────────────── - - [Fact] - public async Task Driver_RunsEveryStepInSequenceOrder_OnExactlyOneWorker() - { - var h = await NewHarnessAsync(); - var run = MakeRun("drv-order", 4, sequential: true); - run.Tasks.Reverse(); // insertion order 3,2,1,0 — only the Sequence sort can recover 0,1,2,3 - - await DriveAsync(h, run); - - Assert.Equal(["drv-order_t0", "drv-order_t1", "drv-order_t2", "drv-order_t3"], h.Svc.Order); - Assert.All(run.Tasks, t => Assert.Equal("Completed", t.Status)); - Assert.Equal(1, h.Svc.Checkouts); // one worker for the whole run - Assert.Equal(1, h.Svc.Reclaims); // reclaimed once, at the end - } - - [Fact] - public async Task Driver_BestEffort_ContinuesPastAFailingStep() - { - var h = await NewHarnessAsync(); - var run = MakeRun("drv-besteffort", 4, sequential: true); - h.Svc.FailIds.Add("drv-besteffort_t1"); // the second step throws - - await DriveAsync(h, run); - - // Every step is still attempted, in order, and the failure does not strand the rest. - Assert.Equal(["drv-besteffort_t0", "drv-besteffort_t1", "drv-besteffort_t2", "drv-besteffort_t3"], - h.Svc.Order); - Assert.Equal("Completed", St(run, 0)); - Assert.Equal("Failed", St(run, 1)); - Assert.Equal("Completed", St(run, 2)); - Assert.Equal("Completed", St(run, 3)); - Assert.Equal(1, h.Svc.Checkouts); // still one worker — a step failure does not re-grab a worker - Assert.Equal(1, h.Svc.Reclaims); - } - - [Fact] - public async Task Driver_Cancellation_MarksRemainingStepsCancelled_AndStops() - { - var h = await NewHarnessAsync(); - var run = MakeRun("drv-cancel", 4, sequential: true); - var cancelled = Get>(h.Svc, "_cancelledRuns"); - h.Svc.OnStep = t => { if (t.Sequence == 1) cancelled[run.Name] = true; }; // cancel while step 1 runs - - await DriveAsync(h, run); - - // Steps 0 and 1 completed; the driver noticed the cancellation before step 2 and stopped there. - Assert.Equal(["drv-cancel_t0", "drv-cancel_t1"], h.Svc.Order); - Assert.Equal("Completed", St(run, 0)); - Assert.Equal("Completed", St(run, 1)); - Assert.Equal("Cancelled", St(run, 2)); - Assert.Equal("Cancelled", St(run, 3)); - Assert.Equal(1, h.Svc.Reclaims); // the pinned worker is still reclaimed on the way out - } - - [Fact] - public async Task Driver_DuplicateEntry_IsANoOp_WhileAnotherDriverIsActive() - { - // A duplicate entry row for a run that already has an active driver must not start a second one. - var h = await NewHarnessAsync(); - var run = MakeRun("drv-dup", 3, sequential: true); - Get>(h.Svc, "_activeSequentialDrivers")[run.Name] = true; - - await DriveAsync(h, run); - - Assert.Empty(h.Svc.Order); // ran nothing - Assert.Equal(0, h.Svc.Checkouts); // never grabbed a worker - Assert.All(run.Tasks, t => Assert.Equal("Pending", t.Status)); // left for the real driver - } - - [Fact] - public async Task Driver_ClearsItsRegistration_WhenDone() - { - var h = await NewHarnessAsync(); - var run = MakeRun("drv-cleanup", 2, sequential: true); - - await DriveAsync(h, run); - - // The run must not stay registered, or its re-drive would be suppressed forever. - Assert.False(Get>(h.Svc, "_activeSequentialDrivers") - .ContainsKey(run.Name)); - } - - [Fact] - public async Task Driver_PostExecutionRun_CapturesAndStoresStepOutput() - { - var h = await NewHarnessAsync(); - var run = MakeRun("drv-postexec", 2, sequential: true); - run.PostExecFunctionName = "Push-Aggregate"; // capture path - h.Svc.StepOutput = "{\"ok\":true}"; - - await DriveAsync(h, run); - - var results = await h.Store.GetResultsAsync(run.Name); - Assert.Equal(2, results.Length); - Assert.All(results, r => Assert.Contains("ok", r)); - } - - [Fact] - public async Task ResolveTaskWork_ForSequentialRun_ReturnsAndRunsTheDriver() - { - // Routing: a claimed entry row for a sequential run resolves to the pinned driver (not the parallel - // per-task work), and invoking it drives the whole run on one worker. - var h = await NewHarnessAsync(); - var run = MakeRun("resolve-seq", 3, sequential: true); - Get>(h.Svc, "_activeRuns")[run.Name] = run; - Get>(h.Svc, "_taskScriptPaths")[run.Name] = TaskFunc; - Get>(h.Svc, "_finalizingRuns")[run.Name] = true; - - var mi = typeof(OrchestratorService).GetMethod("ResolveTaskWorkAsync", - BindingFlags.NonPublic | BindingFlags.Instance)!; - var task = (Task)mi.Invoke(h.Svc, [new JobDescriptor(run.Name, run.Tasks[0].Id, 4), CancellationToken.None])!; - await task; - var work = (Func?)task.GetType().GetProperty("Result")!.GetValue(task); - - Assert.NotNull(work); - await work!(CancellationToken.None); - - Assert.Equal(["resolve-seq_t0", "resolve-seq_t1", "resolve-seq_t2"], h.Svc.Order); - Assert.Equal(1, h.Svc.Checkouts); - } - - // ─── re-drive restriction ─────────────────────────────────────────────────────────────────────── - - [Fact] - public async Task Redrive_Sequential_DoesNothing_WhileADriverIsActive() - { - // The driver runs every step inline; the not-yet-reached steps deliberately have no queue row. The - // watchdog must not treat them as orphaned and enqueue them — that would spawn a second driver. - var h = await NewHarnessAsync(); - var run = MakeRun("rd-driver", 3, sequential: true); - Get>(h.Svc, "_activeSequentialDrivers")[run.Name] = true; - - await Redrive(h, run); - await Task.Delay(100); // give any (erroneous) fire-and-forget requeue time to land - - Assert.Empty(await QueuedAsync(h, run.Name)); - } - - [Fact] - public async Task Redrive_Sequential_DoesNothing_WhileTheEntryJobIsStillQueued() - { - // No driver registered yet, but the entry job is still Queued/Running in the JobManager (the window - // between the pump claiming the entry row and the driver registering). The watchdog must still leave - // the run alone. - var h = await NewHarnessAsync(); - var run = MakeRun("rd-entryjob", 3, sequential: true); - var jobs = Get>(h.Jobs, "_jobs"); - jobs[$"{run.Name}-{run.Tasks[0].Id}"] = new JobRecord - { - Id = $"{run.Name}-{run.Tasks[0].Id}", - Name = TaskFunc, - RunName = run.Name, - Priority = 4, - Status = "Running", - QueuedUtc = DateTime.UtcNow - }; - - await Redrive(h, run); - await Task.Delay(100); - - Assert.Empty(await QueuedAsync(h, run.Name)); - } - - private static T Get(JobManager jm, string field) => - (T)typeof(JobManager).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)!.GetValue(jm)!; - - [Fact] - public async Task Redrive_Sequential_NoOp_WhenTheEntryStepStillHasItsRow() - { - // Nothing running, entry step (Sequence 0) is queued and waiting; later steps have no row. The entry - // is dispatchable (row present) so not orphaned, and the later steps must not be touched. - var h = await NewHarnessAsync(); - var run = MakeRun("rd-waiting", 3, sequential: true); - await h.Queue.EnqueueBatchAsync(run.Name, [(run.Tasks[0].Id, 4)], DateTime.UtcNow); - - await Redrive(h, run); - await Task.Delay(100); - - Assert.Equal(["rd-waiting_t0"], await QueuedAsync(h, run.Name)); // unchanged; t1/t2 not enqueued - } - - [Fact] - public async Task Redrive_Sequential_ReDrivesOnlyTheCurrentStep_WhenStalled() - { - // Driver gone and the entry row lost: current step (Sequence 0) has no queue row and no driver is - // active. The watchdog re-enqueues exactly the current step — and none of the not-yet-reached ones — - // so a fresh driver resumes the run. - var h = await NewHarnessAsync(); - var run = MakeRun("rd-stalled", 3, sequential: true); - - await Redrive(h, run); - - var queued = await WaitQueuedAsync(h, run.Name, 1); - Assert.Equal(["rd-stalled_t0"], queued); // only the current; t1/t2 stay unqueued - } - - [Fact] - public async Task Redrive_FanOut_ReEnqueuesAllOrphanedTasks() - { - // The default fan-out behaviour is unchanged: every orphaned (Pending, no queue row) task is - // re-driven, not just the first. - var h = await NewHarnessAsync(); - var run = MakeRun("rd-fanout", 3, sequential: false); - - await Redrive(h, run); - - var queued = await WaitQueuedAsync(h, run.Name, 3); - Assert.Equal(3, queued.Count); - } -} diff --git a/tests/Craft.Tests/OrchestratorStaleRunningTests.cs b/tests/Craft.Tests/OrchestratorStaleRunningTests.cs deleted file mode 100644 index 2d6c5fb..0000000 --- a/tests/Craft.Tests/OrchestratorStaleRunningTests.cs +++ /dev/null @@ -1,207 +0,0 @@ -using System.Collections.Concurrent; -using System.Reflection; -using Craft.Configuration; -using Craft.Orchestration; -using Craft.PowerShellHost; -using Craft.Services; -using Craft.Storage; -using Microsoft.Extensions.Configuration; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// A "Running" status this process did not write must not be trusted as "a worker here has it". -/// -/// The resolver drops a descriptor whose task reads Running, which is right for a duplicate queue row -/// claimed while THIS process is executing the task. But Running also reaches the live graph from -/// storage: the durable pre-invoke marker another process wrote before it died. Two production paths -/// put it there, both seen on a hosted instance across an App Service container swap (old and new -/// containers overlap for tens of seconds to minutes, sharing one storage account): -/// -/// A. The old container creates or advances a run after the new container's startup recovery has -/// already passed. It dies holding claims whose tasks are marked Running. The new container's pump -/// claims one of the run's rows, ResolveTaskWorkAsync rehydrates the run from storage with -/// those markers, and once the dead claims' leases lapse and those rows are claimed, the resolver -/// drops each one as "already running" — Skipped, row deleted. Nothing ever re-drives a Running -/// task, and the run can never finalize (observed: 505/508 for 31 hours). -/// -/// B. The pump starts claiming at host start, before ResumeInterruptedRunsAsync (which waits -/// for the worker pool). Its rehydrated copy goes into _activeRuns first; recovery then -/// loads its OWN copy, flips Running→Pending on that copy and in storage, releases the dead claims -/// and calls DispatchPendingTasksAsync, whose _activeRuns.TryAdd loses to the pump's copy. -/// The live graph keeps the stale Running; the released row is claimed and dropped exactly as in A. -/// -/// Holding the queue claim is the ownership proof: only this process's pump enqueues descriptor jobs, -/// and it only does so for a row it has claimed. So a Running task that no worker in this process -/// started is an interrupted task, and is handled the way recovery handles one — attempt counted -/// (poison bound kept) and run. -/// -public class OrchestratorStaleRunningTests -{ - private const string TaskFunc = "Invoke-CraftTask"; - - private sealed record Harness(OrchestratorService Svc, OrchestratorTableStore Store, JobQueueStore Queue, - ConcurrentDictionary Active); - - private static async Task NewHarnessAsync() - { - var settings = new CraftSettings { Orchestrator = { TablePrefix = "stale" + Guid.NewGuid().ToString("N")[..8] } }; - settings.Orchestrator.BatchStatusWrites = false; - var config = new ConfigurationBuilder().AddInMemoryCollection([]).Build(); - - var backing = new RunRemainingCounterTests.ConditionalStore(); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); - var queue = new JobQueueStore(NullLogger.Instance, settings, backing); - await store.InitializeAsync(); - await queue.InitializeAsync(); - var writer = new OrchestratorStatusWriter(store, NullLogger.Instance, settings); - - var repo = new ScriptRepository(NullLogger.Instance, settings); - typeof(ScriptRepository).GetField("_moduleFunctionNames", BindingFlags.NonPublic | BindingFlags.Instance)! - .SetValue(repo, new HashSet([TaskFunc], StringComparer.OrdinalIgnoreCase)); - var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); - var runner = new PowerShellRunnerService(NullLogger.Instance, pool, repo, settings); - - var svc = (OrchestratorService)System.Runtime.CompilerServices.RuntimeHelpers - .GetUninitializedObject(typeof(OrchestratorService)); - var active = new ConcurrentDictionary(); - Set(svc, "_logger", NullLogger.Instance); - Set(svc, "_store", store); - Set(svc, "_queue", queue); - Set(svc, "_writer", writer); - Set(svc, "_psRunner", runner); - Set(svc, "_settings", settings); - Set(svc, "_lock", new object()); - Set(svc, "_activeRuns", active); - Set(svc, "_taskScriptPaths", new ConcurrentDictionary()); - Set(svc, "_finalizingRuns", new ConcurrentDictionary()); - Set(svc, "_finalizeDeferrals", new ConcurrentDictionary()); - Set(svc, "_childRuns", new ConcurrentDictionary>()); - Set(svc, "_recoveringChildren", new ConcurrentDictionary()); - Set(svc, "_pendingChildRuns", new ConcurrentDictionary()); - Set(svc, "_cancelledRuns", new ConcurrentDictionary()); - Set(svc, "_activeSequentialDrivers", new ConcurrentDictionary()); - Set(svc, "_requeueFailures", new ConcurrentDictionary()); - Set(svc, "_deferrals", NewFieldValue(svc, "_deferrals")); - Set(svc, "_redriveBackoff", NewFieldValue(svc, "_redriveBackoff")); - Set(svc, "_shedParameters", false); - var jm = (JobManager)System.Runtime.CompilerServices.RuntimeHelpers.GetUninitializedObject(typeof(JobManager)); - var jobsField = typeof(JobManager).GetField("_jobs", BindingFlags.NonPublic | BindingFlags.Instance)!; - jobsField.SetValue(jm, Activator.CreateInstance(jobsField.FieldType)); - Set(svc, "_jobManager", jm); - return new Harness(svc, store, queue, active); - } - - private static object NewFieldValue(object svc, string field) => - Activator.CreateInstance(typeof(OrchestratorService) - .GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)!.FieldType)!; - - private static void Set(object target, string field, object? value) => - typeof(OrchestratorService).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)! - .SetValue(target, value); - - private static async Task ResolveAsync(OrchestratorService svc, string run, string taskId) - { - var mi = typeof(OrchestratorService).GetMethod("ResolveTaskWorkAsync", BindingFlags.NonPublic | BindingFlags.Instance)!; - try - { - var task = (Task)mi.Invoke(svc, [new JobDescriptor(run, taskId, 4), CancellationToken.None])!; - await task; - return task.GetType().GetProperty("Result")!.GetValue(task); - } - catch (TargetInvocationException ex) { throw ex.InnerException ?? ex; } - } - - /// Storage as another process left it: its in-flight task carries the durable Running marker. - private static async Task SeedAsync(Harness h, string name, int staleAttempts = 0) - { - var tasks = new List - { - new() { Id = "t-stale", Status = "Running", AttemptCount = staleAttempts, Parameters = new() { ["FunctionName"] = "Push-Noop" } }, - new() { Id = "t-next", Status = "Pending", Parameters = new() { ["FunctionName"] = "Push-Noop" } }, - new() { Id = "t-other", Status = "Pending", Parameters = new() { ["FunctionName"] = "Push-Noop" } }, - }; - await h.Store.UpsertRunAsync(new OrchestratorRun - { - Name = name, - Status = "Running", - Priority = 4, - StartedUtc = DateTime.UtcNow, - TaskScriptName = TaskFunc, - Tasks = tasks - }); - foreach (var t in tasks) await h.Store.UpsertTaskAsync(name, t); - } - - // ── Mode A ───────────────────────────────────────────────────────────────────────────────────── - - [Fact] - public async Task ModeA_RunningMarkerFromADeadProcess_IsRunWhenThisProcessClaimsItsRow() - { - var h = await NewHarnessAsync(); - await SeedAsync(h, "run-a"); - - // The dead process's lease lapsed; this process claimed the row. Previously: null (dropped forever). - var work = await ResolveAsync(h.Svc, "run-a", "t-stale"); - - Assert.NotNull(work); - var task = h.Active["run-a"].Tasks.Single(t => t.Id == "t-stale"); - Assert.Equal(1, task.AttemptCount); // the interrupted attempt is counted, as recovery counts it - } - - [Fact] - public async Task ModeA_PoisonBoundHolds_AThirdInterruptedAttemptFailsTheTask() - { - var h = await NewHarnessAsync(); - await SeedAsync(h, "run-poison", staleAttempts: 2); - - Assert.Null(await ResolveAsync(h.Svc, "run-poison", "t-stale")); - - var task = h.Active["run-poison"].Tasks.Single(t => t.Id == "t-stale"); - Assert.Equal("Failed", task.Status); - } - - // ── Mode B ───────────────────────────────────────────────────────────────────────────────────── - - [Fact] - public async Task ModeB_PumpClaimBeforeRecovery_LeavesAStaleRunningInTheLiveGraph_WhichMustStillRun() - { - var h = await NewHarnessAsync(); - await SeedAsync(h, "run-b"); - - // 1. Host start: the pump claims a sibling row before recovery has run. The rehydrated copy — - // carrying the dead process's Running marker — becomes the live graph. - Assert.NotNull(await ResolveAsync(h.Svc, "run-b", "t-next")); - var live = h.Active["run-b"]; - - // 2. Recovery runs on its own copy: flips t-stale to Pending in storage, releases the claims and - // re-dispatches — but its TryAdd loses to the pump's copy. - await h.Svc.ResumeInterruptedRunsAsync(CancellationToken.None); - Assert.Same(live, h.Active["run-b"]); - - // 3. The released row is claimed. Previously: dropped as "already running", never run again. - Assert.NotNull(await ResolveAsync(h.Svc, "run-b", "t-stale")); - } - - // ── The guard's real job is kept ─────────────────────────────────────────────────────────────── - - [Fact] - public async Task ADuplicateRowForATaskThisProcessIsExecuting_IsStillDropped() - { - var h = await NewHarnessAsync(); - await SeedAsync(h, "run-dup"); - await h.Store.UpsertTaskAsync("run-dup", - new OrchestratorTaskItem { Id = "t-stale", Status = "Pending", Parameters = new() { ["FunctionName"] = "Push-Noop" } }); - - // Start the task the way dispatch does, up to the point it is running on a worker here. - var work = (Func)(await ResolveAsync(h.Svc, "run-dup", "t-stale"))!; - var task = h.Active["run-dup"].Tasks.Single(t => t.Id == "t-stale"); - _ = Task.Run(() => work(CancellationToken.None)); // marks Running, then blocks on worker checkout - for (var i = 0; i < 200 && task.Status != "Running"; i++) await Task.Delay(10); - Assert.Equal("Running", task.Status); - - // A second row for the same task, claimed mid-flight, must not start another copy. - Assert.Null(await ResolveAsync(h.Svc, "run-dup", "t-stale")); - } -} diff --git a/tests/Craft.Tests/OrchestratorTaskLargeParametersAzuriteTests.cs b/tests/Craft.Tests/OrchestratorTaskLargeParametersAzuriteTests.cs index b44a86c..37fd671 100644 --- a/tests/Craft.Tests/OrchestratorTaskLargeParametersAzuriteTests.cs +++ b/tests/Craft.Tests/OrchestratorTaskLargeParametersAzuriteTests.cs @@ -1,19 +1,15 @@ using System.Text.Json; using Craft.Configuration; -using Craft.Orchestration; using Craft.Storage; using Microsoft.Extensions.Logging.Abstractions; namespace Craft.Tests; /// -/// End-to-end guard for the failure this whole change exists for: a scheduled task whose Parameters -/// embed a whole policy template serialize to more than Azure Table's 64 KiB-per-property limit, so the -/// orchestrator's Tasks row used to 400 with PropertyValueTooLarge, get dropped, and the task could never -/// be rehydrated at dispatch ("Parameters could not be rehydrated at dispatch — the Tasks-table row is -/// missing"). With large-entity splitting in the backing store, the task row persists and both read paths -/// the orchestrator uses — GetRunAsync (partition scan) and GetTaskParametersAsync (point read, the -/// dispatch rehydrate) — return the Parameters byte-for-byte. +/// End-to-end guard against a real table backend: a scheduled task whose parameters embed a whole policy +/// template is far over Azure Table's 64 KiB-per-property limit, and so can be a run's PostExecution +/// parameters. The payload row must persist and rehydrate byte-for-byte, and the run must still move through +/// claim, finish and its barrier: those are real entity-group transactions here, which cannot split a row. /// /// Azurite by default, a real account via CRAFT_TEST_TABLE_CONNECTION; skipped, not failed, when neither /// is reachable. @@ -21,7 +17,7 @@ namespace Craft.Tests; [Collection(LargeAllocationSerialTests.Name)] public class OrchestratorTaskLargeParametersAzuriteTests { - private static async Task TryConnectAsync() + private static async Task TryConnectAsync() { var settings = new CraftSettings(); var connection = Environment.GetEnvironmentVariable("CRAFT_TEST_TABLE_CONNECTION"); @@ -43,7 +39,7 @@ public class OrchestratorTaskLargeParametersAzuriteTests return null; } - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); + var store = new WorkStore(NullLogger.Instance, settings, backing); await store.InitializeAsync(); return store; } @@ -56,54 +52,49 @@ public class OrchestratorTaskLargeParametersAzuriteTests }; [Theory] - [InlineData(80_000)] // > 64 KiB per-property: the exact reported failure (column split) + [InlineData(80_000)] // > 64 KiB per-property: column split [InlineData(1_200_000)] // > 1 MiB entity: forces a cross-row split too - public async Task ATaskWhoseParametersExceedTheLimit_PersistsAndRehydrates(int settingsChars) + public async Task ARunWithParametersOverTheLimit_PersistsRehydratesAndCompletes(int settingsChars) { var store = await TryConnectAsync(); if (store == null) return; - const string run = "UserTaskOrchestrator_contoso.com"; var big = new string('T', settingsChars); - var parameters = new Dictionary + var started = DateTime.UtcNow; + var header = new RunHeader { - ["Tenant"] = "contoso.com", - ["Settings"] = big, + RunKey = WorkStore.RunKeyFor("UserTaskOrchestrator_contoso.com", started), + Name = "UserTaskOrchestrator_contoso.com", + Priority = 2, + StartedUtc = started, + TaskScriptName = "Invoke-CraftTask", + PostExecFunctionName = "Agg", + PostExecParametersJson = JsonSerializer.Serialize(new { blob = big }), }; try { - await store.UpsertRunAsync(new OrchestratorRun - { - Name = run, - Status = "Running", - Priority = 2, - StartedUtc = DateTime.UtcNow, - TaskScriptName = "ExecScheduledCommand", - Tasks = [new OrchestratorTaskItem { Id = "task-0", Status = "Pending" }] - }); + await store.CreateRunAsync(header, + [new WorkStore.NewTask("task-0", new() { ["Tenant"] = "contoso.com", ["Settings"] = big })]); - // The exact write path Start-UserTasksOrchestrator uses to enqueue a task's payload. - await store.UpsertTaskBatchAsync(run, new List - { - new() { Id = "task-0", Status = "Pending", Parameters = parameters } - }); - - // Dispatch rehydrate: a point read of the one task's Parameters. - var rehydrated = await store.GetTaskParametersAsync(run, "task-0"); - Assert.NotNull(rehydrated); + var claim = Assert.Single(await store.ClaimAsync(header.RunKey, 10, "w", TimeSpan.FromMinutes(5), true)); + var rehydrated = await store.GetPayloadAsync(header.RunKey, claim.Seq); Assert.Equal("contoso.com", AsString(rehydrated!["Tenant"])); Assert.Equal(big, AsString(rehydrated["Settings"])); - // Whole-run read: the partition scan reassembles the same task. - var loaded = await store.GetRunAsync(run); - Assert.NotNull(loaded); - var task = Assert.Single(loaded!.Tasks); - Assert.Equal(big, AsString(task.Parameters["Settings"])); + var barrier = await store.FinishAsync(header.RunKey, [new WorkStore.Finish(claim.Seq, "Completed", Owner: "w")]); + Assert.True(barrier!.ReachedBarrier); + Assert.Equal(header.PostExecParametersJson, await store.GetPostExecParametersAsync(header.RunKey)); + + var aggregate = Assert.Single(await store.ClaimAsync(header.RunKey, 10, "w", TimeSpan.FromMinutes(5), true)); + Assert.Equal(WorkStore.AggregateSeq, aggregate.Seq); + var done = await store.FinishAsync(header.RunKey, [new WorkStore.Finish(aggregate.Seq, "Completed", Owner: "w")]); + Assert.True(done!.Completed); + Assert.Equal("Completed", done.Header.Status); } finally { - await store.CleanupRunAsync(run); + await store.DeleteRunAsync(header.RunKey); } } } diff --git a/tests/Craft.Tests/OrchestratorTaskPathRehydrationTests.cs b/tests/Craft.Tests/OrchestratorTaskPathRehydrationTests.cs deleted file mode 100644 index 0684b09..0000000 --- a/tests/Craft.Tests/OrchestratorTaskPathRehydrationTests.cs +++ /dev/null @@ -1,123 +0,0 @@ -using System.Collections.Concurrent; -using System.Reflection; -using Craft.Configuration; -using Craft.Orchestration; -using Craft.PowerShellHost; -using Craft.Services; -using Craft.Storage; -using Microsoft.Extensions.Configuration; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// The resolver must rebuild a task's script from the run record when the in-memory path cache misses, -/// instead of dropping the task. -/// -/// The cache (_taskScriptPaths) is written only by DispatchPendingTasksAsync. The JobQueuePump, a -/// BackgroundService, starts claiming persisted queue rows at host start — before the scheduler has -/// even waited for the worker pool, let alone run ResumeInterruptedRunsAsync, which is what dispatches -/// (and so caches the path for) a resumed run. In that window every claimed row resolved to a cache -/// miss and the task was dropped: the resolver returned null, the JobManager marked the job Skipped, -/// and the pump deleted the queue row — permanently, for a task still Pending in a run the pump would -/// never see re-queued. -/// -/// The run record persists TaskScriptName, and the ScriptRepository is fully loaded before the pump's -/// first claim, so the path is rebuildable from storage exactly as the resume path rebuilds it. These -/// tests pin that: a cache miss on a live run rehydrates and caches the path; only a run with no -/// resolvable script at all is still dropped. -/// -public class OrchestratorTaskPathRehydrationTests -{ - private static (OrchestratorService Svc, OrchestratorTableStore Store, ConcurrentDictionary Paths) - NewService(params string[] knownModuleFunctions) - { - var settings = new CraftSettings - { - Orchestrator = { TablePrefix = "rehyd" + Guid.NewGuid().ToString("N")[..6] } - }; - var config = new ConfigurationBuilder().AddInMemoryCollection([]).Build(); - - var repo = new ScriptRepository(NullLogger.Instance, settings); - // IsModuleFunction short-circuits on a non-null _moduleFunctionNames, so FindScript resolves - // these names without a ScriptBlock or a module on disk. - typeof(ScriptRepository).GetField("_moduleFunctionNames", BindingFlags.NonPublic | BindingFlags.Instance)! - .SetValue(repo, new HashSet(knownModuleFunctions, StringComparer.OrdinalIgnoreCase)); - - var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); - var runner = new PowerShellRunnerService(NullLogger.Instance, pool, repo, settings); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, - new RunRemainingCounterTests.ConditionalStore()); - var paths = new ConcurrentDictionary(); - - var svc = (OrchestratorService)System.Runtime.CompilerServices.RuntimeHelpers - .GetUninitializedObject(typeof(OrchestratorService)); - Set(svc, "_logger", NullLogger.Instance); - Set(svc, "_store", store); - Set(svc, "_psRunner", runner); - Set(svc, "_activeRuns", new ConcurrentDictionary()); - Set(svc, "_taskScriptPaths", paths); // deliberately EMPTY — the pump-before-resume window - Set(svc, "_lock", new object()); - return (svc, store, paths); - } - - private static void Set(object target, string field, object value) => - typeof(OrchestratorService).GetField(field, BindingFlags.NonPublic | BindingFlags.Instance)! - .SetValue(target, value); - - private static async Task ResolveAsync(OrchestratorService svc, string runName, string taskId) - { - var mi = typeof(OrchestratorService).GetMethod("ResolveTaskWorkAsync", - BindingFlags.NonPublic | BindingFlags.Instance)!; - try - { - var task = (Task)mi.Invoke(svc, [new JobDescriptor(runName, taskId, 4), CancellationToken.None])!; - await task; - return task.GetType().GetProperty("Result")!.GetValue(task); - } - catch (TargetInvocationException ex) - { - throw ex.InnerException ?? ex; - } - } - - private static async Task SeedRunAsync(OrchestratorTableStore store, string name, string taskScriptName) - { - await store.InitializeAsync(); - await store.UpsertRunAsync(new OrchestratorRun - { - Name = name, - Status = "Running", - Priority = 4, - StartedUtc = DateTime.UtcNow, - TaskScriptName = taskScriptName, - Tasks = [new OrchestratorTaskItem { Id = "task-0", Status = "Pending" }] - }); - await store.UpsertTaskAsync(name, new OrchestratorTaskItem { Id = "task-0", Status = "Pending" }); - } - - [Fact] - public async Task CacheMissOnALiveRun_RehydratesThePathFromTaskScriptName_AndCachesIt() - { - var (svc, store, paths) = NewService("Invoke-CraftTask"); - await SeedRunAsync(store, "AuditLogProcessV2-contoso.com", "Invoke-CraftTask"); - - var work = await ResolveAsync(svc, "AuditLogProcessV2-contoso.com", "task-0"); - - Assert.NotNull(work); // previously: dropped, because _taskScriptPaths had no entry yet - Assert.Equal("Invoke-CraftTask", paths["AuditLogProcessV2-contoso.com"]); - } - - [Fact] - public async Task ARunWithNoResolvableTaskScript_IsStillDropped() - { - // Empty TaskScriptName, and the naming-convention fallback (Invoke-Task) resolves to nothing - // because the repo knows no such function. This is the only case that should still drop. - var (svc, store, _) = NewService(/* no known functions */); - await SeedRunAsync(store, "mysteryrun", taskScriptName: ""); - - var work = await ResolveAsync(svc, "mysteryrun", "task-0"); - - Assert.Null(work); - } -} diff --git a/tests/Craft.Tests/PowerShellWorkerOutputTests.cs b/tests/Craft.Tests/PowerShellWorkerOutputTests.cs new file mode 100644 index 0000000..11a4f9f --- /dev/null +++ b/tests/Craft.Tests/PowerShellWorkerOutputTests.cs @@ -0,0 +1,49 @@ +using System.Management.Automation; +using System.Management.Automation.Runspaces; +using Craft.PowerShellHost; +using Microsoft.Extensions.Logging.Abstractions; + +namespace Craft.Tests; + +/// +/// A worker is reused across invocations, and an invocation that throws must not cost the next one its output. +/// It did: after a failed asynchronous invocation, the PowerShell object's own output buffer came back null from +/// the next EndInvoke, so that call returned nothing. A sequential run lost the result of the step after every +/// failed step; any reused worker lost one call's output after any failure. +/// +public class PowerShellWorkerOutputTests +{ + private static async Task NewWorkerAsync() + { + var iss = InitialSessionState.CreateDefault2(); + iss.Commands.Add(new SessionStateFunctionEntry("Step", "param($N) if ($N -eq 2) { throw \"boom $N\" }; \"out$N\"")); + var worker = new PowerShellWorker(96, iss, NullLogger.Instance); + worker.Runspace.ThreadOptions = PSThreadOptions.ReuseThread; + if (worker.Runspace.RunspaceStateInfo.State == RunspaceState.BeforeOpen) worker.Runspace.Open(); + await worker.InvokeScriptAsync(ScriptBlock.Create("$null")); + return worker; + } + + [Fact] + public async Task TheCallAfterAFailedCall_StillReturnsItsOutput() + { + using var worker = await NewWorkerAsync(); + var outputs = new List(); + for (var i = 0; i < 5; i++) + { + try { outputs.Add(string.Join(",", (await worker.InvokeAsync("Step", new() { ["N"] = i })).Select(o => o.ToString()))); } + catch (RuntimeException) { outputs.Add("threw"); } + } + + Assert.Equal(["out0", "out1", "threw", "out3", "out4"], outputs); + } + + [Fact] + public async Task TheScriptAfterAFailedScript_StillReturnsItsOutput() + { + using var worker = await NewWorkerAsync(); + await Assert.ThrowsAsync(() => worker.InvokeScriptAsync(ScriptBlock.Create("throw 'boom'"))); + + Assert.Equal("after", Assert.Single(await worker.InvokeScriptAsync(ScriptBlock.Create("'after'"))).ToString()); + } +} diff --git a/tests/Craft.Tests/RunRemainingCounterTests.cs b/tests/Craft.Tests/RunRemainingCounterTests.cs deleted file mode 100644 index 38a287e..0000000 --- a/tests/Craft.Tests/RunRemainingCounterTests.cs +++ /dev/null @@ -1,345 +0,0 @@ -using Craft.Configuration; -using Craft.Orchestration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// The remaining-task counter that makes run completion a fact in storage rather than a property of one -/// process's memory. -/// -/// Completion is decided today by CheckRunCompletion walking run.Tasks in memory, which is -/// why ResolveTaskWorkAsync documents object identity with the live run graph as a requirement -/// rather than an optimization — a worker handed a deserialized copy would mutate something the graph -/// never sees and the run would never finalize. A counter in storage removes that requirement. -/// -/// Its one hard property is being decremented exactly once per task. The status writer retries runs -/// whose batch failed, so anything that decrements separately from the task's own terminal write gets -/// replayed and double-counts, stranding a run that has actually finished. -/// -public class RunRemainingCounterTests -{ - /// - /// A fake that models the two things the counter depends on: ETag concurrency, and transactions that - /// are all-or-nothing within a partition. - /// - internal sealed class ConditionalStore : ICraftTableStore - { - private readonly Dictionary> _tables = new(); - private long _etag; - - public int ConditionalWrites { get; private set; } - public int RejectedWrites { get; private set; } - - /// Set to run just before a conditional write lands — lets a test interleave a competitor. - public Action? OnBeforeConditionalWrite { get; set; } - - /// Set to run at the start of any table query — lets a test model unreachable storage. - public Action? OnBeforeQuery { get; set; } - - private Dictionary<(string, string), StoreRow> Table(string t) => - _tables.TryGetValue(t, out var x) ? x : _tables[t] = new(); - - private StoreRow Stamp(StoreRow row) => new(row.PartitionKey, row.RowKey) - { - ETag = $"W/\"{Interlocked.Increment(ref _etag)}\"", - Properties = new Dictionary(row.Properties), - }; - - - /// - /// Azure Table returns rows ordered by partition key then row key, and the queue's whole - /// priority scheme rests on that. A fake handing back insertion order would let an ordering bug - /// pass, so model the real contract. - /// - /// Copies, like GetAsync: yielding the stored instances would let a caller that mutates what it - /// read — which claiming does, by design — change storage without ever writing. - /// - private List Ordered(string table) => Table(table).Values - .OrderBy(r => r.PartitionKey, StringComparer.Ordinal) - .ThenBy(r => r.RowKey, StringComparer.Ordinal) - .Select(r => new StoreRow(r.PartitionKey, r.RowKey) - { - ETag = r.ETag, - Properties = new Dictionary(r.Properties), - }) - .ToList(); - - public Task PingAsync(CancellationToken ct = default) => Task.CompletedTask; - public Task EnsureTableAsync(string table, CancellationToken ct = default) { Table(table); return Task.CompletedTask; } - - public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) - { - Table(table)[(row.PartitionKey, row.RowKey)] = Stamp(row); - return Task.CompletedTask; - } - - public Task UpsertBatchAsync(string table, string partitionKey, IReadOnlyList rows, CancellationToken ct = default) - { - foreach (var r in rows) Table(table)[(r.PartitionKey, r.RowKey)] = Stamp(r); - return Task.CompletedTask; - } - - public Task TryReplaceBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) - { - OnBeforeConditionalWrite?.Invoke(); - - var t = Table(table); - // All-or-nothing: verify every guard BEFORE mutating anything. - foreach (var r in rows) - { - if (!t.TryGetValue((r.PartitionKey, r.RowKey), out var cur) || cur.ETag != r.ETag) - { - RejectedWrites++; - return Task.FromResult(false); - } - } - - foreach (var r in rows) t[(r.PartitionKey, r.RowKey)] = Stamp(r); - ConditionalWrites++; - return Task.FromResult(true); - } - - public Task GetAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) - { - // Hand back a copy: a caller mutating what it read must not mutate the store in place. - if (!Table(table).TryGetValue((partitionKey, rowKey), out var r)) return Task.FromResult(null); - return Task.FromResult(new StoreRow(r.PartitionKey, r.RowKey) - { - ETag = r.ETag, - Properties = new Dictionary(r.Properties), - }); - } - - public async IAsyncEnumerable QueryPartitionAsync(string table, string partitionKey, - [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) - { - foreach (var r in Ordered(table).Where(r => r.PartitionKey == partitionKey)) - { - yield return r; - await Task.Yield(); - } - } - - public async IAsyncEnumerable QueryTableAsync(string table, - [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken ct = default) - { - OnBeforeQuery?.Invoke(); - foreach (var r in Ordered(table)) { yield return r; await Task.Yield(); } - } - - public Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) - { - Table(table).Remove((partitionKey, rowKey)); - return Task.CompletedTask; - } - - public Task DeletePartitionAsync(string table, string partitionKey, CancellationToken ct = default) - { - foreach (var k in Table(table).Keys.Where(k => k.Item1 == partitionKey).ToList()) Table(table).Remove(k); - return Task.CompletedTask; - } - } - - private const string Run = "StandardsApply"; - - private static (OrchestratorTableStore Store, ConditionalStore Backing) NewStore() - { - var backing = new ConditionalStore(); - var settings = new CraftSettings(); - return (new OrchestratorTableStore(NullLogger.Instance, settings, backing), backing); - } - - private static OrchestratorTaskItem Task_(string id, string status = "Completed") => - new() { Id = id, Status = status, CompletedUtc = DateTime.UtcNow }; - - private static async Task SeededAsync(ConditionalStore backing, int taskCount) - { - var settings = new CraftSettings(); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); - await store.InitializeAsync(); - - var tasks = Enumerable.Range(0, taskCount).Select(i => Task_($"task-{i}", "Pending")).ToList(); - await store.UpsertTaskBatchAsync(Run, tasks); - await store.InitRemainingAsync(Run, taskCount); - return store; - } - - /// - /// The counter shares the task partition so it can ride the same transaction. Nothing that reads - /// tasks may mistake it for one — a phantom task never completes, and the run never finalizes. - /// - [Fact] - public async Task CounterRowIsNotReturnedAsATask() - { - var backing = new ConditionalStore(); - var store = await SeededAsync(backing, 4); - await store.UpsertRunAsync(new OrchestratorRun { Name = Run, Priority = 4 }); - - var run = await store.GetRunAsync(Run); - - Assert.NotNull(run); - Assert.Equal(4, run!.Tasks.Count); - Assert.DoesNotContain(run.Tasks, t => t.Id.Contains("counter", StringComparison.OrdinalIgnoreCase)); - } - - /// - /// The live path. Terminal transitions reach storage through the coalescing status writer in - /// batches, so the counter has to fall by the number of terminal rows in a batch rather than by one - /// per call — routing each completion through the single-task primitive would be three round-trips - /// per task, roughly 22,000 for the run that prompted this work. - /// - /// Non-terminal rows in the same batch (a "Running" marker) must not count. - /// - [Fact] - public async Task BatchedTerminalWritesDecrementByTheirTerminalCount() - { - var backing = new ConditionalStore(); - var store = await SeededAsync(backing, 10); - - var writes = new List - { - new(Run, "task-0", "Completed", null, 1, null, DateTime.UtcNow, null), - new(Run, "task-1", "Failed", null, 1, "boom", DateTime.UtcNow, null), - new(Run, "task-2", "Cancelled", null, 1, null, DateTime.UtcNow, null), - new(Run, "task-3", "Running", null, 1, null, null, null), // not terminal — must not count - }; - - var failed = await store.WriteTaskStatusBatchAsync(writes); - - Assert.Empty(failed); - Assert.Equal(7, await store.GetRemainingAsync(Run)); - } - - [Fact] - public async Task MissingCounterReportsNullRatherThanGuessing() - { - var (store, _) = NewStore(); - await store.InitializeAsync(); - - Assert.Null(await store.GetRemainingAsync("never-seeded")); - Assert.Null(await store.DecrementRemainingAsync("never-seeded", 1)); - } - - // ─── Reconciliation (the lost-decrement repair) ─── - - /// - /// A decrement that exhausts its retries is never re-sent — the terminal rows landed, the counter - /// didn't move, and from then on it overstates the run's outstanding work forever. Production - /// symptom: "complete in memory but storage shows N outstanding - deferring finalize" on every 60s - /// tick for the life of the process. Reconcile recounts the rows and repairs the counter. - /// - [Fact] - public async Task ReconcileRepairsALostDecrement() - { - var backing = new ConditionalStore(); - var store = await SeededAsync(backing, 3); - - // Terminal rows written WITHOUT the counter moving — exactly what a lost decrement leaves. - await store.UpsertTaskAsync(Run, Task_("task-0")); - await store.UpsertTaskAsync(Run, Task_("task-1", "Failed")); - Assert.Equal(3, await store.GetRemainingAsync(Run)); - - Assert.Equal(1, await store.ReconcileRemainingAsync(Run)); - Assert.Equal(1, await store.GetRemainingAsync(Run)); - } - - [Fact] - public async Task ReconcileWithoutDriftChangesNothing() - { - var backing = new ConditionalStore(); - var store = await SeededAsync(backing, 2); - - var writesBefore = backing.ConditionalWrites; - - Assert.Equal(2, await store.ReconcileRemainingAsync(Run)); - Assert.Equal(writesBefore, backing.ConditionalWrites); - } - - [Fact] - public async Task ReconcileWithoutACounterRowIsANoOp() - { - var (store, _) = NewStore(); - await store.InitializeAsync(); - - Assert.Null(await store.ReconcileRemainingAsync("never-seeded")); - } - - /// - /// A decrement landing between reconcile's recount and its write must win. The recount is stale the - /// moment a competitor moves the counter, so the ETag guard has to reject the repair — reporting - /// null sends the caller back around rather than letting an old count overwrite fresh progress. - /// - [Fact] - public async Task ReconcileLosingARaceDoesNotClobberTheCompetitor() - { - var backing = new ConditionalStore(); - var store = await SeededAsync(backing, 3); - - // Manufacture drift so reconcile attempts a write at all. - await store.UpsertTaskAsync(Run, Task_("task-0")); - - backing.OnBeforeConditionalWrite = () => - { - backing.OnBeforeConditionalWrite = null; - store.DecrementRemainingAsync(Run, 1).GetAwaiter().GetResult(); - }; - - Assert.Null(await store.ReconcileRemainingAsync(Run)); - // The competitor's decrement survived: 3 seeded − 1 decremented-by-competitor. - Assert.Equal(2, await store.GetRemainingAsync(Run)); - } - - // ─── Status-guarded cancel (the cancel-a-run write) ─── - - [Fact] - public async Task CancellingAPendingTaskDecrementsTheCounter() - { - var backing = new ConditionalStore(); - var store = await SeededAsync(backing, 3); - await store.UpsertRunAsync(new OrchestratorRun { Name = Run, Priority = 4 }); - - var result = await store.CancelPendingTaskAsync(Run, Task_("task-0", "Cancelled")); - - Assert.True(result.Cancelled); - Assert.Equal(2, await store.GetRemainingAsync(Run)); - Assert.Equal("Cancelled", (await store.GetRunAsync(Run))?.Tasks.Single(t => t.Id == "task-0").Status); - } - - /// - /// The guard that makes cancel safe against dispatch. A task that moved Pending → Running between - /// the caller's read and this write must be left alone: clobbering it would have the task's real - /// completion decrement the counter a second time, and the run would finalize with work - /// outstanding. - /// - [Fact] - public async Task CancellingATaskThatStartedRunningIsRefused() - { - var backing = new ConditionalStore(); - var store = await SeededAsync(backing, 3); - await store.UpsertRunAsync(new OrchestratorRun { Name = Run, Priority = 4 }); - await store.UpsertTaskAsync(Run, Task_("task-0", "Running")); - - var result = await store.CancelPendingTaskAsync(Run, Task_("task-0", "Cancelled")); - - Assert.False(result.Cancelled); - Assert.Equal("Running", result.CurrentStatus); - Assert.Equal(3, await store.GetRemainingAsync(Run)); - Assert.Equal("Running", (await store.GetRunAsync(Run))?.Tasks.Single(t => t.Id == "task-0").Status); - } - - [Fact] - public async Task CancellingOnAPreCounterRunStillWritesTheStatus() - { - var (store, _) = NewStore(); - await store.InitializeAsync(); - await store.UpsertTaskBatchAsync("old-run", [Task_("task-0", "Pending")]); - - var result = await store.CancelPendingTaskAsync("old-run", Task_("task-0", "Cancelled")); - - Assert.True(result.Cancelled); - Assert.Null(await store.GetRemainingAsync("old-run")); - } -} diff --git a/tests/Craft.Tests/StartupClaimGateTests.cs b/tests/Craft.Tests/StartupClaimGateTests.cs deleted file mode 100644 index 7f9a9ec..0000000 --- a/tests/Craft.Tests/StartupClaimGateTests.cs +++ /dev/null @@ -1,140 +0,0 @@ -using System.Reflection; -using System.Runtime.CompilerServices; -using Craft.Configuration; -using Craft.Endpoints; -using Craft.Orchestration; -using Craft.PowerShellHost; -using Craft.Services; -using Craft.Storage; -using Microsoft.Extensions.Configuration; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// The pump claims nothing until startup recovery is done, and every way startup can end opens the gate. -/// -/// A claim taken before recovery rehydrates its run from storage — the previous process's Running markers -/// included — into the live graph. Recovery then resets those markers on its own copy, which loses the -/// _activeRuns race, so the live graph keeps the stale Running (see OrchestratorStaleRunningTests, -/// mode B). Gating the first claim on recovery removes the race; the resolver's ownership guard stays as -/// defense in depth. -/// -public class StartupClaimGateTests -{ - /// An orchestrator carrying only the gate. Field initializers do not run on an uninitialized - /// object, so the gate is installed by hand; every other field stays null. - private static OrchestratorService GatedOrchestrator() - { - var svc = (OrchestratorService)RuntimeHelpers.GetUninitializedObject(typeof(OrchestratorService)); - typeof(OrchestratorService).GetField("_recoveryDone", BindingFlags.NonPublic | BindingFlags.Instance)! - .SetValue(svc, new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously)); - typeof(OrchestratorService).GetField("_logger", BindingFlags.NonPublic | BindingFlags.Instance)! - .SetValue(svc, NullLogger.Instance); - return svc; - } - - private static (JobQueuePump Pump, JobManager Jobs) NewPump(OrchestratorService orchestrator) - { - var settings = new CraftSettings(); - settings.Worker.BgPoolSize = 4; - var config = new ConfigurationBuilder().AddInMemoryCollection(new Dictionary - { - ["JobQueuePollIntervalMs"] = "100", - }).Build(); - - var queue = new JobQueueStore(NullLogger.Instance, settings, new RunRemainingCounterTests.ConditionalStore()); - queue.InitializeAsync().GetAwaiter().GetResult(); - queue.EnqueueBatchAsync("StandardsApply", - Enumerable.Range(0, 20).Select(i => ($"task-{i:D3}", 4)).ToList(), DateTime.UtcNow).GetAwaiter().GetResult(); - - var repo = new ScriptRepository(NullLogger.Instance, settings); - var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); - var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); - var jobs = new JobManager(NullLogger.Instance, settings, limiter); - return (new JobQueuePump(NullLogger.Instance, queue, jobs, config, settings, orchestrator), jobs); - } - - private static async Task WaitUntil(Func condition, int timeoutMs = 5000) - { - var deadline = Environment.TickCount64 + timeoutMs; - while (Environment.TickCount64 < deadline) - { - if (condition()) return true; - await Task.Delay(20); - } - return condition(); - } - - [Fact] - public async Task PumpClaimsNothingBeforeRecovery_AndClaimsNormallyAfter() - { - var orchestrator = GatedOrchestrator(); - var (pump, jobs) = NewPump(orchestrator); - - await pump.StartAsync(CancellationToken.None); - try - { - await Task.Delay(500); // five poll intervals with rows waiting - Assert.Equal(0, jobs.QueuedCount); - - orchestrator.MarkRecoveryDone(); - - Assert.True(await WaitUntil(() => jobs.QueuedCount > 0), "the pump did not claim once recovery was done"); - } - finally - { - await Task.WhenAny(pump.StopAsync(CancellationToken.None), Task.Delay(3000)); - } - } - - private static SchedulerService NewScheduler(OrchestratorService orchestrator, bool poolReady) - { - var settings = new CraftSettings(); - var config = new ConfigurationBuilder().AddInMemoryCollection([]).Build(); - var repo = new ScriptRepository(NullLogger.Instance, settings); - var pool = new PowerShellWorkerPool(repo, NullLogger.Instance, config, settings); - if (poolReady) pool.Initialize(enableHttp: false, enableBg: false); // signals ready, builds nothing - var limiter = new BackgroundTaskLimiter(NullLogger.Instance, config, settings, pool); - var runner = new PowerShellRunnerService(NullLogger.Instance, pool, repo, settings); - var health = new StorageHealthMonitor(new RunRemainingCounterTests.ConditionalStore(), NullLogger.Instance); - return new SchedulerService(NullLogger.Instance, runner, limiter, orchestrator, - new JobManager(NullLogger.Instance, settings, limiter), settings, pool, health, - NativeScheduledTasks.Empty, null!); - } - - [Fact] - public async Task ARecoveryThatThrows_StillOpensTheGate() - { - // The gated orchestrator has no store, so ResumeInterruptedRunsAsync throws on its first line. - var orchestrator = GatedOrchestrator(); - var scheduler = NewScheduler(orchestrator, poolReady: true); - using var cts = new CancellationTokenSource(); - - await scheduler.StartAsync(cts.Token); - try - { - Assert.True(await Task.WhenAny(orchestrator.RecoveryDone, Task.Delay(10_000)) == orchestrator.RecoveryDone, - "a failed recovery left the claim gate shut — the pump would never claim"); - } - finally - { - cts.Cancel(); - try { await scheduler.StopAsync(CancellationToken.None); } catch { /* the gutted orchestrator may fault the loop */ } - } - } - - [Fact] - public async Task ShutdownBeforeTheWorkerPoolIsReady_StillOpensTheGate() - { - var orchestrator = GatedOrchestrator(); - var scheduler = NewScheduler(orchestrator, poolReady: false); - using var cts = new CancellationTokenSource(); - - await scheduler.StartAsync(cts.Token); - cts.Cancel(); - try { await scheduler.StopAsync(CancellationToken.None); } catch { /* cancellation */ } - - Assert.True(await Task.WhenAny(orchestrator.RecoveryDone, Task.Delay(5_000)) == orchestrator.RecoveryDone); - } -} diff --git a/tests/Craft.Tests/StatusWriterDurabilityTests.cs b/tests/Craft.Tests/StatusWriterDurabilityTests.cs deleted file mode 100644 index 3b8b246..0000000 --- a/tests/Craft.Tests/StatusWriterDurabilityTests.cs +++ /dev/null @@ -1,499 +0,0 @@ -using System.Runtime.CompilerServices; -using Craft.Configuration; -using Craft.Orchestration; -using Craft.Storage; -using Microsoft.Extensions.Logging.Abstractions; - -namespace Craft.Tests; - -/// -/// The status writer sits on the critical path between a task being dispatched and that task checking -/// out a worker. Production wedged twice in one morning — 101 and 75 minutes — with all 8 limiter slots -/// held by tasks blocked on its barrier, every one of the 8 BG workers idle, 1,919 jobs queued and the -/// heap at 24% of its cap. The worker-health snapshot read Jobs.Running=8 / BgPool.BusyCount=0, which is -/// only reachable if the jobs never got as far as PowerShell. -/// -/// These tests pin the two properties that failure needed: -/// LIVENESS — no wait here is unbounded, and the drain loop cannot die. -/// DURABILITY — nothing bounded is ever dropped. A write that does not persist is retried; a task that -/// cannot be marked Running is deferred, never failed, and stays Pending in storage. -/// -public class StatusWriterDurabilityTests -{ - /// An in-memory store whose writes can be made to hang or fail on demand. - private sealed class ControllableStore : ICraftTableStore - { - - // Claims are not exercised by this fake. Fail loudly rather than pretend the guard held — - // a silent 'true' here would look exactly like a successful claim. - public Task TryReplaceBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) => throw new NotSupportedException(); - private readonly Dictionary> _tables = new(); - private readonly object _sync = new(); - - public ManualResetEventSlim BatchGate { get; } = new(initialState: true); - public volatile bool FailBatches; - - public int BatchCalls; - public int MaxConcurrentBatches; - private int _inFlight; - - /// Single-row upserts. Counted separately from batches because a flush that writes N rows - /// one at a time costs N round-trips no matter how fast each one is — the cost the batch path exists - /// to avoid. - public int SingleUpsertCalls; - - public List Rows(string table) - { - lock (_sync) return _tables.TryGetValue(table, out var t) ? t.Values.ToList() : new List(); - } - - public Task PingAsync(CancellationToken ct = default) => Task.CompletedTask; - - public Task EnsureTableAsync(string table, CancellationToken ct = default) - { - lock (_sync) { if (!_tables.ContainsKey(table)) _tables[table] = new(); } - return Task.CompletedTask; - } - - public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) - { - // Counted, not timed: the single-vs-batch distinction the tests assert on is the NUMBER of - // round-trips, so this needs no simulated latency. - Interlocked.Increment(ref SingleUpsertCalls); - lock (_sync) - { - if (!_tables.ContainsKey(table)) _tables[table] = new(); - _tables[table][(row.PartitionKey, row.RowKey)] = row; - } - return Task.CompletedTask; - } - - public async Task UpsertBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) - { - await Task.Yield(); // a real storage call never completes synchronously - Interlocked.Increment(ref BatchCalls); - var now = Interlocked.Increment(ref _inFlight); - InterlockedMax(ref MaxConcurrentBatches, now); - try - { - // Block here to model a stalled storage call, honouring cancellation so a flush timeout works. - while (!BatchGate.IsSet) - { - ct.ThrowIfCancellationRequested(); - await Task.Delay(5, ct); - } - if (FailBatches) throw new InvalidOperationException("storage unavailable"); - - lock (_sync) - { - if (!_tables.ContainsKey(table)) _tables[table] = new(); - foreach (var r in rows) _tables[table][(r.PartitionKey, r.RowKey)] = r; - } - } - finally - { - Interlocked.Decrement(ref _inFlight); - } - } - - public Task GetAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) - { - lock (_sync) - return Task.FromResult(_tables.TryGetValue(table, out var t) - && t.TryGetValue((partitionKey, rowKey), out var r) ? r : null); - } - - public async IAsyncEnumerable QueryPartitionAsync(string table, string partitionKey, - [EnumeratorCancellation] CancellationToken ct = default) - { - List snapshot; - lock (_sync) - snapshot = _tables.TryGetValue(table, out var t) - ? t.Where(k => k.Key.Item1 == partitionKey).Select(k => k.Value).ToList() - : new List(); - foreach (var r in snapshot) { yield return r; await Task.Yield(); } - } - - public async IAsyncEnumerable QueryTableAsync(string table, - [EnumeratorCancellation] CancellationToken ct = default) - { - List snapshot; - lock (_sync) snapshot = _tables.TryGetValue(table, out var t) ? t.Values.ToList() : new List(); - foreach (var r in snapshot) { yield return r; await Task.Yield(); } - } - - public Task DeleteAsync(string table, string partitionKey, string rowKey, CancellationToken ct = default) - { - lock (_sync) { if (_tables.TryGetValue(table, out var t)) t.Remove((partitionKey, rowKey)); } - return Task.CompletedTask; - } - - public Task DeletePartitionAsync(string table, string partitionKey, CancellationToken ct = default) - { - lock (_sync) - { - if (!_tables.TryGetValue(table, out var t)) return Task.CompletedTask; - foreach (var k in t.Keys.Where(k => k.Item1 == partitionKey).ToList()) t.Remove(k); - } - return Task.CompletedTask; - } - - private static void InterlockedMax(ref int target, int value) - { - int cur; - while (value > (cur = Volatile.Read(ref target))) - if (Interlocked.CompareExchange(ref target, value, cur) == cur) return; - } - } - - private static (OrchestratorStatusWriter Writer, ControllableStore Backing) NewWriter( - int barrierTimeoutSec = 2, int flushTimeoutSec = 1, int concurrency = 8) - { - var settings = new CraftSettings(); - settings.Orchestrator.RunningBarrierTimeoutSeconds = barrierTimeoutSec; - settings.Orchestrator.StatusFlushTimeoutSeconds = flushTimeoutSec; - settings.Orchestrator.StatusFlushConcurrency = concurrency; - - var backing = new ControllableStore(); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); - var writer = new OrchestratorStatusWriter(store, NullLogger.Instance, settings); - return (writer, backing); - } - - private static OrchestratorTaskItem Task1(string id = "t1") => - new() { Id = id, Status = "Running", Parameters = new Dictionary { ["TenantFilter"] = "x.com" } }; - - // ── R2: COALESCED SMALL-RESULT WRITES ───────────────────────────────────────────────────────────── - - /// - /// A small result rides the writer and is durable after a flush; one too large for a single table - /// property is refused, so the caller keeps the chunked StoreResultAsync path. - /// - [Fact] - public async Task SmallResult_Coalesces_AndIsDurable_WhileLargeResultIsRefused() - { - var settings = new CraftSettings(); - var backing = new ControllableStore(); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); - using var writer = new OrchestratorStatusWriter(store, NullLogger.Instance, settings); - - Assert.True(writer.TryQueueResult("run", "t1", "{\"ok\":true}")); - // 30 001 chars is one over the single-property bound — must fall back to the direct chunked path. - Assert.False(writer.TryQueueResult("run", "t2", new string('x', 30_001))); - - await writer.FlushAsync(); - - Assert.Contains("{\"ok\":true}", await store.GetResultsAsync("run")); - } - - /// With result-batching off, TryQueueResult refuses so the caller writes results directly. - [Fact] - public void TryQueueResult_IsRefused_WhenResultBatchingIsOff() - { - var settings = new CraftSettings(); - settings.Orchestrator.BatchResultWrites = false; - var backing = new ControllableStore(); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); - using var writer = new OrchestratorStatusWriter(store, NullLogger.Instance, settings); - - Assert.False(writer.TryQueueResult("run", "t1", "{\"ok\":true}")); - } - - // ── LIVENESS ──────────────────────────────────────────────────────────────────────────────────── - - /// - /// THE regression guard. A stalled storage write must not hold the caller forever — that wait is what - /// consumed all 8 slots while every worker sat idle. - /// - [Fact] - public async Task MarkRunning_TimesOut_WhenStorageStalls_RatherThanHangingForever() - { - var (writer, backing) = NewWriter(barrierTimeoutSec: 2, flushTimeoutSec: 1); - using var _ = writer; - - backing.BatchGate.Reset(); // storage stalls from here on - await writer.QueueRunWarmup(); // ensure the drain loop is running - - var sw = System.Diagnostics.Stopwatch.StartNew(); - await Assert.ThrowsAsync(() => writer.MarkRunningAsync("run", Task1())); - sw.Stop(); - - backing.BatchGate.Set(); - Assert.True(sw.Elapsed < TimeSpan.FromSeconds(30), - $"MarkRunningAsync took {sw.Elapsed.TotalSeconds:F1}s — the barrier is not bounded"); - } - - /// A stalled flush must not stop LATER work once storage recovers — the loop has to survive. - [Fact] - public async Task DrainLoop_KeepsWorking_AfterAFlushTimesOut() - { - var (writer, backing) = NewWriter(barrierTimeoutSec: 2, flushTimeoutSec: 1); - using var _ = writer; - - backing.BatchGate.Reset(); - await Assert.ThrowsAsync(() => writer.MarkRunningAsync("run", Task1("stalled"))); - - backing.BatchGate.Set(); // storage recovers - - // A brand-new marker must now succeed, proving the loop is still alive. - await writer.MarkRunningAsync("run", Task1("after-recovery")); - } - - /// FlushAsync is the other barrier consumer — no run could finalize while it hung. - [Fact] - public async Task FlushAsync_ReturnsWithinTheBound_WhenStorageStalls() - { - var (writer, backing) = NewWriter(barrierTimeoutSec: 2, flushTimeoutSec: 1); - using var _ = writer; - - backing.BatchGate.Reset(); - writer.QueueTask("run", Task1()); - - var sw = System.Diagnostics.Stopwatch.StartNew(); - await writer.FlushAsync(); // must not throw and must not hang - sw.Stop(); - - backing.BatchGate.Set(); - Assert.True(sw.Elapsed < TimeSpan.FromSeconds(30), - $"FlushAsync took {sw.Elapsed.TotalSeconds:F1}s — it is not bounded"); - } - - // ── DURABILITY ────────────────────────────────────────────────────────────────────────────────── - - /// - /// The write that could not be persisted must be retried, not dropped. Dropping it — the previous - /// behaviour on ANY exception — silently lost terminal task states, leaving finished tasks looking - /// Pending forever and re-run by the next recovery pass. - /// - [Fact] - public async Task WritesThatFail_AreRetried_NotLost() - { - var (writer, backing) = NewWriter(barrierTimeoutSec: 2, flushTimeoutSec: 1); - using var _ = writer; - - backing.FailBatches = true; - writer.QueueTask("run", new OrchestratorTaskItem { Id = "t1", Status = "Completed" }); - - await Task.Delay(400); // several failing flushes - Assert.Empty(backing.Rows("OrchestratorTasks")); // nothing persisted yet - - backing.FailBatches = false; // storage recovers - - var deadline = Environment.TickCount64 + 5000; - while (Environment.TickCount64 < deadline && backing.Rows("OrchestratorTasks").Count == 0) - await Task.Delay(20); - - var rows = backing.Rows("OrchestratorTasks"); - Assert.Single(rows); - Assert.Equal("Completed", rows[0].GetString("Status")); // the state survived the outage - } - - /// A newer state for the same task must not be clobbered by a retry of an older snapshot. - [Fact] - public async Task Retry_DoesNotOverwrite_NewerStateForTheSameTask() - { - var (writer, backing) = NewWriter(barrierTimeoutSec: 2, flushTimeoutSec: 1); - using var _ = writer; - - backing.FailBatches = true; - writer.QueueTask("run", new OrchestratorTaskItem { Id = "t1", Status = "Running" }); - await Task.Delay(200); - - writer.QueueTask("run", new OrchestratorTaskItem { Id = "t1", Status = "Completed" }); - backing.FailBatches = false; - - var deadline = Environment.TickCount64 + 5000; - while (Environment.TickCount64 < deadline && backing.Rows("OrchestratorTasks").Count == 0) - await Task.Delay(20); - - var rows = backing.Rows("OrchestratorTasks"); - Assert.Single(rows); - Assert.Equal("Completed", rows[0].GetString("Status")); - } - - /// Shutdown must still push everything pending to storage. - [Fact] - public async Task Dispose_DrainsPendingWrites() - { - var (writer, backing) = NewWriter(barrierTimeoutSec: 2, flushTimeoutSec: 5); - - writer.QueueTask("run", new OrchestratorTaskItem { Id = "t1", Status = "Completed" }); - writer.Dispose(); // final drain runs here - await Task.Delay(50); - - Assert.Single(backing.Rows("OrchestratorTasks")); - } - - // ── THROUGHPUT SHAPE ──────────────────────────────────────────────────────────────────────────── - - /// - /// Writes group by run because a batch shares a partition key. CIPP's workload is ~600 runs of ONE - /// task each, so sequential groups meant hundreds of round-trips per flush with the whole process - /// waiting. They must overlap. - /// - [Fact] - public async Task ManySingleTaskRuns_AreWrittenConcurrently_NotOneAtATime() - { - var settings = new CraftSettings(); - var backing = new ControllableStore(); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); - - // The shape that broke production: one run per tenant, one task each. - var writes = Enumerable.Range(0, 200) - .Select(i => new TaskStatusWrite($"AuditLogIngestV2-tenant{i:D3}.dk", "t1", "Completed", "{}", 0, null, null, null)) - .ToList(); - - // Hold every batch inside the store, so overlap is FORCED rather than raced for. - // - // Two earlier versions of this assertion measured the machine instead of the code. It first - // asserted the whole thing finished inside two seconds — that failed on a box running a 124-task - // orchestration and passed on the same commit once idle. Replacing it with "peak concurrency - // reaches the cap of 8" was no better: under a fully saturated CPU the observed peak was 3, - // because starvation delays the continuations ENTERING the counted region, so it narrows the - // window rather than widening it. - // - // With the gate closed every batch blocks after being counted, so the first 8 occupy the - // semaphore and stay there no matter how slow the scheduler is. The only thing left to wait for - // is progress, and the bound below is generous enough that a slow machine takes longer rather - // than failing. - backing.BatchGate.Reset(); - var pending = store.WriteTaskStatusBatchAsync(writes, maxConcurrency: 8); - - for (var i = 0; i < 400 && Volatile.Read(ref backing.MaxConcurrentBatches) < 8; i++) - await Task.Delay(25); - - // Exactly 8: fewer means the per-run writes are not overlapping as designed, more means the - // requested cap is not being honoured. - Assert.Equal(8, Volatile.Read(ref backing.MaxConcurrentBatches)); - - backing.BatchGate.Set(); - var failed = await pending; - - Assert.Empty(failed); - Assert.Equal(200, backing.BatchCalls); - } - - /// One run's failure must not discard the other 199. - [Fact] - public async Task OneFailingRun_DoesNotDiscardTheRest() - { - var settings = new CraftSettings(); - var backing = new FlakyStore(failFor: "AuditLogIngestV2-tenant005.dk"); - var store = new OrchestratorTableStore(NullLogger.Instance, settings, backing); - - var writes = Enumerable.Range(0, 20) - .Select(i => new TaskStatusWrite($"AuditLogIngestV2-tenant{i:D3}.dk", "t1", "Completed", "{}", 0, null, null, null)) - .ToList(); - - var failed = await store.WriteTaskStatusBatchAsync(writes, maxConcurrency: 4); - - Assert.Equal(["AuditLogIngestV2-tenant005.dk"], failed); - Assert.Equal(19, backing.Written); - } - - private sealed class FlakyStore : ICraftTableStore - { - - // Claims are not exercised by this fake. Fail loudly rather than pretend the guard held — - // a silent 'true' here would look exactly like a successful claim. - public Task TryReplaceBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) => throw new NotSupportedException(); - private readonly string _failFor; - public int Written; - public FlakyStore(string failFor) => _failFor = failFor; - - public Task PingAsync(CancellationToken ct = default) => Task.CompletedTask; - public Task EnsureTableAsync(string table, CancellationToken ct = default) => Task.CompletedTask; - public Task UpsertAsync(string table, StoreRow row, CancellationToken ct = default) => Task.CompletedTask; - - public Task UpsertBatchAsync(string table, string partitionKey, IReadOnlyList rows, - CancellationToken ct = default) - { - if (partitionKey == _failFor) throw new InvalidOperationException("partition unavailable"); - Interlocked.Increment(ref Written); - return Task.CompletedTask; - } - - public Task GetAsync(string t, string p, string r, CancellationToken ct = default) => Task.FromResult(null); - public async IAsyncEnumerable QueryPartitionAsync(string t, string p, - [EnumeratorCancellation] CancellationToken ct = default) - { - await Task.CompletedTask; - yield break; - } - - public async IAsyncEnumerable QueryTableAsync(string t, - [EnumeratorCancellation] CancellationToken ct = default) - { - await Task.CompletedTask; - yield break; - } - - public Task DeleteAsync(string t, string p, string r, CancellationToken ct = default) => Task.CompletedTask; - public Task DeletePartitionAsync(string t, string p, CancellationToken ct = default) => Task.CompletedTask; - } - - // ── RUN-ROW WRITE COST ────────────────────────────────────────────────────────────────────────── - - /// - /// Run rows all share the constant "Run" partition key, so a flush carrying N of them can persist - /// them in ceil(N/100) transactions. Writing them one at a time instead costs N round-trips inside a - /// flush that is bounded by StatusFlushTimeoutSeconds — the cost that pushed real flushes past 30s, - /// then past the 90s barrier, deferring every waiting task until it was abandoned as Pending. - /// - /// Guards the write SHAPE, not wall-clock: a timing assertion here would be flaky on a loaded agent. - /// - [Fact] - public async Task RunRows_ArePersistedInBatches_NotOnePerRoundTrip() - { - var (writer, backing) = NewWriter(barrierTimeoutSec: 30, flushTimeoutSec: 20); - using var _ = writer; - - const int runCount = 150; - for (var i = 0; i < runCount; i++) - writer.QueueRun(new OrchestratorRun { Name = $"run-{i:D3}", Status = "Running" }); - - await writer.FlushAsync(); - - Assert.Equal(runCount, backing.Rows("OrchestratorRuns").Count); - - // 150 rows in one partition = 2 transactions of 100 + 50. Allow generous headroom for the - // warmup row and flush-cycle boundaries, but nothing close to one call per row. - Assert.True(backing.SingleUpsertCalls <= 10, - $"run rows were written with {backing.SingleUpsertCalls} single upserts for {runCount} runs — " + - "they share one partition key and should be batched"); - } - - /// - /// A batch transaction fails atomically, so one bad row would take out the other 99. The write path - /// must fall back to per-row writes for that chunk rather than reporting all of them unwritten. - /// - [Fact] - public async Task RunRows_FallBackToIndividualWrites_WhenABatchFails() - { - var (writer, backing) = NewWriter(barrierTimeoutSec: 30, flushTimeoutSec: 20); - using var _ = writer; - - backing.FailBatches = true; // every batch transaction rejects - - for (var i = 0; i < 5; i++) - writer.QueueRun(new OrchestratorRun { Name = $"run-{i}", Status = "Running" }); - - await writer.FlushAsync(); - - // Batching failed, so each row must still have reached storage individually. - Assert.Equal(5, backing.Rows("OrchestratorRuns").Count); - } -} - -internal static class StatusWriterTestExtensions -{ - /// Nudge the drain loop so it is definitely running before a test manipulates storage. - public static Task QueueRunWarmup(this OrchestratorStatusWriter writer) - { - writer.QueueRun(new OrchestratorRun { Name = "warmup", Status = "Running" }); - return Task.Delay(60); - } -} diff --git a/tests/Craft.Tests/TableKeyTests.cs b/tests/Craft.Tests/TableKeyTests.cs index c8e469f..552236b 100644 --- a/tests/Craft.Tests/TableKeyTests.cs +++ b/tests/Craft.Tests/TableKeyTests.cs @@ -91,7 +91,7 @@ public void TwoNamesThatFoldOntoTheSameIdStayDistinct() [Fact] public void QueueRowKeyIsLegalForARepoNamedTask() { - // The end of the chain the bug actually travelled: id → BuildRowKey → upsert → 400. + // The end of the chain the bug actually travelled: id → row key → upsert → 400. var id = IdFor(""" { "FunctionName": "ExecScheduledCommand", @@ -99,8 +99,8 @@ public void QueueRowKeyIsLegalForARepoNamedTask() } """); - var rowKey = JobQueueStore.BuildRowKey("UserTaskOrchestrator_No tenant", id); - - Assert.True(TableKeys.IsSafe(rowKey)); + // The id keys the task's result row; the run key is the partition of both tables. + Assert.True(TableKeys.IsSafe(id)); + Assert.True(TableKeys.IsSafe(WorkStore.RunKeyFor("UserTaskOrchestrator_No tenant", DateTime.UnixEpoch))); } } diff --git a/tests/Craft.Tests/WorkStoreTests.cs b/tests/Craft.Tests/WorkStoreTests.cs new file mode 100644 index 0000000..53b0292 --- /dev/null +++ b/tests/Craft.Tests/WorkStoreTests.cs @@ -0,0 +1,400 @@ +using Craft.Configuration; +using Craft.Storage; +using Microsoft.Extensions.Logging.Abstractions; + +namespace Craft.Tests; + +/// +/// The storage model: a task's state is one row in its run's partition, and every transition is one +/// transaction that also moves the run's counts. These pin the transitions and the guarantees that replace +/// the old reconciliation machinery: no task is claimed twice, a lost lease cannot finish someone else's +/// task, the counts reach the barrier exactly once, and a task that keeps dying is failed rather than retried +/// forever. +/// +public class WorkStoreTests +{ + private static readonly TimeSpan Lease = TimeSpan.FromMinutes(30); + + private static (WorkStore Store, MemoryTableStore Mem) New() + { + var mem = new MemoryTableStore(); + return (new WorkStore(NullLogger.Instance, new CraftSettings(), mem), mem); + } + + private static RunHeader Header(string name, int minute = 0, int priority = 4, string? postExec = null) => new() + { + RunKey = WorkStore.RunKeyFor(name, At(minute)), + Name = name, + Priority = priority, + StartedUtc = At(minute), + TaskScriptName = "Invoke-CraftTask", + PostExecFunctionName = postExec, + }; + + private static DateTime At(int minute) => new(2026, 10, 5, 3, minute, 0, DateTimeKind.Utc); + + private static List Tasks(int n) => + Enumerable.Range(0, n).Select(i => new WorkStore.NewTask($"t{i}", new() { ["i"] = i })).ToList(); + + private static async Task CreateAsync(WorkStore s, string name, int tasks, int minute = 0, + int priority = 4, string? postExec = null) => + await s.CreateRunAsync(Header(name, minute, priority, postExec), Tasks(tasks)); + + private static List Done(IEnumerable claims, string owner = "w") => + claims.Select(c => new WorkStore.Finish(c.Seq, "Completed", Owner: owner)).ToList(); + + [Fact] + public async Task ARunsTasksAreClaimedOnce_AndTheLastFinishCompletesTheRun() + { + var (s, _) = New(); + var run = await CreateAsync(s, "R", 3); + + var first = await s.ClaimAsync(run.RunKey, 2, "w", Lease, false); + var second = await s.ClaimAsync(run.RunKey, 2, "w", Lease, false); + Assert.Equal([0, 1], first.Select(c => c.Seq)); + Assert.Equal([2], second.Select(c => c.Seq)); + Assert.Empty(await s.ClaimAsync(run.RunKey, 2, "w", Lease, false)); + + var partial = await s.FinishAsync(run.RunKey, Done(first)); + Assert.False(partial!.Completed); + Assert.Equal(2, partial.Header.Done); + + var last = await s.FinishAsync(run.RunKey, Done(second)); + Assert.True(last!.ReachedBarrier); + Assert.True(last.Completed); + Assert.Equal("Completed", last.Header.Status); + Assert.Empty(await ReadyAsync(s)); + } + + [Fact] + public async Task TheBarrierQueuesTheAggregation_WhichCompletesTheRun() + { + var (s, _) = New(); + var run = await CreateAsync(s, "R", 1, postExec: "StoreThings"); + + var task = await s.ClaimAsync(run.RunKey, 8, "w", Lease, false); + var barrier = await s.FinishAsync(run.RunKey, Done(task)); + Assert.True(barrier!.ReachedBarrier); + Assert.False(barrier.Completed); + Assert.Equal(RunPhase.Aggregate, barrier.Header.Phase); + + var agg = Assert.Single(await s.ClaimAsync(run.RunKey, 8, "w", Lease, false)); + Assert.Equal(WorkStore.AggregateSeq, agg.Seq); + var done = await s.FinishAsync(run.RunKey, Done([agg])); + Assert.True(done!.Completed); + Assert.Equal("Completed", done.Header.PostExecStatus); + Assert.Equal(1, done.Header.Done); + } + + [Fact] + public async Task AFinishFromAWorkerThatLostItsLease_IsIgnored() + { + var (s, _) = New(); + var run = await CreateAsync(s, "R", 1); + var claim = await s.ClaimAsync(run.RunKey, 1, "old", TimeSpan.FromMilliseconds(1), false); + await Task.Delay(20); + Assert.Single(await s.ClaimAsync(run.RunKey, 1, "new", Lease, reclaimExpired: true)); + + var stale = await s.FinishAsync(run.RunKey, Done(claim, "old")); + + Assert.Equal(0, stale!.Applied); + Assert.Equal(0, (await s.GetRunAsync(run.RunKey))!.Done); + } + + [Fact] + public async Task ATaskThatKeepsDying_IsFailedAfterThreeAttempts() + { + var (s, _) = New(); + var run = await CreateAsync(s, "R", 1); + + for (var i = 0; i < 3; i++) + { + Assert.Single(await s.ClaimAsync(run.RunKey, 1, $"w{i}", TimeSpan.FromMilliseconds(1), reclaimExpired: true)); + await Task.Delay(20); + } + Assert.Empty(await s.ClaimAsync(run.RunKey, 1, "w4", Lease, reclaimExpired: true)); + + var header = (await s.GetRunAsync(run.RunKey))!; + Assert.Equal(1, header.Failed); + Assert.Equal("CompletedWithErrors", header.Status); + } + + private static Task CreateModeAsync(WorkStore s, string name, int tasks, bool sequential = false, + bool stopOnFailure = false, string? postExec = null) + { + var h = Header(name); + return s.CreateRunAsync(new RunHeader + { + RunKey = h.RunKey, + Name = h.Name, + StartedUtc = h.StartedUtc, + TaskScriptName = h.TaskScriptName, + Sequential = sequential, + StopOnFailure = stopOnFailure, + PostExecFunctionName = postExec, + }, Tasks(tasks)); + } + + // ── a sequential step: finish it and claim the next in one transaction ── + + [Fact] + public async Task FinishingASequentialStep_ClaimsTheNextOne_WithItsPayload_UnderTheDriverLease() + { + var (s, _) = New(); + var run = await CreateModeAsync(s, "Step", 3, sequential: true); + var first = (await s.ClaimSequentialAsync(run.RunKey, "w", Lease))!; + + var r = await s.FinishStepAsync(run.RunKey, new WorkStore.Finish(first.Seq, "Completed", Owner: "w"), "w", Lease); + + Assert.Equal((1, "t1", 1), (r.Next!.Seq, r.Next.TaskId, r.Next.Attempt)); + Assert.Equal("1", r.Payload!["i"].ToString()); + Assert.Equal("Completed", Assert.Single(await s.GetTasksAsync(run.RunKey, 'D')).Status); + Assert.Equal("w", Assert.Single(await s.GetTasksAsync(run.RunKey, 'R')).Owner); + var header = (await s.GetRunAsync(run.RunKey))!; + Assert.Equal(("w", 1), (header.DriverOwner, header.Done)); + Assert.Equal(header.Done, r.Outcome!.Header.Done); + } + + [Fact] + public async Task FinishingTheLastSequentialStep_ClaimsTheAggregationDirectly() + { + var (s, _) = New(); + var run = await CreateModeAsync(s, "StepAgg", 2, sequential: true, postExec: "Agg"); + var step = (await s.ClaimSequentialAsync(run.RunKey, "w", Lease))!; + step = (await s.FinishStepAsync(run.RunKey, new WorkStore.Finish(step.Seq, "Completed", Owner: "w"), "w", Lease)).Next!; + + var r = await s.FinishStepAsync(run.RunKey, new WorkStore.Finish(step.Seq, "Failed", "x", "w"), "w", Lease); + + Assert.Equal(WorkStore.AggregateSeq, r.Next!.Seq); + Assert.Null(r.Payload); + Assert.True(r.Outcome!.ReachedBarrier); + Assert.Empty(await s.GetTasksAsync(run.RunKey, 'P')); + Assert.Equal(WorkStore.AggregateSeq, Assert.Single(await s.GetTasksAsync(run.RunKey, 'R')).Seq); + var header = (await s.GetRunAsync(run.RunKey))!; + Assert.Equal((RunPhase.Aggregate, "Pending", 1), (header.Phase, header.PostExecStatus, header.Failed)); + } + + [Fact] + public async Task FinishingTheLastSequentialStep_WithoutAggregation_CompletesTheRun_AndClaimsNothing() + { + var (s, _) = New(); + var run = await CreateModeAsync(s, "StepLast", 1, sequential: true); + var step = (await s.ClaimSequentialAsync(run.RunKey, "w", Lease))!; + + var r = await s.FinishStepAsync(run.RunKey, new WorkStore.Finish(step.Seq, "Completed", Owner: "w"), "w", Lease); + + Assert.Null(r.Next); + Assert.True(r.Outcome!.Completed); + Assert.Equal("Completed", (await s.GetRunAsync(run.RunKey))!.Status); + } + + [Fact] + public async Task ACancelRequest_StopsTheNextStepBeingClaimed() + { + var (s, _) = New(); + var run = await CreateModeAsync(s, "StepCancel", 3, sequential: true); + var step = (await s.ClaimSequentialAsync(run.RunKey, "w", Lease))!; + await s.RequestCancelAsync(run.RunKey); + + var r = await s.FinishStepAsync(run.RunKey, new WorkStore.Finish(step.Seq, "Completed", Owner: "w"), "w", Lease); + + Assert.Null(r.Next); + Assert.True(r.Outcome!.Header.CancelRequested); + Assert.Equal(2, (await s.GetTasksAsync(run.RunKey, 'P')).Count); + Assert.Empty(await s.GetTasksAsync(run.RunKey, 'R')); + } + + [Fact] + public async Task AStepTakenOverByAnotherDriver_IsNeitherFinishedNorFollowed_ByItsFormerOwner() + { + var (s, _) = New(); + var run = await CreateModeAsync(s, "StepLost", 3, sequential: true); + Assert.NotNull(await s.ClaimSequentialAsync(run.RunKey, "old", TimeSpan.FromMilliseconds(1))); + await Task.Delay(20); + var taken = (await s.ClaimSequentialAsync(run.RunKey, "new", Lease))!; + + var r = await s.FinishStepAsync(run.RunKey, new WorkStore.Finish(taken.Seq, "Completed", Owner: "old"), "old", Lease); + + Assert.Null(r.Next); + Assert.Equal(0, r.Outcome!.Applied); + Assert.Equal("new", Assert.Single(await s.GetTasksAsync(run.RunKey, 'R')).Owner); + Assert.Equal("new", (await s.GetRunAsync(run.RunKey))!.DriverOwner); + } + + [Fact] + public async Task ASequentialStepThatKeepsDying_StopsAStopOnFailureRun() + { + var (s, _) = New(); + var run = await CreateModeAsync(s, "SeqPoison", 4, sequential: true, stopOnFailure: true); + for (var i = 0; i < 3; i++) + { + Assert.NotNull(await s.ClaimSequentialAsync(run.RunKey, $"w{i}", TimeSpan.FromMilliseconds(1))); + await Task.Delay(20); + } + + Assert.Null(await s.ClaimSequentialAsync(run.RunKey, "w9", Lease)); + var done = (await s.GetTasksAsync(run.RunKey, 'D')).OrderBy(t => t.Seq).ToList(); + Assert.Equal(["Failed", "Cancelled", "Cancelled", "Cancelled"], done.Select(t => t.Status)); + Assert.True((await s.GetRunAsync(run.RunKey))!.IsFinished); + } + + [Fact] + public async Task ASequentialStepThatKeepsDying_IsFailed_AndTheRunCarriesOn_ByDefault() + { + var (s, _) = New(); + var run = await CreateModeAsync(s, "SeqPoisonCarry", 3, sequential: true); + for (var i = 0; i < 3; i++) + { + Assert.NotNull(await s.ClaimSequentialAsync(run.RunKey, $"w{i}", TimeSpan.FromMilliseconds(1))); + await Task.Delay(20); + } + + var next = await s.ClaimSequentialAsync(run.RunKey, "w9", Lease); + Assert.Equal(1, next!.Seq); + Assert.Equal("Failed", Assert.Single(await s.GetTasksAsync(run.RunKey, 'D')).Status); + } + + [Fact] + public async Task TwoClaimersRacingForTheSameRows_NeverBothWin() + { + var (s, mem) = New(); + var run = await CreateAsync(s, "R", 2); + var raced = false; + mem.BeforeSubmit = async () => + { + if (raced) return; + raced = true; + mem.BeforeSubmit = null; + await s.ClaimAsync(run.RunKey, 2, "rival", Lease, false); + }; + + Assert.Empty(await s.ClaimAsync(run.RunKey, 2, "me", Lease, false)); + Assert.All(await s.GetTasksAsync(run.RunKey, 'R'), t => Assert.Equal("rival", t.Owner)); + } + + [Fact] + public async Task CancellingARun_CancelsPendingTasks_AndRunningOnesFinishTheRun() + { + var (s, _) = New(); + var run = await CreateAsync(s, "R", 3); + var running = await s.ClaimAsync(run.RunKey, 1, "w", Lease, false); + + var (cancelled, _) = await s.CancelPendingAsync(run.RunKey); + Assert.Equal(2, cancelled); + Assert.False((await s.GetRunAsync(run.RunKey))!.IsFinished); + + var last = await s.FinishAsync(run.RunKey, Done(running)); + Assert.True(last!.Completed); + Assert.Equal("CompletedWithErrors", last.Header.Status); + Assert.Equal(2, last.Header.Cancelled); + } + + [Fact] + public async Task AParentWaitsForItsChild_ThenCompletes() + { + var (s, _) = New(); + var parent = await CreateAsync(s, "Parent", 1); + var task = await s.ClaimAsync(parent.RunKey, 1, "w", Lease, false); + + Assert.True(await s.AddChildAsync(parent.RunKey, "Child~1")); + var tasksDone = await s.FinishAsync(parent.RunKey, Done(task)); + Assert.False(tasksDone!.Completed); + + var childDone = await s.FinishAsync(parent.RunKey, [new WorkStore.Finish(0, "Completed", ChildKey: "Child~1")]); + Assert.True(childDone!.Completed); + Assert.Equal(2, childDone.Header.Done); + } + + [Fact] + public async Task ARunQueuedFromAnAggregation_IsNotAChild() + { + var (s, _) = New(); + var parent = await CreateAsync(s, "Parent", 1, postExec: "Agg"); + await s.FinishAsync(parent.RunKey, Done(await s.ClaimAsync(parent.RunKey, 1, "w", Lease, false))); + + Assert.False(await s.AddChildAsync(parent.RunKey, "Child~1")); + } + + [Fact] + public async Task ReadyListsTheBestBandFirst_ThenTheOldestRun() + { + var (s, _) = New(); + await CreateAsync(s, "Zeta", 1, minute: 0); + await CreateAsync(s, "Alpha", 1, minute: 5); + await CreateAsync(s, "Urgent", 1, minute: 9, priority: 1); + + Assert.Equal(["Urgent", "Zeta", "Alpha"], (await ReadyAsync(s)).Select(e => e.Name)); + } + + [Fact] + public async Task AReleasedTaskGoesBackToPending_WithItsAttemptRefunded() + { + var (s, _) = New(); + var run = await CreateAsync(s, "R", 1); + var claim = Assert.Single(await s.ClaimAsync(run.RunKey, 1, "w", Lease, false)); + + Assert.True(await s.ReleaseAsync(run.RunKey, claim.Seq, "w", refundAttempt: true)); + + var again = Assert.Single(await s.ClaimAsync(run.RunKey, 1, "w", Lease, false)); + Assert.Equal(1, again.Attempt); + } + + [Fact] + public async Task RetentionDeletesRunsFinishedBeforeTheCutoff() + { + var (s, mem) = New(); + var run = await CreateAsync(s, "R", 1); + await s.FinishAsync(run.RunKey, Done(await s.ClaimAsync(run.RunKey, 1, "w", Lease, false))); + + Assert.Equal(0, await s.SweepFinishedAsync(TimeSpan.FromHours(1))); + Assert.Equal(1, await s.SweepFinishedAsync(TimeSpan.Zero)); + Assert.Null(await s.GetRunAsync(run.RunKey)); + Assert.Null(await s.GetRunByNameAsync("R")); + Assert.DoesNotContain(mem.All($"{new CraftSettings().Orchestrator.TablePrefix}Work"), r => r.PartitionKey == run.RunKey); + } + + /// + /// The header is rewritten in every finish transaction, and a transaction cannot split a row, so a + /// property over Azure's 64 KiB limit there would fail every finish of the run. Checked on the row itself, + /// which catches it without needing a real backend or a run large enough to hit the limit. + /// + [Fact] + public async Task TheHeaderStaysSmall_HoweverLargeThePostExecutionParameters() + { + var (s, mem) = New(); + var big = new string('x', 200_000); + var header = Header("Big", postExec: "Agg"); + await s.CreateRunAsync(new RunHeader + { + RunKey = header.RunKey, + Name = header.Name, + StartedUtc = header.StartedUtc, + TaskScriptName = header.TaskScriptName, + PostExecFunctionName = "Agg", + PostExecParametersJson = big, + }, Tasks(1)); + + var row = mem.All("OrchestratorWork").Single(r => r.RowKey == WorkStore.HeaderKey); + Assert.All(row.Properties.Values, v => Assert.True((v as string)?.Length is null or < 32_000)); + Assert.Equal(big, await s.GetPostExecParametersAsync(header.RunKey)); + } + + [Fact] + public async Task APayloadRoundTripsToTheTaskThatWasClaimed() + { + var (s, _) = New(); + var run = await CreateAsync(s, "R", 2); + var claim = (await s.ClaimAsync(run.RunKey, 2, "w", Lease, false))[1]; + + var payload = await s.GetPayloadAsync(run.RunKey, claim.Seq); + Assert.Equal("t1", claim.TaskId); + Assert.Equal("1", payload!["i"].ToString()); + } + + private static async Task> ReadyAsync(WorkStore s) + { + var list = new List(); + await foreach (var e in s.ReadReadyAsync()) list.Add(e); + return list; + } +}