import Foundation import Testing @testable import RunnerCore /// Tests for the pure scheduling state machine. /// /// Every case drives ``SchedulerCore/plan(state:queuedJobs:labels:maxVMs:now:jobTimeout:bootTimeout:)`` /// with an injected `now`, so nothing here touches a clock, the network, or a VM. @Suite("SchedulerCore") struct SchedulerCoreTests { // MARK: - Fixtures static let labels = LabelSet(["macos-arm64", "macos"]) static let now = Date(timeIntervalSince1970: 1_700_000_000) static let jobTimeout: TimeInterval = 3600 static let bootTimeout: TimeInterval = 300 static func job(_ id: Int64, labels: [String] = ["macos-arm64"]) -> WorkflowJob { WorkflowJob(id: id, runID: id * 10, name: "job-\(id)", status: "queued", labels: labels) } /// Plans one tick with the suite's fixed labels and timeouts. static func tick( _ state: SchedulerState, _ queued: [WorkflowJob], maxVMs: Int = 2, now: Date = SchedulerCoreTests.now ) -> (SchedulerState, [SchedulerAction]) { SchedulerCore.plan( state: state, queuedJobs: queued, labels: labels, maxVMs: maxVMs, now: now, jobTimeout: jobTimeout, bootTimeout: bootTimeout ) } // MARK: - Baseline @Test("a fresh state has no VMs and no ledger") func freshStateIsAllIdle() { let state = SchedulerState(slotCount: 2) #expect(state.slots.count == 2) #expect(state.idleSlots.count == 2) #expect(state.occupiedSlots.isEmpty) #expect(state.dispatchedJobIDs.isEmpty) } @Test("an empty queue is a no-op") func emptyQueueDoesNothing() { let state = SchedulerState(slotCount: 2) let (next, actions) = Self.tick(state, []) #expect(actions.isEmpty) #expect(next == state) } // MARK: - Booting @Test("one queued job boots one VM") func oneJobBootsOneVM() { let (next, actions) = Self.tick(SchedulerState(slotCount: 2), [Self.job(1)]) #expect(actions == [.bootVM(slot: 0, jobHint: 1)]) #expect(next.slots[0].state == .provisioning(since: Self.now)) #expect(next.slots[1].state == .idle) #expect(next.dispatchedJobIDs == [1]) } @Test("the same job across two ticks boots only one VM") func dedupsAcrossTicks() { let queue = [Self.job(1)] let (afterFirst, firstActions) = Self.tick(SchedulerState(slotCount: 2), queue) // The job is still queued a poll later: the VM has not registered yet. let (afterSecond, secondActions) = Self.tick( afterFirst, queue, now: Self.now.addingTimeInterval(10)) #expect(firstActions == [.bootVM(slot: 0, jobHint: 1)]) #expect(secondActions.isEmpty) #expect(afterSecond.occupiedSlots.count == 1) #expect(afterSecond.dispatchedJobIDs == [1]) } @Test("a job still queued while its VM is running does not boot a second VM") func dedupSurvivesTheRunningTransition() { let queue = [Self.job(1)] let (booted, _) = Self.tick(SchedulerState(slotCount: 2), queue) let running = SchedulerCore.markRunning(state: booted, slot: 0, jobHint: 1, now: Self.now) let (next, actions) = Self.tick(running, queue, now: Self.now.addingTimeInterval(30)) #expect(actions.isEmpty) #expect(next.occupiedSlots.count == 1) } @Test("two jobs boot two VMs but a third waits for capacity") func capacityIsCapped() { let queue = [Self.job(1), Self.job(2), Self.job(3)] let (next, actions) = Self.tick(SchedulerState(slotCount: 2), queue) #expect(actions == [.bootVM(slot: 0, jobHint: 1), .bootVM(slot: 1, jobHint: 2)]) #expect(next.occupiedSlots.count == 2) // Job 3 never entered the ledger, so it is eligible the moment a slot frees. #expect(next.dispatchedJobIDs == [1, 2]) } @Test("maxVMs above two is clamped to the kernel's concurrent-guest limit") func maxVMsIsClampedToTwo() { let queue = [Self.job(1), Self.job(2), Self.job(3), Self.job(4)] let (next, actions) = Self.tick(SchedulerState(slotCount: 4), queue, maxVMs: 5) #expect(actions.count == 2) #expect(next.occupiedSlots.count == 2) } @Test("maxVMs of zero boots nothing") func zeroCapacityBootsNothing() { let (next, actions) = Self.tick(SchedulerState(slotCount: 2), [Self.job(1)], maxVMs: 0) #expect(actions.isEmpty) #expect(next.occupiedSlots.isEmpty) } // MARK: - Label matching @Test("jobs whose labels do not match are ignored") func nonMatchingLabelsAreIgnored() { let queue = [ Self.job(1, labels: ["ubuntu-latest"]), Self.job(2, labels: ["windows-2022", "self-hosted"]), ] let (next, actions) = Self.tick(SchedulerState(slotCount: 2), queue) #expect(actions.isEmpty) #expect(next.dispatchedJobIDs.isEmpty) #expect(next.occupiedSlots.isEmpty) } @Test("a matching job among non-matching ones still boots") func matchingJobIsPickedOutOfAMixedQueue() { let queue = [ Self.job(1, labels: ["ubuntu-latest"]), Self.job(2, labels: ["macos-arm64"]), Self.job(3, labels: ["ubuntu-latest"]), ] let (_, actions) = Self.tick(SchedulerState(slotCount: 2), queue) #expect(actions == [.bootVM(slot: 0, jobHint: 2)]) } // MARK: - Ledger expiry @Test("a job that leaves the queue drops out of the dedup ledger") func dequeuedJobClearsItsLedgerEntry() { let (booted, _) = Self.tick(SchedulerState(slotCount: 2), [Self.job(1)]) let running = SchedulerCore.markRunning(state: booted, slot: 0, jobHint: 1, now: Self.now) #expect(running.dispatchedJobIDs == [1]) // The VM registered and claimed job 1, so Gitea no longer reports it queued. let (next, actions) = Self.tick(running, [], now: Self.now.addingTimeInterval(60)) #expect(actions.isEmpty) #expect(next.dispatchedJobIDs.isEmpty) #expect(next.occupiedSlots.count == 1) } @Test("releasing a job lets a still-queued job boot again after a failure") func releasedJobIsRedispatched() { // A boot that failed (no disk, clone error, dead SSH) leaves its job // queued, so the ledger's "no longer queued" expiry never fires for it. // Without the explicit release the job is stranded for good. let (booted, _) = Self.tick(SchedulerState(slotCount: 2), [Self.job(1)]) #expect(booted.dispatchedJobIDs == [1]) let failed = SchedulerCore.releaseJob( state: SchedulerCore.markIdle(state: booted, slot: 0), jobID: 1 ) #expect(failed.dispatchedJobIDs.isEmpty) let (next, actions) = Self.tick(failed, [Self.job(1)], now: Self.now.addingTimeInterval(30)) #expect(actions == [.bootVM(slot: 0, jobHint: 1)]) #expect(next.dispatchedJobIDs == [1]) } @Test("releasing an id that was never dispatched is a no-op") func releasingAnUnknownJobIsHarmless() { let (booted, _) = Self.tick(SchedulerState(slotCount: 2), [Self.job(1)]) let after = SchedulerCore.releaseJob(state: booted, jobID: 99) #expect(after.dispatchedJobIDs == [1]) #expect(SchedulerCore.releaseJob(state: after, jobID: 1).dispatchedJobIDs.isEmpty) } @Test("a freed slot is re-earned by a genuinely new job") func freedCapacityServesTheNextJob() { let (booted, _) = Self.tick(SchedulerState(slotCount: 2), [Self.job(1)]) let done = SchedulerCore.markIdle(state: booted, slot: 0) let (next, actions) = Self.tick(done, [Self.job(2)], now: Self.now.addingTimeInterval(120)) #expect(actions == [.bootVM(slot: 0, jobHint: 2)]) #expect(next.dispatchedJobIDs == [2]) } // MARK: - Timeouts @Test("a stuck boot is torn down and replaced in the same pass") func bootTimeoutTearsDownAndAllowsAReplacement() { let stuckSince = Self.now.addingTimeInterval(-(Self.bootTimeout + 60)) let state = SchedulerState( slots: [ VMSlot(id: 0, state: .provisioning(since: stuckSince)), VMSlot(id: 1, state: .idle), ], dispatchedJobIDs: [1] ) let (next, actions) = Self.tick(state, [Self.job(1)]) // Teardown first so the orchestrator frees the slot before reusing it. #expect(actions.count == 2) if case .teardownVM(let slot, let reason) = actions[0] { #expect(slot == 0) #expect(reason.contains("boot timeout")) } else { Issue.record("expected a teardown first, got \(actions[0])") } #expect(actions[1] == .bootVM(slot: 0, jobHint: 1)) #expect(next.slots[0].state == .provisioning(since: Self.now)) #expect(next.dispatchedJobIDs == [1]) } @Test("a boot inside its timeout is left alone") func youngBootIsNotTornDown() { let state = SchedulerState( slots: [VMSlot(id: 0, state: .provisioning(since: Self.now.addingTimeInterval(-10)))], dispatchedJobIDs: [1] ) let (next, actions) = Self.tick(state, [Self.job(1)]) #expect(actions.isEmpty) #expect(next == state) } @Test("a run that overshoots the job timeout is torn down") func jobTimeoutTearsDownARunningSlot() { let startedAt = Self.now.addingTimeInterval(-(Self.jobTimeout + 300)) let state = SchedulerState( slots: [ VMSlot(id: 0, state: .running(jobHint: 7, since: startedAt)), VMSlot(id: 1, state: .idle), ], dispatchedJobIDs: [7] ) let (next, actions) = Self.tick(state, []) #expect(actions.count == 1) if case .teardownVM(let slot, let reason) = actions[0] { #expect(slot == 0) #expect(reason.contains("job timeout")) } else { Issue.record("expected a teardown, got \(actions[0])") } #expect(next.slots[0].state == .idle) #expect(next.dispatchedJobIDs.isEmpty) } @Test("a run inside its timeout is left alone") func youngRunIsNotTornDown() { let state = SchedulerState( slots: [VMSlot(id: 0, state: .running(jobHint: 7, since: Self.now.addingTimeInterval(-60)))] ) let (next, actions) = Self.tick(state, []) #expect(actions.isEmpty) #expect(next == state) } @Test("both slots can time out on the same tick") func bothSlotsCanTimeOutTogether() { let state = SchedulerState( slots: [ VMSlot(id: 0, state: .provisioning(since: Self.now.addingTimeInterval(-1000))), VMSlot(id: 1, state: .running(jobHint: 9, since: Self.now.addingTimeInterval(-100_000))), ], dispatchedJobIDs: [9] ) let (next, actions) = Self.tick(state, []) #expect(actions.count == 2) #expect(next.occupiedSlots.isEmpty) } // MARK: - Transitions @Test("the mark* helpers move a slot through its lifecycle") func slotTransitions() { var state = SchedulerState(slotCount: 2) state = SchedulerCore.markProvisioning(state: state, slot: 1, jobHint: 42, now: Self.now) #expect(state.slots[1].state == .provisioning(since: Self.now)) #expect(state.dispatchedJobIDs == [42]) let live = Self.now.addingTimeInterval(90) state = SchedulerCore.markRunning(state: state, slot: 1, jobHint: 42, now: live) #expect(state.slots[1].state == .running(jobHint: 42, since: live)) state = SchedulerCore.markIdle(state: state, slot: 1) #expect(state.slots[1].state == .idle) #expect(state.occupiedSlots.isEmpty) } @Test("markRunning keeps an existing hint and tolerates an unknown slot") func markRunningIsForgiving() { var state = SchedulerState(slotCount: 1) state = SchedulerCore.markRunning(state: state, slot: 0, jobHint: 5, now: Self.now) let later = Self.now.addingTimeInterval(30) let carried = SchedulerCore.markRunning(state: state, slot: 0, now: later) #expect(carried.slots[0].state == .running(jobHint: 5, since: later)) // A slot id we do not own is ignored rather than trapping. #expect(SchedulerCore.markRunning(state: state, slot: 99, now: later) == state) #expect(SchedulerCore.markIdle(state: state, slot: 99) == state) } // MARK: - Purity @Test("planning is deterministic and leaves its input untouched") func planningIsPure() { let state = SchedulerState(slotCount: 2) let queue = [Self.job(1), Self.job(2)] let (firstState, firstActions) = Self.tick(state, queue) let (secondState, secondActions) = Self.tick(state, queue) #expect(firstState == secondState) #expect(firstActions == secondActions) // The value passed in is unchanged — `plan` returns a new state. #expect(state.occupiedSlots.isEmpty) #expect(state.dispatchedJobIDs.isEmpty) } @Test("the plan never contains an explicit no-op action") func noOpIsAnEmptyPlan() { let (_, actions) = Self.tick(SchedulerState(slotCount: 2), [Self.job(1)]) #expect(!actions.contains(.none)) } }