Merge nucleic/mellow-dewy-falcon-rjhr into main
This commit is contained in:
@@ -0,0 +1,344 @@
|
||||
import Foundation
|
||||
import Testing
|
||||
|
||||
@testable import RunnerCore
|
||||
|
||||
/// Tests for the pure scheduling state machine.
|
||||
///
|
||||
/// Every case drives ``SchedulerCore/plan(state:queuedJobs:labels:maxVMs:now:jobTimeout:bootTimeout:)``
|
||||
/// with an injected `now`, so nothing here touches a clock, the network, or a VM.
|
||||
@Suite("SchedulerCore")
|
||||
struct SchedulerCoreTests {
|
||||
// MARK: - Fixtures
|
||||
|
||||
static let labels = LabelSet(["macos-arm64", "macos"])
|
||||
static let now = Date(timeIntervalSince1970: 1_700_000_000)
|
||||
static let jobTimeout: TimeInterval = 3600
|
||||
static let bootTimeout: TimeInterval = 300
|
||||
|
||||
static func job(_ id: Int64, labels: [String] = ["macos-arm64"]) -> WorkflowJob {
|
||||
WorkflowJob(id: id, runID: id * 10, name: "job-\(id)", status: "queued", labels: labels)
|
||||
}
|
||||
|
||||
/// Plans one tick with the suite's fixed labels and timeouts.
|
||||
static func tick(
|
||||
_ state: SchedulerState,
|
||||
_ queued: [WorkflowJob],
|
||||
maxVMs: Int = 2,
|
||||
now: Date = SchedulerCoreTests.now
|
||||
) -> (SchedulerState, [SchedulerAction]) {
|
||||
SchedulerCore.plan(
|
||||
state: state,
|
||||
queuedJobs: queued,
|
||||
labels: labels,
|
||||
maxVMs: maxVMs,
|
||||
now: now,
|
||||
jobTimeout: jobTimeout,
|
||||
bootTimeout: bootTimeout
|
||||
)
|
||||
}
|
||||
|
||||
// MARK: - Baseline
|
||||
|
||||
@Test("a fresh state has no VMs and no ledger")
|
||||
func freshStateIsAllIdle() {
|
||||
let state = SchedulerState(slotCount: 2)
|
||||
#expect(state.slots.count == 2)
|
||||
#expect(state.idleSlots.count == 2)
|
||||
#expect(state.occupiedSlots.isEmpty)
|
||||
#expect(state.dispatchedJobIDs.isEmpty)
|
||||
}
|
||||
|
||||
@Test("an empty queue is a no-op")
|
||||
func emptyQueueDoesNothing() {
|
||||
let state = SchedulerState(slotCount: 2)
|
||||
let (next, actions) = Self.tick(state, [])
|
||||
#expect(actions.isEmpty)
|
||||
#expect(next == state)
|
||||
}
|
||||
|
||||
// MARK: - Booting
|
||||
|
||||
@Test("one queued job boots one VM")
|
||||
func oneJobBootsOneVM() {
|
||||
let (next, actions) = Self.tick(SchedulerState(slotCount: 2), [Self.job(1)])
|
||||
|
||||
#expect(actions == [.bootVM(slot: 0, jobHint: 1)])
|
||||
#expect(next.slots[0].state == .provisioning(since: Self.now))
|
||||
#expect(next.slots[1].state == .idle)
|
||||
#expect(next.dispatchedJobIDs == [1])
|
||||
}
|
||||
|
||||
@Test("the same job across two ticks boots only one VM")
|
||||
func dedupsAcrossTicks() {
|
||||
let queue = [Self.job(1)]
|
||||
let (afterFirst, firstActions) = Self.tick(SchedulerState(slotCount: 2), queue)
|
||||
// The job is still queued a poll later: the VM has not registered yet.
|
||||
let (afterSecond, secondActions) = Self.tick(
|
||||
afterFirst, queue, now: Self.now.addingTimeInterval(10))
|
||||
|
||||
#expect(firstActions == [.bootVM(slot: 0, jobHint: 1)])
|
||||
#expect(secondActions.isEmpty)
|
||||
#expect(afterSecond.occupiedSlots.count == 1)
|
||||
#expect(afterSecond.dispatchedJobIDs == [1])
|
||||
}
|
||||
|
||||
@Test("a job still queued while its VM is running does not boot a second VM")
|
||||
func dedupSurvivesTheRunningTransition() {
|
||||
let queue = [Self.job(1)]
|
||||
let (booted, _) = Self.tick(SchedulerState(slotCount: 2), queue)
|
||||
let running = SchedulerCore.markRunning(state: booted, slot: 0, jobHint: 1, now: Self.now)
|
||||
|
||||
let (next, actions) = Self.tick(running, queue, now: Self.now.addingTimeInterval(30))
|
||||
#expect(actions.isEmpty)
|
||||
#expect(next.occupiedSlots.count == 1)
|
||||
}
|
||||
|
||||
@Test("two jobs boot two VMs but a third waits for capacity")
|
||||
func capacityIsCapped() {
|
||||
let queue = [Self.job(1), Self.job(2), Self.job(3)]
|
||||
let (next, actions) = Self.tick(SchedulerState(slotCount: 2), queue)
|
||||
|
||||
#expect(actions == [.bootVM(slot: 0, jobHint: 1), .bootVM(slot: 1, jobHint: 2)])
|
||||
#expect(next.occupiedSlots.count == 2)
|
||||
// Job 3 never entered the ledger, so it is eligible the moment a slot frees.
|
||||
#expect(next.dispatchedJobIDs == [1, 2])
|
||||
}
|
||||
|
||||
@Test("maxVMs above two is clamped to the kernel's concurrent-guest limit")
|
||||
func maxVMsIsClampedToTwo() {
|
||||
let queue = [Self.job(1), Self.job(2), Self.job(3), Self.job(4)]
|
||||
let (next, actions) = Self.tick(SchedulerState(slotCount: 4), queue, maxVMs: 5)
|
||||
|
||||
#expect(actions.count == 2)
|
||||
#expect(next.occupiedSlots.count == 2)
|
||||
}
|
||||
|
||||
@Test("maxVMs of zero boots nothing")
|
||||
func zeroCapacityBootsNothing() {
|
||||
let (next, actions) = Self.tick(SchedulerState(slotCount: 2), [Self.job(1)], maxVMs: 0)
|
||||
#expect(actions.isEmpty)
|
||||
#expect(next.occupiedSlots.isEmpty)
|
||||
}
|
||||
|
||||
// MARK: - Label matching
|
||||
|
||||
@Test("jobs whose labels do not match are ignored")
|
||||
func nonMatchingLabelsAreIgnored() {
|
||||
let queue = [
|
||||
Self.job(1, labels: ["ubuntu-latest"]),
|
||||
Self.job(2, labels: ["windows-2022", "self-hosted"]),
|
||||
]
|
||||
let (next, actions) = Self.tick(SchedulerState(slotCount: 2), queue)
|
||||
|
||||
#expect(actions.isEmpty)
|
||||
#expect(next.dispatchedJobIDs.isEmpty)
|
||||
#expect(next.occupiedSlots.isEmpty)
|
||||
}
|
||||
|
||||
@Test("a matching job among non-matching ones still boots")
|
||||
func matchingJobIsPickedOutOfAMixedQueue() {
|
||||
let queue = [
|
||||
Self.job(1, labels: ["ubuntu-latest"]),
|
||||
Self.job(2, labels: ["macos-arm64"]),
|
||||
Self.job(3, labels: ["ubuntu-latest"]),
|
||||
]
|
||||
let (_, actions) = Self.tick(SchedulerState(slotCount: 2), queue)
|
||||
#expect(actions == [.bootVM(slot: 0, jobHint: 2)])
|
||||
}
|
||||
|
||||
// MARK: - Ledger expiry
|
||||
|
||||
@Test("a job that leaves the queue drops out of the dedup ledger")
|
||||
func dequeuedJobClearsItsLedgerEntry() {
|
||||
let (booted, _) = Self.tick(SchedulerState(slotCount: 2), [Self.job(1)])
|
||||
let running = SchedulerCore.markRunning(state: booted, slot: 0, jobHint: 1, now: Self.now)
|
||||
#expect(running.dispatchedJobIDs == [1])
|
||||
|
||||
// The VM registered and claimed job 1, so Gitea no longer reports it queued.
|
||||
let (next, actions) = Self.tick(running, [], now: Self.now.addingTimeInterval(60))
|
||||
#expect(actions.isEmpty)
|
||||
#expect(next.dispatchedJobIDs.isEmpty)
|
||||
#expect(next.occupiedSlots.count == 1)
|
||||
}
|
||||
|
||||
@Test("releasing a job lets a still-queued job boot again after a failure")
|
||||
func releasedJobIsRedispatched() {
|
||||
// A boot that failed (no disk, clone error, dead SSH) leaves its job
|
||||
// queued, so the ledger's "no longer queued" expiry never fires for it.
|
||||
// Without the explicit release the job is stranded for good.
|
||||
let (booted, _) = Self.tick(SchedulerState(slotCount: 2), [Self.job(1)])
|
||||
#expect(booted.dispatchedJobIDs == [1])
|
||||
|
||||
let failed = SchedulerCore.releaseJob(
|
||||
state: SchedulerCore.markIdle(state: booted, slot: 0),
|
||||
jobID: 1
|
||||
)
|
||||
#expect(failed.dispatchedJobIDs.isEmpty)
|
||||
|
||||
let (next, actions) = Self.tick(failed, [Self.job(1)], now: Self.now.addingTimeInterval(30))
|
||||
#expect(actions == [.bootVM(slot: 0, jobHint: 1)])
|
||||
#expect(next.dispatchedJobIDs == [1])
|
||||
}
|
||||
|
||||
@Test("releasing an id that was never dispatched is a no-op")
|
||||
func releasingAnUnknownJobIsHarmless() {
|
||||
let (booted, _) = Self.tick(SchedulerState(slotCount: 2), [Self.job(1)])
|
||||
let after = SchedulerCore.releaseJob(state: booted, jobID: 99)
|
||||
#expect(after.dispatchedJobIDs == [1])
|
||||
#expect(SchedulerCore.releaseJob(state: after, jobID: 1).dispatchedJobIDs.isEmpty)
|
||||
}
|
||||
|
||||
@Test("a freed slot is re-earned by a genuinely new job")
|
||||
func freedCapacityServesTheNextJob() {
|
||||
let (booted, _) = Self.tick(SchedulerState(slotCount: 2), [Self.job(1)])
|
||||
let done = SchedulerCore.markIdle(state: booted, slot: 0)
|
||||
|
||||
let (next, actions) = Self.tick(done, [Self.job(2)], now: Self.now.addingTimeInterval(120))
|
||||
#expect(actions == [.bootVM(slot: 0, jobHint: 2)])
|
||||
#expect(next.dispatchedJobIDs == [2])
|
||||
}
|
||||
|
||||
// MARK: - Timeouts
|
||||
|
||||
@Test("a stuck boot is torn down and replaced in the same pass")
|
||||
func bootTimeoutTearsDownAndAllowsAReplacement() {
|
||||
let stuckSince = Self.now.addingTimeInterval(-(Self.bootTimeout + 60))
|
||||
let state = SchedulerState(
|
||||
slots: [
|
||||
VMSlot(id: 0, state: .provisioning(since: stuckSince)),
|
||||
VMSlot(id: 1, state: .idle),
|
||||
],
|
||||
dispatchedJobIDs: [1]
|
||||
)
|
||||
|
||||
let (next, actions) = Self.tick(state, [Self.job(1)])
|
||||
|
||||
// Teardown first so the orchestrator frees the slot before reusing it.
|
||||
#expect(actions.count == 2)
|
||||
if case .teardownVM(let slot, let reason) = actions[0] {
|
||||
#expect(slot == 0)
|
||||
#expect(reason.contains("boot timeout"))
|
||||
} else {
|
||||
Issue.record("expected a teardown first, got \(actions[0])")
|
||||
}
|
||||
#expect(actions[1] == .bootVM(slot: 0, jobHint: 1))
|
||||
#expect(next.slots[0].state == .provisioning(since: Self.now))
|
||||
#expect(next.dispatchedJobIDs == [1])
|
||||
}
|
||||
|
||||
@Test("a boot inside its timeout is left alone")
|
||||
func youngBootIsNotTornDown() {
|
||||
let state = SchedulerState(
|
||||
slots: [VMSlot(id: 0, state: .provisioning(since: Self.now.addingTimeInterval(-10)))],
|
||||
dispatchedJobIDs: [1]
|
||||
)
|
||||
let (next, actions) = Self.tick(state, [Self.job(1)])
|
||||
#expect(actions.isEmpty)
|
||||
#expect(next == state)
|
||||
}
|
||||
|
||||
@Test("a run that overshoots the job timeout is torn down")
|
||||
func jobTimeoutTearsDownARunningSlot() {
|
||||
let startedAt = Self.now.addingTimeInterval(-(Self.jobTimeout + 300))
|
||||
let state = SchedulerState(
|
||||
slots: [
|
||||
VMSlot(id: 0, state: .running(jobHint: 7, since: startedAt)),
|
||||
VMSlot(id: 1, state: .idle),
|
||||
],
|
||||
dispatchedJobIDs: [7]
|
||||
)
|
||||
|
||||
let (next, actions) = Self.tick(state, [])
|
||||
|
||||
#expect(actions.count == 1)
|
||||
if case .teardownVM(let slot, let reason) = actions[0] {
|
||||
#expect(slot == 0)
|
||||
#expect(reason.contains("job timeout"))
|
||||
} else {
|
||||
Issue.record("expected a teardown, got \(actions[0])")
|
||||
}
|
||||
#expect(next.slots[0].state == .idle)
|
||||
#expect(next.dispatchedJobIDs.isEmpty)
|
||||
}
|
||||
|
||||
@Test("a run inside its timeout is left alone")
|
||||
func youngRunIsNotTornDown() {
|
||||
let state = SchedulerState(
|
||||
slots: [VMSlot(id: 0, state: .running(jobHint: 7, since: Self.now.addingTimeInterval(-60)))]
|
||||
)
|
||||
let (next, actions) = Self.tick(state, [])
|
||||
#expect(actions.isEmpty)
|
||||
#expect(next == state)
|
||||
}
|
||||
|
||||
@Test("both slots can time out on the same tick")
|
||||
func bothSlotsCanTimeOutTogether() {
|
||||
let state = SchedulerState(
|
||||
slots: [
|
||||
VMSlot(id: 0, state: .provisioning(since: Self.now.addingTimeInterval(-1000))),
|
||||
VMSlot(id: 1, state: .running(jobHint: 9, since: Self.now.addingTimeInterval(-100_000))),
|
||||
],
|
||||
dispatchedJobIDs: [9]
|
||||
)
|
||||
let (next, actions) = Self.tick(state, [])
|
||||
#expect(actions.count == 2)
|
||||
#expect(next.occupiedSlots.isEmpty)
|
||||
}
|
||||
|
||||
// MARK: - Transitions
|
||||
|
||||
@Test("the mark* helpers move a slot through its lifecycle")
|
||||
func slotTransitions() {
|
||||
var state = SchedulerState(slotCount: 2)
|
||||
|
||||
state = SchedulerCore.markProvisioning(state: state, slot: 1, jobHint: 42, now: Self.now)
|
||||
#expect(state.slots[1].state == .provisioning(since: Self.now))
|
||||
#expect(state.dispatchedJobIDs == [42])
|
||||
|
||||
let live = Self.now.addingTimeInterval(90)
|
||||
state = SchedulerCore.markRunning(state: state, slot: 1, jobHint: 42, now: live)
|
||||
#expect(state.slots[1].state == .running(jobHint: 42, since: live))
|
||||
|
||||
state = SchedulerCore.markIdle(state: state, slot: 1)
|
||||
#expect(state.slots[1].state == .idle)
|
||||
#expect(state.occupiedSlots.isEmpty)
|
||||
}
|
||||
|
||||
@Test("markRunning keeps an existing hint and tolerates an unknown slot")
|
||||
func markRunningIsForgiving() {
|
||||
var state = SchedulerState(slotCount: 1)
|
||||
state = SchedulerCore.markRunning(state: state, slot: 0, jobHint: 5, now: Self.now)
|
||||
|
||||
let later = Self.now.addingTimeInterval(30)
|
||||
let carried = SchedulerCore.markRunning(state: state, slot: 0, now: later)
|
||||
#expect(carried.slots[0].state == .running(jobHint: 5, since: later))
|
||||
|
||||
// A slot id we do not own is ignored rather than trapping.
|
||||
#expect(SchedulerCore.markRunning(state: state, slot: 99, now: later) == state)
|
||||
#expect(SchedulerCore.markIdle(state: state, slot: 99) == state)
|
||||
}
|
||||
|
||||
// MARK: - Purity
|
||||
|
||||
@Test("planning is deterministic and leaves its input untouched")
|
||||
func planningIsPure() {
|
||||
let state = SchedulerState(slotCount: 2)
|
||||
let queue = [Self.job(1), Self.job(2)]
|
||||
|
||||
let (firstState, firstActions) = Self.tick(state, queue)
|
||||
let (secondState, secondActions) = Self.tick(state, queue)
|
||||
|
||||
#expect(firstState == secondState)
|
||||
#expect(firstActions == secondActions)
|
||||
// The value passed in is unchanged — `plan` returns a new state.
|
||||
#expect(state.occupiedSlots.isEmpty)
|
||||
#expect(state.dispatchedJobIDs.isEmpty)
|
||||
}
|
||||
|
||||
@Test("the plan never contains an explicit no-op action")
|
||||
func noOpIsAnEmptyPlan() {
|
||||
let (_, actions) = Self.tick(SchedulerState(slotCount: 2), [Self.job(1)])
|
||||
#expect(!actions.contains(.none))
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user