Merge nucleic/mellow-dewy-falcon-rjhr into main

This commit is contained in:
2026-08-07 00:44:36 -07:00
parent 749f0be4fb
commit bc2b6cd33b
47 changed files with 13159 additions and 0 deletions
@@ -0,0 +1,344 @@
import Foundation
import Testing
@testable import RunnerCore
/// Tests for the pure scheduling state machine.
///
/// Every case drives ``SchedulerCore/plan(state:queuedJobs:labels:maxVMs:now:jobTimeout:bootTimeout:)``
/// with an injected `now`, so nothing here touches a clock, the network, or a VM.
@Suite("SchedulerCore")
struct SchedulerCoreTests {
// MARK: - Fixtures
static let labels = LabelSet(["macos-arm64", "macos"])
static let now = Date(timeIntervalSince1970: 1_700_000_000)
static let jobTimeout: TimeInterval = 3600
static let bootTimeout: TimeInterval = 300
static func job(_ id: Int64, labels: [String] = ["macos-arm64"]) -> WorkflowJob {
WorkflowJob(id: id, runID: id * 10, name: "job-\(id)", status: "queued", labels: labels)
}
/// Plans one tick with the suite's fixed labels and timeouts.
static func tick(
_ state: SchedulerState,
_ queued: [WorkflowJob],
maxVMs: Int = 2,
now: Date = SchedulerCoreTests.now
) -> (SchedulerState, [SchedulerAction]) {
SchedulerCore.plan(
state: state,
queuedJobs: queued,
labels: labels,
maxVMs: maxVMs,
now: now,
jobTimeout: jobTimeout,
bootTimeout: bootTimeout
)
}
// MARK: - Baseline
@Test("a fresh state has no VMs and no ledger")
func freshStateIsAllIdle() {
let state = SchedulerState(slotCount: 2)
#expect(state.slots.count == 2)
#expect(state.idleSlots.count == 2)
#expect(state.occupiedSlots.isEmpty)
#expect(state.dispatchedJobIDs.isEmpty)
}
@Test("an empty queue is a no-op")
func emptyQueueDoesNothing() {
let state = SchedulerState(slotCount: 2)
let (next, actions) = Self.tick(state, [])
#expect(actions.isEmpty)
#expect(next == state)
}
// MARK: - Booting
@Test("one queued job boots one VM")
func oneJobBootsOneVM() {
let (next, actions) = Self.tick(SchedulerState(slotCount: 2), [Self.job(1)])
#expect(actions == [.bootVM(slot: 0, jobHint: 1)])
#expect(next.slots[0].state == .provisioning(since: Self.now))
#expect(next.slots[1].state == .idle)
#expect(next.dispatchedJobIDs == [1])
}
@Test("the same job across two ticks boots only one VM")
func dedupsAcrossTicks() {
let queue = [Self.job(1)]
let (afterFirst, firstActions) = Self.tick(SchedulerState(slotCount: 2), queue)
// The job is still queued a poll later: the VM has not registered yet.
let (afterSecond, secondActions) = Self.tick(
afterFirst, queue, now: Self.now.addingTimeInterval(10))
#expect(firstActions == [.bootVM(slot: 0, jobHint: 1)])
#expect(secondActions.isEmpty)
#expect(afterSecond.occupiedSlots.count == 1)
#expect(afterSecond.dispatchedJobIDs == [1])
}
@Test("a job still queued while its VM is running does not boot a second VM")
func dedupSurvivesTheRunningTransition() {
let queue = [Self.job(1)]
let (booted, _) = Self.tick(SchedulerState(slotCount: 2), queue)
let running = SchedulerCore.markRunning(state: booted, slot: 0, jobHint: 1, now: Self.now)
let (next, actions) = Self.tick(running, queue, now: Self.now.addingTimeInterval(30))
#expect(actions.isEmpty)
#expect(next.occupiedSlots.count == 1)
}
@Test("two jobs boot two VMs but a third waits for capacity")
func capacityIsCapped() {
let queue = [Self.job(1), Self.job(2), Self.job(3)]
let (next, actions) = Self.tick(SchedulerState(slotCount: 2), queue)
#expect(actions == [.bootVM(slot: 0, jobHint: 1), .bootVM(slot: 1, jobHint: 2)])
#expect(next.occupiedSlots.count == 2)
// Job 3 never entered the ledger, so it is eligible the moment a slot frees.
#expect(next.dispatchedJobIDs == [1, 2])
}
@Test("maxVMs above two is clamped to the kernel's concurrent-guest limit")
func maxVMsIsClampedToTwo() {
let queue = [Self.job(1), Self.job(2), Self.job(3), Self.job(4)]
let (next, actions) = Self.tick(SchedulerState(slotCount: 4), queue, maxVMs: 5)
#expect(actions.count == 2)
#expect(next.occupiedSlots.count == 2)
}
@Test("maxVMs of zero boots nothing")
func zeroCapacityBootsNothing() {
let (next, actions) = Self.tick(SchedulerState(slotCount: 2), [Self.job(1)], maxVMs: 0)
#expect(actions.isEmpty)
#expect(next.occupiedSlots.isEmpty)
}
// MARK: - Label matching
@Test("jobs whose labels do not match are ignored")
func nonMatchingLabelsAreIgnored() {
let queue = [
Self.job(1, labels: ["ubuntu-latest"]),
Self.job(2, labels: ["windows-2022", "self-hosted"]),
]
let (next, actions) = Self.tick(SchedulerState(slotCount: 2), queue)
#expect(actions.isEmpty)
#expect(next.dispatchedJobIDs.isEmpty)
#expect(next.occupiedSlots.isEmpty)
}
@Test("a matching job among non-matching ones still boots")
func matchingJobIsPickedOutOfAMixedQueue() {
let queue = [
Self.job(1, labels: ["ubuntu-latest"]),
Self.job(2, labels: ["macos-arm64"]),
Self.job(3, labels: ["ubuntu-latest"]),
]
let (_, actions) = Self.tick(SchedulerState(slotCount: 2), queue)
#expect(actions == [.bootVM(slot: 0, jobHint: 2)])
}
// MARK: - Ledger expiry
@Test("a job that leaves the queue drops out of the dedup ledger")
func dequeuedJobClearsItsLedgerEntry() {
let (booted, _) = Self.tick(SchedulerState(slotCount: 2), [Self.job(1)])
let running = SchedulerCore.markRunning(state: booted, slot: 0, jobHint: 1, now: Self.now)
#expect(running.dispatchedJobIDs == [1])
// The VM registered and claimed job 1, so Gitea no longer reports it queued.
let (next, actions) = Self.tick(running, [], now: Self.now.addingTimeInterval(60))
#expect(actions.isEmpty)
#expect(next.dispatchedJobIDs.isEmpty)
#expect(next.occupiedSlots.count == 1)
}
@Test("releasing a job lets a still-queued job boot again after a failure")
func releasedJobIsRedispatched() {
// A boot that failed (no disk, clone error, dead SSH) leaves its job
// queued, so the ledger's "no longer queued" expiry never fires for it.
// Without the explicit release the job is stranded for good.
let (booted, _) = Self.tick(SchedulerState(slotCount: 2), [Self.job(1)])
#expect(booted.dispatchedJobIDs == [1])
let failed = SchedulerCore.releaseJob(
state: SchedulerCore.markIdle(state: booted, slot: 0),
jobID: 1
)
#expect(failed.dispatchedJobIDs.isEmpty)
let (next, actions) = Self.tick(failed, [Self.job(1)], now: Self.now.addingTimeInterval(30))
#expect(actions == [.bootVM(slot: 0, jobHint: 1)])
#expect(next.dispatchedJobIDs == [1])
}
@Test("releasing an id that was never dispatched is a no-op")
func releasingAnUnknownJobIsHarmless() {
let (booted, _) = Self.tick(SchedulerState(slotCount: 2), [Self.job(1)])
let after = SchedulerCore.releaseJob(state: booted, jobID: 99)
#expect(after.dispatchedJobIDs == [1])
#expect(SchedulerCore.releaseJob(state: after, jobID: 1).dispatchedJobIDs.isEmpty)
}
@Test("a freed slot is re-earned by a genuinely new job")
func freedCapacityServesTheNextJob() {
let (booted, _) = Self.tick(SchedulerState(slotCount: 2), [Self.job(1)])
let done = SchedulerCore.markIdle(state: booted, slot: 0)
let (next, actions) = Self.tick(done, [Self.job(2)], now: Self.now.addingTimeInterval(120))
#expect(actions == [.bootVM(slot: 0, jobHint: 2)])
#expect(next.dispatchedJobIDs == [2])
}
// MARK: - Timeouts
@Test("a stuck boot is torn down and replaced in the same pass")
func bootTimeoutTearsDownAndAllowsAReplacement() {
let stuckSince = Self.now.addingTimeInterval(-(Self.bootTimeout + 60))
let state = SchedulerState(
slots: [
VMSlot(id: 0, state: .provisioning(since: stuckSince)),
VMSlot(id: 1, state: .idle),
],
dispatchedJobIDs: [1]
)
let (next, actions) = Self.tick(state, [Self.job(1)])
// Teardown first so the orchestrator frees the slot before reusing it.
#expect(actions.count == 2)
if case .teardownVM(let slot, let reason) = actions[0] {
#expect(slot == 0)
#expect(reason.contains("boot timeout"))
} else {
Issue.record("expected a teardown first, got \(actions[0])")
}
#expect(actions[1] == .bootVM(slot: 0, jobHint: 1))
#expect(next.slots[0].state == .provisioning(since: Self.now))
#expect(next.dispatchedJobIDs == [1])
}
@Test("a boot inside its timeout is left alone")
func youngBootIsNotTornDown() {
let state = SchedulerState(
slots: [VMSlot(id: 0, state: .provisioning(since: Self.now.addingTimeInterval(-10)))],
dispatchedJobIDs: [1]
)
let (next, actions) = Self.tick(state, [Self.job(1)])
#expect(actions.isEmpty)
#expect(next == state)
}
@Test("a run that overshoots the job timeout is torn down")
func jobTimeoutTearsDownARunningSlot() {
let startedAt = Self.now.addingTimeInterval(-(Self.jobTimeout + 300))
let state = SchedulerState(
slots: [
VMSlot(id: 0, state: .running(jobHint: 7, since: startedAt)),
VMSlot(id: 1, state: .idle),
],
dispatchedJobIDs: [7]
)
let (next, actions) = Self.tick(state, [])
#expect(actions.count == 1)
if case .teardownVM(let slot, let reason) = actions[0] {
#expect(slot == 0)
#expect(reason.contains("job timeout"))
} else {
Issue.record("expected a teardown, got \(actions[0])")
}
#expect(next.slots[0].state == .idle)
#expect(next.dispatchedJobIDs.isEmpty)
}
@Test("a run inside its timeout is left alone")
func youngRunIsNotTornDown() {
let state = SchedulerState(
slots: [VMSlot(id: 0, state: .running(jobHint: 7, since: Self.now.addingTimeInterval(-60)))]
)
let (next, actions) = Self.tick(state, [])
#expect(actions.isEmpty)
#expect(next == state)
}
@Test("both slots can time out on the same tick")
func bothSlotsCanTimeOutTogether() {
let state = SchedulerState(
slots: [
VMSlot(id: 0, state: .provisioning(since: Self.now.addingTimeInterval(-1000))),
VMSlot(id: 1, state: .running(jobHint: 9, since: Self.now.addingTimeInterval(-100_000))),
],
dispatchedJobIDs: [9]
)
let (next, actions) = Self.tick(state, [])
#expect(actions.count == 2)
#expect(next.occupiedSlots.isEmpty)
}
// MARK: - Transitions
@Test("the mark* helpers move a slot through its lifecycle")
func slotTransitions() {
var state = SchedulerState(slotCount: 2)
state = SchedulerCore.markProvisioning(state: state, slot: 1, jobHint: 42, now: Self.now)
#expect(state.slots[1].state == .provisioning(since: Self.now))
#expect(state.dispatchedJobIDs == [42])
let live = Self.now.addingTimeInterval(90)
state = SchedulerCore.markRunning(state: state, slot: 1, jobHint: 42, now: live)
#expect(state.slots[1].state == .running(jobHint: 42, since: live))
state = SchedulerCore.markIdle(state: state, slot: 1)
#expect(state.slots[1].state == .idle)
#expect(state.occupiedSlots.isEmpty)
}
@Test("markRunning keeps an existing hint and tolerates an unknown slot")
func markRunningIsForgiving() {
var state = SchedulerState(slotCount: 1)
state = SchedulerCore.markRunning(state: state, slot: 0, jobHint: 5, now: Self.now)
let later = Self.now.addingTimeInterval(30)
let carried = SchedulerCore.markRunning(state: state, slot: 0, now: later)
#expect(carried.slots[0].state == .running(jobHint: 5, since: later))
// A slot id we do not own is ignored rather than trapping.
#expect(SchedulerCore.markRunning(state: state, slot: 99, now: later) == state)
#expect(SchedulerCore.markIdle(state: state, slot: 99) == state)
}
// MARK: - Purity
@Test("planning is deterministic and leaves its input untouched")
func planningIsPure() {
let state = SchedulerState(slotCount: 2)
let queue = [Self.job(1), Self.job(2)]
let (firstState, firstActions) = Self.tick(state, queue)
let (secondState, secondActions) = Self.tick(state, queue)
#expect(firstState == secondState)
#expect(firstActions == secondActions)
// The value passed in is unchanged — `plan` returns a new state.
#expect(state.occupiedSlots.isEmpty)
#expect(state.dispatchedJobIDs.isEmpty)
}
@Test("the plan never contains an explicit no-op action")
func noOpIsAnEmptyPlan() {
let (_, actions) = Self.tick(SchedulerState(slotCount: 2), [Self.job(1)])
#expect(!actions.contains(.none))
}
}