374 lines
15 KiB
Swift
374 lines
15 KiB
Swift
import Foundation
|
|
import RunnerCore
|
|
import Virtualization
|
|
|
|
/// Why a VM stopped.
|
|
public enum VMStopReason: Sendable, Equatable {
|
|
/// The guest shut itself down (our normal path: the SSH session runs
|
|
/// `shutdown`, or `gitea-runner daemon` exits and provisioning halts it).
|
|
case guestInitiated
|
|
/// We asked it to stop and it complied.
|
|
case requested
|
|
/// The framework reported an error.
|
|
case failed(String)
|
|
}
|
|
|
|
/// Owns one live `VZVirtualMachine` and exposes it as an `async` API.
|
|
///
|
|
/// ## Threading
|
|
///
|
|
/// `VZVirtualMachine` is not thread-safe and must be used only from the queue it
|
|
/// was created with. This class creates it with
|
|
/// `VZVirtualMachine(configuration:queue:)` on a **private serial queue** and
|
|
/// funnels every call through that queue, bridging the framework's
|
|
/// completion-handler API to `async` with continuations. That is why the daemon
|
|
/// can drive two VMs from an actor without ever touching the main queue for VM
|
|
/// control — though the process still needs a running `NSApplication` main loop
|
|
/// for the framework itself (see ``CommandDaemon``).
|
|
public final class VMInstance: @unchecked Sendable {
|
|
|
|
/// The bundle this instance was created from.
|
|
public let bundle: VMBundle
|
|
|
|
/// A caller-supplied label used in log messages, typically `slot-0`.
|
|
public let label: String
|
|
|
|
/// The private serial queue every `VZVirtualMachine` call and every delegate
|
|
/// callback runs on. `VZVirtualMachine` is not thread-safe; this queue *is*
|
|
/// its thread-safety.
|
|
private let queue: DispatchQueue
|
|
|
|
/// Only ever touched on ``queue``.
|
|
private let vm: VZVirtualMachine
|
|
|
|
/// Retained explicitly: `VZVirtualMachine.delegate` is a weak reference.
|
|
private let vmDelegate: VMInstanceDelegate
|
|
|
|
/// Guards ``stopReason`` and ``activityToken``. A plain lock rather than an
|
|
/// actor so the delegate callback — which arrives on ``queue`` and must not
|
|
/// block on an await — can publish the stop synchronously.
|
|
private let lock = NSLock()
|
|
private var stopReason: VMStopReason?
|
|
private var activityToken: (any NSObjectProtocol)?
|
|
|
|
/// Wraps a non-`Sendable` value so it can cross into a `@Sendable` closure
|
|
/// that immediately hops onto ``queue``, which is the only place it is used.
|
|
private struct Unchecked<T>: @unchecked Sendable {
|
|
let value: T
|
|
}
|
|
|
|
/// Creates an instance and its underlying `VZVirtualMachine`.
|
|
///
|
|
/// - Parameters:
|
|
/// - bundle: The VM bundle to boot. Usually an ephemeral clone.
|
|
/// - label: Log label.
|
|
/// - headless: Passed through to ``VZConfigFactory``.
|
|
/// - Throws: Configuration or validation failures.
|
|
public init(bundle: VMBundle, label: String, headless: Bool = true) throws {
|
|
self.bundle = bundle
|
|
self.label = label
|
|
let vmQueue = DispatchQueue(label: "vm.\(label).\(bundle.name)", qos: .userInitiated)
|
|
self.queue = vmQueue
|
|
|
|
let configuration = try VZConfigFactory.makeConfiguration(bundle: bundle, headless: headless)
|
|
let delegate = VMInstanceDelegate()
|
|
self.vmDelegate = delegate
|
|
|
|
// Constructed on the queue it will be driven from, so no VZ object is
|
|
// ever created on one thread and used from another.
|
|
self.vm = vmQueue.sync {
|
|
let machine = VZVirtualMachine(configuration: configuration, queue: vmQueue)
|
|
machine.delegate = delegate
|
|
return machine
|
|
}
|
|
|
|
delegate.onStop = { [weak self] reason in
|
|
self?.finishStop(reason)
|
|
}
|
|
}
|
|
|
|
/// The framework's current state, read on the VM queue.
|
|
public var state: VZVirtualMachine.State {
|
|
get async {
|
|
// The raw value crosses the concurrency boundary rather than the
|
|
// enum, so no assumption is made about the imported type's Sendable
|
|
// conformance.
|
|
let raw: Int = await withCheckedContinuation { continuation in
|
|
queue.async {
|
|
continuation.resume(returning: self.vm.state.rawValue)
|
|
}
|
|
}
|
|
return VZVirtualMachine.State(rawValue: raw) ?? .stopped
|
|
}
|
|
}
|
|
|
|
/// Whether the VM is running or in a transitional state.
|
|
public var isActive: Bool {
|
|
get async {
|
|
switch await state {
|
|
case .stopped, .error:
|
|
return false
|
|
default:
|
|
return true
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Starts the VM.
|
|
///
|
|
/// - Parameter options: Optional start options. The install/provision path
|
|
/// passes a `VZMacOSVirtualMachineStartOptions` — on macOS 27+ hosts that
|
|
/// is also where Setup Assistant automation is attached (see
|
|
/// ``GuestProvisioner`` and docs/DESIGN.md, Verified Fact 9). Pass `nil`
|
|
/// for a normal boot of an already-provisioned clone.
|
|
/// - Throws: ``CoreError/vmLimitExceeded`` when Apple's kernel-enforced cap
|
|
/// of **two** concurrent macOS guests is hit — the framework raises
|
|
/// `VZError.virtualMachineLimitExceeded` from `start()`, and that case is
|
|
/// translated here rather than propagated, because the scheduler treats it
|
|
/// as transient back-pressure rather than a failure.
|
|
public func start(options: VZMacOSVirtualMachineStartOptions? = nil) async throws {
|
|
clearStopReason()
|
|
|
|
let boxed = Unchecked(value: options)
|
|
do {
|
|
try await withCheckedThrowingContinuation {
|
|
(continuation: CheckedContinuation<Void, any Error>) in
|
|
queue.async {
|
|
if let options = boxed.value {
|
|
// The install/provision path: on a macOS 27+ host these
|
|
// options carry the Setup Assistant automation. Note the
|
|
// options-taking overload reports failure as an optional
|
|
// Error, not a Result.
|
|
self.vm.start(options: options) { error in
|
|
if let error {
|
|
continuation.resume(throwing: error)
|
|
} else {
|
|
continuation.resume()
|
|
}
|
|
}
|
|
} else {
|
|
self.vm.start { result in
|
|
switch result {
|
|
case .success:
|
|
continuation.resume()
|
|
case .failure(let error):
|
|
continuation.resume(throwing: error)
|
|
}
|
|
}
|
|
}
|
|
}
|
|
}
|
|
} catch {
|
|
throw Self.mapVZError(error)
|
|
}
|
|
|
|
// Hold a power assertion for the VM's lifetime: a CI guest that is
|
|
// building for twenty minutes over SSH looks completely idle to the host,
|
|
// and letting the Mac sleep underneath it would stall the job.
|
|
beginActivityAssertion()
|
|
}
|
|
|
|
/// NSLock's `lock`/`unlock` are unavailable from an async context, so every
|
|
/// critical section lives in a synchronous helper.
|
|
private func clearStopReason() {
|
|
lock.lock()
|
|
defer { lock.unlock() }
|
|
stopReason = nil
|
|
}
|
|
|
|
/// Takes the power assertion, unless the VM already stopped in the meantime.
|
|
private func beginActivityAssertion() {
|
|
let token = ProcessInfo.processInfo.beginActivity(
|
|
options: [.userInitiated, .idleSystemSleepDisabled],
|
|
reason: "running macOS CI guest \(label) (\(bundle.name))"
|
|
)
|
|
lock.lock()
|
|
let alreadyStopped = stopReason != nil
|
|
if !alreadyStopped {
|
|
activityToken = token
|
|
}
|
|
lock.unlock()
|
|
if alreadyStopped {
|
|
// Raced with an immediate stop; don't strand the assertion.
|
|
ProcessInfo.processInfo.endActivity(token)
|
|
}
|
|
}
|
|
|
|
/// Publishes a terminal stop and releases the power assertion. Idempotent:
|
|
/// the first reason wins, so a `didStopWithError` following a `requestStop`
|
|
/// cannot overwrite an already-recorded outcome.
|
|
private func finishStop(_ reason: VMStopReason) {
|
|
lock.lock()
|
|
if stopReason == nil {
|
|
stopReason = reason
|
|
}
|
|
let token = activityToken
|
|
activityToken = nil
|
|
lock.unlock()
|
|
if let token {
|
|
ProcessInfo.processInfo.endActivity(token)
|
|
}
|
|
}
|
|
|
|
/// The recorded stop reason, if the VM has already stopped.
|
|
private var recordedStopReason: VMStopReason? {
|
|
lock.lock()
|
|
defer { lock.unlock() }
|
|
return stopReason
|
|
}
|
|
|
|
/// Asks the guest to shut down, then force-stops if it does not.
|
|
///
|
|
/// Tries `requestStop()` first — that delivers an ACPI-equivalent power
|
|
/// button press, giving the guest a chance to flush its filesystem — and
|
|
/// falls back to `stop()` after `gracePeriod`. Never throws: teardown must
|
|
/// always complete so the slot can be recycled.
|
|
///
|
|
/// - Parameter gracePeriod: How long to wait for a graceful stop.
|
|
/// - Returns: Why the VM ended up stopped.
|
|
@discardableResult
|
|
public func requestStopThenForce(gracePeriod: Duration = .seconds(30)) async -> VMStopReason {
|
|
if let reason = recordedStopReason { return reason }
|
|
if await !isActive {
|
|
// Stopped without a delegate callback ever landing (for example a
|
|
// start() that failed outright). Record it so waiters unblock.
|
|
finishStop(.requested)
|
|
return recordedStopReason ?? .requested
|
|
}
|
|
|
|
// Guest-cooperative first: requestStop() is the equivalent of a power
|
|
// button press, which lets the guest flush its filesystem.
|
|
_ = await withCheckedContinuation { (continuation: CheckedContinuation<Bool, Never>) in
|
|
queue.async {
|
|
guard self.vm.canRequestStop else {
|
|
continuation.resume(returning: false)
|
|
return
|
|
}
|
|
do {
|
|
try self.vm.requestStop()
|
|
continuation.resume(returning: true)
|
|
} catch {
|
|
// "not running", or the guest refused. Force is next either way.
|
|
continuation.resume(returning: false)
|
|
}
|
|
}
|
|
}
|
|
|
|
if let reason = await waitForStop(within: gracePeriod) {
|
|
return reason
|
|
}
|
|
|
|
// Grace elapsed — pull the plug. Teardown must always complete so the
|
|
// slot can be recycled, so every failure here is swallowed.
|
|
await withCheckedContinuation { (continuation: CheckedContinuation<Void, Never>) in
|
|
queue.async {
|
|
guard self.vm.canStop else {
|
|
continuation.resume()
|
|
return
|
|
}
|
|
self.vm.stop { _ in
|
|
continuation.resume()
|
|
}
|
|
}
|
|
}
|
|
|
|
if let reason = await waitForStop(within: .seconds(10)) {
|
|
return reason
|
|
}
|
|
// The framework never told us; treat it as stopped regardless rather
|
|
// than leaving the caller blocked on a dead slot.
|
|
finishStop(.requested)
|
|
return recordedStopReason ?? .requested
|
|
}
|
|
|
|
/// Polls for a recorded stop for at most `limit`. Returns `nil` on timeout.
|
|
///
|
|
/// Polling rather than a parked continuation keeps this cancellable and
|
|
/// leak-free: a continuation registered for a VM that never stops would be
|
|
/// stranded forever.
|
|
private func waitForStop(within limit: Duration) async -> VMStopReason? {
|
|
let deadline = ContinuousClock.now.advanced(by: limit)
|
|
while true {
|
|
if let reason = recordedStopReason { return reason }
|
|
if ContinuousClock.now >= deadline { return nil }
|
|
do {
|
|
try await Task.sleep(for: .milliseconds(200))
|
|
} catch {
|
|
return recordedStopReason
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Suspends until the VM stops for any reason.
|
|
///
|
|
/// - Returns: Why it stopped.
|
|
public func waitUntilStopped() async -> VMStopReason {
|
|
while true {
|
|
if let reason = recordedStopReason { return reason }
|
|
// A VM that reaches .stopped or .error without a delegate callback
|
|
// (an unusual but observed path) must not hang the caller.
|
|
if await !isActive {
|
|
finishStop(.guestInitiated)
|
|
return recordedStopReason ?? .guestInitiated
|
|
}
|
|
do {
|
|
try await Task.sleep(for: .milliseconds(500))
|
|
} catch {
|
|
return recordedStopReason ?? .requested
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Translates a Virtualization error into a ``CoreError``.
|
|
///
|
|
/// `VZError.Code.virtualMachineLimitExceeded` becomes
|
|
/// ``CoreError/vmLimitExceeded``; everything else becomes
|
|
/// ``CoreError/provisioningFailed(_:)`` carrying the framework's message.
|
|
public static func mapVZError(_ error: any Error) -> CoreError {
|
|
if let coreError = error as? CoreError { return coreError }
|
|
|
|
// Apple's kernel-enforced cap of two concurrent macOS guests
|
|
// (docs/DESIGN.md, Verified Fact 8). The scheduler treats this as
|
|
// transient back-pressure, so it must stay distinguishable.
|
|
if let vzError = error as? VZError, vzError.code == .virtualMachineLimitExceeded {
|
|
return .vmLimitExceeded
|
|
}
|
|
let nsError = error as NSError
|
|
if nsError.domain == VZErrorDomain,
|
|
nsError.code == VZError.Code.virtualMachineLimitExceeded.rawValue
|
|
{
|
|
return .vmLimitExceeded
|
|
}
|
|
return .provisioningFailed(nsError.localizedDescription)
|
|
}
|
|
}
|
|
|
|
/// Bridges `VZVirtualMachineDelegate` callbacks back into ``VMInstance``.
|
|
///
|
|
/// Kept as a separate object so ``VMInstance`` need not inherit `NSObject`, and
|
|
/// so the delegate's lifetime is explicitly owned rather than accidentally
|
|
/// retained by the framework.
|
|
final class VMInstanceDelegate: NSObject, VZVirtualMachineDelegate {
|
|
/// Invoked on the VM queue whenever the machine stops.
|
|
var onStop: (@Sendable (VMStopReason) -> Void)?
|
|
|
|
func guestDidStop(_ virtualMachine: VZVirtualMachine) {
|
|
onStop?(.guestInitiated)
|
|
}
|
|
|
|
func virtualMachine(_ virtualMachine: VZVirtualMachine, didStopWithError error: any Error) {
|
|
onStop?(.failed((error as NSError).localizedDescription))
|
|
}
|
|
|
|
func virtualMachine(
|
|
_ virtualMachine: VZVirtualMachine,
|
|
networkDevice: VZNetworkDevice,
|
|
attachmentWasDisconnectedWithError error: any Error
|
|
) {
|
|
// NAT attachments do drop transiently. The VM keeps running and the
|
|
// guest's DHCP client recovers, so this is deliberately not treated as a
|
|
// stop — the boot/job timeouts are what catch a guest that never comes
|
|
// back onto the network.
|
|
}
|
|
}
|