Merge nucleic/mellow-dewy-falcon-rjhr into main
This commit is contained in:
@@ -0,0 +1,373 @@
|
||||
import Foundation
|
||||
import RunnerCore
|
||||
import Virtualization
|
||||
|
||||
/// Why a VM stopped.
|
||||
public enum VMStopReason: Sendable, Equatable {
|
||||
/// The guest shut itself down (our normal path: the SSH session runs
|
||||
/// `shutdown`, or `gitea-runner daemon` exits and provisioning halts it).
|
||||
case guestInitiated
|
||||
/// We asked it to stop and it complied.
|
||||
case requested
|
||||
/// The framework reported an error.
|
||||
case failed(String)
|
||||
}
|
||||
|
||||
/// Owns one live `VZVirtualMachine` and exposes it as an `async` API.
|
||||
///
|
||||
/// ## Threading
|
||||
///
|
||||
/// `VZVirtualMachine` is not thread-safe and must be used only from the queue it
|
||||
/// was created with. This class creates it with
|
||||
/// `VZVirtualMachine(configuration:queue:)` on a **private serial queue** and
|
||||
/// funnels every call through that queue, bridging the framework's
|
||||
/// completion-handler API to `async` with continuations. That is why the daemon
|
||||
/// can drive two VMs from an actor without ever touching the main queue for VM
|
||||
/// control — though the process still needs a running `NSApplication` main loop
|
||||
/// for the framework itself (see ``CommandDaemon``).
|
||||
public final class VMInstance: @unchecked Sendable {
|
||||
|
||||
/// The bundle this instance was created from.
|
||||
public let bundle: VMBundle
|
||||
|
||||
/// A caller-supplied label used in log messages, typically `slot-0`.
|
||||
public let label: String
|
||||
|
||||
/// The private serial queue every `VZVirtualMachine` call and every delegate
|
||||
/// callback runs on. `VZVirtualMachine` is not thread-safe; this queue *is*
|
||||
/// its thread-safety.
|
||||
private let queue: DispatchQueue
|
||||
|
||||
/// Only ever touched on ``queue``.
|
||||
private let vm: VZVirtualMachine
|
||||
|
||||
/// Retained explicitly: `VZVirtualMachine.delegate` is a weak reference.
|
||||
private let vmDelegate: VMInstanceDelegate
|
||||
|
||||
/// Guards ``stopReason`` and ``activityToken``. A plain lock rather than an
|
||||
/// actor so the delegate callback — which arrives on ``queue`` and must not
|
||||
/// block on an await — can publish the stop synchronously.
|
||||
private let lock = NSLock()
|
||||
private var stopReason: VMStopReason?
|
||||
private var activityToken: (any NSObjectProtocol)?
|
||||
|
||||
/// Wraps a non-`Sendable` value so it can cross into a `@Sendable` closure
|
||||
/// that immediately hops onto ``queue``, which is the only place it is used.
|
||||
private struct Unchecked<T>: @unchecked Sendable {
|
||||
let value: T
|
||||
}
|
||||
|
||||
/// Creates an instance and its underlying `VZVirtualMachine`.
|
||||
///
|
||||
/// - Parameters:
|
||||
/// - bundle: The VM bundle to boot. Usually an ephemeral clone.
|
||||
/// - label: Log label.
|
||||
/// - headless: Passed through to ``VZConfigFactory``.
|
||||
/// - Throws: Configuration or validation failures.
|
||||
public init(bundle: VMBundle, label: String, headless: Bool = true) throws {
|
||||
self.bundle = bundle
|
||||
self.label = label
|
||||
let vmQueue = DispatchQueue(label: "vm.\(label).\(bundle.name)", qos: .userInitiated)
|
||||
self.queue = vmQueue
|
||||
|
||||
let configuration = try VZConfigFactory.makeConfiguration(bundle: bundle, headless: headless)
|
||||
let delegate = VMInstanceDelegate()
|
||||
self.vmDelegate = delegate
|
||||
|
||||
// Constructed on the queue it will be driven from, so no VZ object is
|
||||
// ever created on one thread and used from another.
|
||||
self.vm = vmQueue.sync {
|
||||
let machine = VZVirtualMachine(configuration: configuration, queue: vmQueue)
|
||||
machine.delegate = delegate
|
||||
return machine
|
||||
}
|
||||
|
||||
delegate.onStop = { [weak self] reason in
|
||||
self?.finishStop(reason)
|
||||
}
|
||||
}
|
||||
|
||||
/// The framework's current state, read on the VM queue.
|
||||
public var state: VZVirtualMachine.State {
|
||||
get async {
|
||||
// The raw value crosses the concurrency boundary rather than the
|
||||
// enum, so no assumption is made about the imported type's Sendable
|
||||
// conformance.
|
||||
let raw: Int = await withCheckedContinuation { continuation in
|
||||
queue.async {
|
||||
continuation.resume(returning: self.vm.state.rawValue)
|
||||
}
|
||||
}
|
||||
return VZVirtualMachine.State(rawValue: raw) ?? .stopped
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether the VM is running or in a transitional state.
|
||||
public var isActive: Bool {
|
||||
get async {
|
||||
switch await state {
|
||||
case .stopped, .error:
|
||||
return false
|
||||
default:
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Starts the VM.
|
||||
///
|
||||
/// - Parameter options: Optional start options. The install/provision path
|
||||
/// passes a `VZMacOSVirtualMachineStartOptions` — on macOS 27+ hosts that
|
||||
/// is also where Setup Assistant automation is attached (see
|
||||
/// ``GuestProvisioner`` and docs/DESIGN.md, Verified Fact 9). Pass `nil`
|
||||
/// for a normal boot of an already-provisioned clone.
|
||||
/// - Throws: ``CoreError/vmLimitExceeded`` when Apple's kernel-enforced cap
|
||||
/// of **two** concurrent macOS guests is hit — the framework raises
|
||||
/// `VZError.virtualMachineLimitExceeded` from `start()`, and that case is
|
||||
/// translated here rather than propagated, because the scheduler treats it
|
||||
/// as transient back-pressure rather than a failure.
|
||||
public func start(options: VZMacOSVirtualMachineStartOptions? = nil) async throws {
|
||||
clearStopReason()
|
||||
|
||||
let boxed = Unchecked(value: options)
|
||||
do {
|
||||
try await withCheckedThrowingContinuation {
|
||||
(continuation: CheckedContinuation<Void, any Error>) in
|
||||
queue.async {
|
||||
if let options = boxed.value {
|
||||
// The install/provision path: on a macOS 27+ host these
|
||||
// options carry the Setup Assistant automation. Note the
|
||||
// options-taking overload reports failure as an optional
|
||||
// Error, not a Result.
|
||||
self.vm.start(options: options) { error in
|
||||
if let error {
|
||||
continuation.resume(throwing: error)
|
||||
} else {
|
||||
continuation.resume()
|
||||
}
|
||||
}
|
||||
} else {
|
||||
self.vm.start { result in
|
||||
switch result {
|
||||
case .success:
|
||||
continuation.resume()
|
||||
case .failure(let error):
|
||||
continuation.resume(throwing: error)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
throw Self.mapVZError(error)
|
||||
}
|
||||
|
||||
// Hold a power assertion for the VM's lifetime: a CI guest that is
|
||||
// building for twenty minutes over SSH looks completely idle to the host,
|
||||
// and letting the Mac sleep underneath it would stall the job.
|
||||
beginActivityAssertion()
|
||||
}
|
||||
|
||||
/// NSLock's `lock`/`unlock` are unavailable from an async context, so every
|
||||
/// critical section lives in a synchronous helper.
|
||||
private func clearStopReason() {
|
||||
lock.lock()
|
||||
defer { lock.unlock() }
|
||||
stopReason = nil
|
||||
}
|
||||
|
||||
/// Takes the power assertion, unless the VM already stopped in the meantime.
|
||||
private func beginActivityAssertion() {
|
||||
let token = ProcessInfo.processInfo.beginActivity(
|
||||
options: [.userInitiated, .idleSystemSleepDisabled],
|
||||
reason: "running macOS CI guest \(label) (\(bundle.name))"
|
||||
)
|
||||
lock.lock()
|
||||
let alreadyStopped = stopReason != nil
|
||||
if !alreadyStopped {
|
||||
activityToken = token
|
||||
}
|
||||
lock.unlock()
|
||||
if alreadyStopped {
|
||||
// Raced with an immediate stop; don't strand the assertion.
|
||||
ProcessInfo.processInfo.endActivity(token)
|
||||
}
|
||||
}
|
||||
|
||||
/// Publishes a terminal stop and releases the power assertion. Idempotent:
|
||||
/// the first reason wins, so a `didStopWithError` following a `requestStop`
|
||||
/// cannot overwrite an already-recorded outcome.
|
||||
private func finishStop(_ reason: VMStopReason) {
|
||||
lock.lock()
|
||||
if stopReason == nil {
|
||||
stopReason = reason
|
||||
}
|
||||
let token = activityToken
|
||||
activityToken = nil
|
||||
lock.unlock()
|
||||
if let token {
|
||||
ProcessInfo.processInfo.endActivity(token)
|
||||
}
|
||||
}
|
||||
|
||||
/// The recorded stop reason, if the VM has already stopped.
|
||||
private var recordedStopReason: VMStopReason? {
|
||||
lock.lock()
|
||||
defer { lock.unlock() }
|
||||
return stopReason
|
||||
}
|
||||
|
||||
/// Asks the guest to shut down, then force-stops if it does not.
|
||||
///
|
||||
/// Tries `requestStop()` first — that delivers an ACPI-equivalent power
|
||||
/// button press, giving the guest a chance to flush its filesystem — and
|
||||
/// falls back to `stop()` after `gracePeriod`. Never throws: teardown must
|
||||
/// always complete so the slot can be recycled.
|
||||
///
|
||||
/// - Parameter gracePeriod: How long to wait for a graceful stop.
|
||||
/// - Returns: Why the VM ended up stopped.
|
||||
@discardableResult
|
||||
public func requestStopThenForce(gracePeriod: Duration = .seconds(30)) async -> VMStopReason {
|
||||
if let reason = recordedStopReason { return reason }
|
||||
if await !isActive {
|
||||
// Stopped without a delegate callback ever landing (for example a
|
||||
// start() that failed outright). Record it so waiters unblock.
|
||||
finishStop(.requested)
|
||||
return recordedStopReason ?? .requested
|
||||
}
|
||||
|
||||
// Guest-cooperative first: requestStop() is the equivalent of a power
|
||||
// button press, which lets the guest flush its filesystem.
|
||||
_ = await withCheckedContinuation { (continuation: CheckedContinuation<Bool, Never>) in
|
||||
queue.async {
|
||||
guard self.vm.canRequestStop else {
|
||||
continuation.resume(returning: false)
|
||||
return
|
||||
}
|
||||
do {
|
||||
try self.vm.requestStop()
|
||||
continuation.resume(returning: true)
|
||||
} catch {
|
||||
// "not running", or the guest refused. Force is next either way.
|
||||
continuation.resume(returning: false)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if let reason = await waitForStop(within: gracePeriod) {
|
||||
return reason
|
||||
}
|
||||
|
||||
// Grace elapsed — pull the plug. Teardown must always complete so the
|
||||
// slot can be recycled, so every failure here is swallowed.
|
||||
await withCheckedContinuation { (continuation: CheckedContinuation<Void, Never>) in
|
||||
queue.async {
|
||||
guard self.vm.canStop else {
|
||||
continuation.resume()
|
||||
return
|
||||
}
|
||||
self.vm.stop { _ in
|
||||
continuation.resume()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if let reason = await waitForStop(within: .seconds(10)) {
|
||||
return reason
|
||||
}
|
||||
// The framework never told us; treat it as stopped regardless rather
|
||||
// than leaving the caller blocked on a dead slot.
|
||||
finishStop(.requested)
|
||||
return recordedStopReason ?? .requested
|
||||
}
|
||||
|
||||
/// Polls for a recorded stop for at most `limit`. Returns `nil` on timeout.
|
||||
///
|
||||
/// Polling rather than a parked continuation keeps this cancellable and
|
||||
/// leak-free: a continuation registered for a VM that never stops would be
|
||||
/// stranded forever.
|
||||
private func waitForStop(within limit: Duration) async -> VMStopReason? {
|
||||
let deadline = ContinuousClock.now.advanced(by: limit)
|
||||
while true {
|
||||
if let reason = recordedStopReason { return reason }
|
||||
if ContinuousClock.now >= deadline { return nil }
|
||||
do {
|
||||
try await Task.sleep(for: .milliseconds(200))
|
||||
} catch {
|
||||
return recordedStopReason
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Suspends until the VM stops for any reason.
|
||||
///
|
||||
/// - Returns: Why it stopped.
|
||||
public func waitUntilStopped() async -> VMStopReason {
|
||||
while true {
|
||||
if let reason = recordedStopReason { return reason }
|
||||
// A VM that reaches .stopped or .error without a delegate callback
|
||||
// (an unusual but observed path) must not hang the caller.
|
||||
if await !isActive {
|
||||
finishStop(.guestInitiated)
|
||||
return recordedStopReason ?? .guestInitiated
|
||||
}
|
||||
do {
|
||||
try await Task.sleep(for: .milliseconds(500))
|
||||
} catch {
|
||||
return recordedStopReason ?? .requested
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Translates a Virtualization error into a ``CoreError``.
|
||||
///
|
||||
/// `VZError.Code.virtualMachineLimitExceeded` becomes
|
||||
/// ``CoreError/vmLimitExceeded``; everything else becomes
|
||||
/// ``CoreError/provisioningFailed(_:)`` carrying the framework's message.
|
||||
public static func mapVZError(_ error: any Error) -> CoreError {
|
||||
if let coreError = error as? CoreError { return coreError }
|
||||
|
||||
// Apple's kernel-enforced cap of two concurrent macOS guests
|
||||
// (docs/DESIGN.md, Verified Fact 8). The scheduler treats this as
|
||||
// transient back-pressure, so it must stay distinguishable.
|
||||
if let vzError = error as? VZError, vzError.code == .virtualMachineLimitExceeded {
|
||||
return .vmLimitExceeded
|
||||
}
|
||||
let nsError = error as NSError
|
||||
if nsError.domain == VZErrorDomain,
|
||||
nsError.code == VZError.Code.virtualMachineLimitExceeded.rawValue
|
||||
{
|
||||
return .vmLimitExceeded
|
||||
}
|
||||
return .provisioningFailed(nsError.localizedDescription)
|
||||
}
|
||||
}
|
||||
|
||||
/// Bridges `VZVirtualMachineDelegate` callbacks back into ``VMInstance``.
|
||||
///
|
||||
/// Kept as a separate object so ``VMInstance`` need not inherit `NSObject`, and
|
||||
/// so the delegate's lifetime is explicitly owned rather than accidentally
|
||||
/// retained by the framework.
|
||||
final class VMInstanceDelegate: NSObject, VZVirtualMachineDelegate {
|
||||
/// Invoked on the VM queue whenever the machine stops.
|
||||
var onStop: (@Sendable (VMStopReason) -> Void)?
|
||||
|
||||
func guestDidStop(_ virtualMachine: VZVirtualMachine) {
|
||||
onStop?(.guestInitiated)
|
||||
}
|
||||
|
||||
func virtualMachine(_ virtualMachine: VZVirtualMachine, didStopWithError error: any Error) {
|
||||
onStop?(.failed((error as NSError).localizedDescription))
|
||||
}
|
||||
|
||||
func virtualMachine(
|
||||
_ virtualMachine: VZVirtualMachine,
|
||||
networkDevice: VZNetworkDevice,
|
||||
attachmentWasDisconnectedWithError error: any Error
|
||||
) {
|
||||
// NAT attachments do drop transiently. The VM keeps running and the
|
||||
// guest's DHCP client recovers, so this is deliberately not treated as a
|
||||
// stop — the boot/job timeouts are what catch a guest that never comes
|
||||
// back onto the network.
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user