Fix the recurring session-stall pair: MainActor lock-reconcile hang + vminitd relay spin
Host (dominant): AppStore.reconcileLocks polls every 3s on the MainActor while any lock is held — effectively forever, since interrupted/errored sessions deliberately retain locks. Each pass ran heldPathDisposition's diverges() as three held×unmerged scans with two split-allocations per pathsOverlap call, pinning the main thread for tens of seconds per pass on a diverged trunk (hang-reports 2026-07-28: 100% of samples in reconcileLocks→heldPathDisposition→pathsOverlap). That froze running sessions' transcripts and starved the spawn path into the 60s "produced no output" watchdog. pathsOverlap is now allocation-free bytewise comparison with identical semantics, and divergentHeldPaths answers all three questions from one O((held+unmerged)·depth) set. Guest (persistence): VsockProxy threaded ONE offset pair through BOTH relay directions; once the EAGAIN-return backpressure patch let pending bytes persist, traffic in the other direction skewed the shared counters, made the write leg unreachable, and spun the single ProcessSupervisor poller thread forever — container-wide dead control plane until VM recreation, triggered by exactly the backpressure the host hang created. Each direction now owns its own pipe and counters (OSFile.RelayDirection), and source EOF is only surfaced after the pipe drains so SHUT_WR can't truncate a parked tail. Vendored patch docs updated (#15); inert until the initfs image is rebuilt+repointed. Co-Authored-By: Claude Fable 5 <[email protected]>
This commit is contained in:
@@ -20,95 +20,103 @@ import Foundation
|
||||
import LCShim
|
||||
|
||||
extension OSFile {
|
||||
struct SpliceFile: Sendable {
|
||||
fileprivate var file: OSFile
|
||||
fileprivate var offset: Int
|
||||
fileprivate let pipe = Pipe()
|
||||
/// [Nucleic vendored patch] One direction of a bidirectional relay: `from` fd → transfer
|
||||
/// pipe → `to` fd. Each direction owns its OWN pipe and byte counters.
|
||||
///
|
||||
/// The previous `SpliceFile` design threaded ONE offset pair through BOTH directions of
|
||||
/// `VsockProxy`'s relay (a fd's struct served as read-counter in one direction and
|
||||
/// write-counter in the other). That was survivable only while every splice call fully
|
||||
/// drained its pipe before returning. Once the EAGAIN-return backpressure patch let pending
|
||||
/// bytes persist across calls, one parked direction skewed the shared counters for the
|
||||
/// other: its write leg's `to.offset < from.offset` guard went false with data still in the
|
||||
/// pipe, and the outer loop then alternated read-EAGAIN/skip-write forever — a hard spin on
|
||||
/// vminitd's single ProcessSupervisor poller thread that froze every exec's stdio and every
|
||||
/// control-plane relay in the container until the VM was recreated (the persistent
|
||||
/// "produced no output within 60s" / dead-control-plane state).
|
||||
struct RelayDirection: Sendable {
|
||||
let from: Int32
|
||||
let to: Int32
|
||||
private let pipe = Pipe()
|
||||
/// Bytes spliced from `from` into the transfer pipe so far.
|
||||
fileprivate var bytesIn = 0
|
||||
/// Bytes spliced from the transfer pipe into `to` so far.
|
||||
fileprivate var bytesOut = 0
|
||||
/// The source reported EOF. The direction only FINISHES (`.eof`) once the pipe has
|
||||
/// also drained, so the stream's tail is never dropped by an early SHUT_WR.
|
||||
fileprivate var sawSourceEOF = false
|
||||
|
||||
var fileDescriptor: Int32 {
|
||||
file.fileDescriptor
|
||||
}
|
||||
/// Bytes read from the source that the destination hasn't accepted yet.
|
||||
var pendingBytes: Int { bytesIn - bytesOut }
|
||||
|
||||
var reader: Int32 {
|
||||
pipe.fileHandleForReading.fileDescriptor
|
||||
}
|
||||
fileprivate var pipeReader: Int32 { pipe.fileHandleForReading.fileDescriptor }
|
||||
fileprivate var pipeWriter: Int32 { pipe.fileHandleForWriting.fileDescriptor }
|
||||
|
||||
var writer: Int32 {
|
||||
pipe.fileHandleForWriting.fileDescriptor
|
||||
}
|
||||
|
||||
init(fd: Int32) {
|
||||
self.file = OSFile(fd: fd)
|
||||
self.offset = 0
|
||||
}
|
||||
|
||||
init(handle: FileHandle) {
|
||||
self.file = OSFile(handle: handle)
|
||||
self.offset = 0
|
||||
}
|
||||
|
||||
init(from: OSFile, withOffset: Int = 0) {
|
||||
self.file = from
|
||||
self.offset = withOffset
|
||||
}
|
||||
|
||||
func close() throws {
|
||||
try self.file.close()
|
||||
init(from: Int32, to: Int32) {
|
||||
self.from = from
|
||||
self.to = to
|
||||
}
|
||||
}
|
||||
|
||||
static func splice(from: inout SpliceFile, to: inout SpliceFile, count: Int = 1 << 16) throws -> (read: Int, wrote: Int, action: IOAction) {
|
||||
let fromOffset = from.offset
|
||||
let toOffset = to.offset
|
||||
/// The terminal state of one `relay` pass over a direction.
|
||||
enum RelayResult: Sendable {
|
||||
/// No more progress possible right now: the source has no data (EAGAIN) or the
|
||||
/// destination is full (its EPOLLOUT edge resumes the flush of `pendingBytes`).
|
||||
case idle
|
||||
/// Source EOF observed AND the pipe fully drained — the direction is complete; the
|
||||
/// caller should SHUT_WR the destination.
|
||||
case eof
|
||||
/// The destination hung up mid-write; nothing further can be delivered.
|
||||
case brokenPipe
|
||||
}
|
||||
|
||||
/// Move as much data as possible along `direction` without blocking. `count` bounds the
|
||||
/// bytes buffered in the transfer pipe (it matches the default pipe capacity).
|
||||
static func relay(_ direction: inout RelayDirection, count: Int = 1 << 16) throws -> RelayResult {
|
||||
let flags = UInt32(bitPattern: LCShim.SPLICE_F_MOVE | LCShim.SPLICE_F_NONBLOCK)
|
||||
while true {
|
||||
while (from.offset - to.offset) < count {
|
||||
let toRead = count - (from.offset - to.offset)
|
||||
let bytesRead = LCShim.splice(from.fileDescriptor, nil, to.writer, nil, toRead, UInt32(bitPattern: LCShim.SPLICE_F_MOVE | LCShim.SPLICE_F_NONBLOCK))
|
||||
if bytesRead == -1 {
|
||||
// Read leg: source → pipe, until the pipe is full, the source runs dry, or EOF.
|
||||
var sourceDry = false
|
||||
if !direction.sawSourceEOF {
|
||||
while direction.pendingBytes < count {
|
||||
let toRead = count - direction.pendingBytes
|
||||
let n = LCShim.splice(direction.from, nil, direction.pipeWriter, nil, toRead, flags)
|
||||
if n == -1 {
|
||||
if errno != EAGAIN && errno != EIO {
|
||||
throw POSIXError(.init(rawValue: errno)!)
|
||||
}
|
||||
sourceDry = true
|
||||
break
|
||||
}
|
||||
if n == 0 {
|
||||
direction.sawSourceEOF = true
|
||||
break
|
||||
}
|
||||
direction.bytesIn += n
|
||||
if n < toRead { break }
|
||||
}
|
||||
}
|
||||
// Write leg: pipe → destination, until drained or the destination pushes back.
|
||||
while direction.pendingBytes > 0 {
|
||||
let n = LCShim.splice(direction.pipeReader, nil, direction.to, nil, direction.pendingBytes, flags)
|
||||
if n == -1 {
|
||||
if errno != EAGAIN && errno != EIO {
|
||||
throw POSIXError(.init(rawValue: errno)!)
|
||||
}
|
||||
break
|
||||
// Destination full: park with the remainder in the pipe. The destination
|
||||
// fd's EPOLLOUT edge re-enters and resumes exactly here — never spin, and
|
||||
// never block the shared poller thread.
|
||||
return .idle
|
||||
}
|
||||
if bytesRead == 0 {
|
||||
return (0, 0, .eof)
|
||||
}
|
||||
from.offset += bytesRead
|
||||
if bytesRead < toRead {
|
||||
break
|
||||
}
|
||||
}
|
||||
if from.offset == to.offset {
|
||||
return (from.offset - fromOffset, to.offset - toOffset, .success)
|
||||
}
|
||||
while to.offset < from.offset {
|
||||
let toWrite = from.offset - to.offset
|
||||
let bytesWrote = LCShim.splice(to.reader, nil, to.fileDescriptor, nil, toWrite, UInt32(bitPattern: LCShim.SPLICE_F_MOVE | LCShim.SPLICE_F_NONBLOCK))
|
||||
if bytesWrote == -1 {
|
||||
if errno != EAGAIN && errno != EIO {
|
||||
throw POSIXError(.init(rawValue: errno)!)
|
||||
}
|
||||
// [Nucleic vendored patch] Destination full: RETURN, don't `break`. Breaking
|
||||
// sent the outer `while true` straight back here — with the source idle and the
|
||||
// destination still full, neither leg could progress and this spun the caller's
|
||||
// thread at 100% until the peer drained. The caller is an epoll handler on
|
||||
// vminitd's SINGLE ProcessSupervisor poller thread, so the spin froze every
|
||||
// exec's stdio and every control-plane relay in the container at once (the
|
||||
// all-sessions "produced no output within 60s" stall / dead control plane).
|
||||
// The un-flushed bytes stay in the transfer pipe (`from.offset > to.offset`
|
||||
// persists in the SpliceFiles); the destination fd is registered for EPOLLOUT,
|
||||
// whose edge re-enters transferData and resumes the flush.
|
||||
return (from.offset - fromOffset, to.offset - toOffset, .success)
|
||||
}
|
||||
to.offset += bytesWrote
|
||||
if bytesWrote == 0 {
|
||||
return (from.offset - fromOffset, to.offset - toOffset, .brokenPipe)
|
||||
}
|
||||
if bytesWrote < toWrite {
|
||||
break
|
||||
if n == 0 {
|
||||
return .brokenPipe
|
||||
}
|
||||
direction.bytesOut += n
|
||||
}
|
||||
// Pipe is drained here.
|
||||
if direction.sawSourceEOF { return .eof }
|
||||
if sourceDry { return .idle }
|
||||
// The read leg stopped only because the pipe filled (or read a full window):
|
||||
// go around again — the source may still have data.
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user