Merge nucleic/tidy-north-gecko-6wqv into dev

This commit is contained in:
2026-07-17 15:35:17 -07:00
parent 9de6a6cee7
commit 3168f3d8b6
4 changed files with 212 additions and 27 deletions
+69 -23
View File
@@ -109,12 +109,24 @@ extension UnixSocketRelay {
$0.t = Task {
do {
for try await connection in connectionStream {
try await self.handleHostUnixConn(
hostConn: connection,
port: self.port,
vm: self.vm,
log: self.log
)
// [Nucleic vendored patch] Contain per-connection failures. A thrown dial
// (guest listener briefly absent, transient vsock error) used to propagate
// out of the loop and end the relay for the CONTAINER'S LIFETIME — every
// later client of this socket then failed, unrecoverably. One bad
// connection must only fail that connection. Handled on its own task so a
// slow dial can't head-of-line-block later connections either.
Task {
do {
try await self.handleHostUnixConn(
hostConn: connection,
port: self.port,
vm: self.vm,
log: self.log
)
} catch {
self.log?.error("failed to relay host unix connection: \(error)")
}
}
}
} catch {
log?.error("failed in unix socket relay loop: \(error)")
@@ -138,18 +150,30 @@ extension UnixSocketRelay {
state.withLock {
$0.listener = listener
$0.t = Task {
do {
defer { try? listener.finish() }
for await connection in listener {
try await self.handleGuestVsockConn(
vsockConn: connection,
hostConnectionPath: hostPath,
port: self.port,
log: self.log
)
defer { try? listener.finish() }
for await connection in listener {
// [Nucleic vendored patch] Contain per-connection failures. This loop is the
// host half of a container's relayed control socket: a single thrown connect
// (host server rebinding, listen backlog momentarily full → ECONNREFUSED, fd
// pressure) used to propagate out of the loop, whose defer then tore down the
// vsock listener — permanently severing EVERY session in the container from the
// host control plane until the container was recreated (an app restart). One
// bad connection must only fail that connection; the client retries. Handled on
// its own task so a slow host connect can't head-of-line-block later guest
// connections.
Task {
do {
try await self.handleGuestVsockConn(
vsockConn: connection,
hostConnectionPath: hostPath,
port: self.port,
log: self.log
)
} catch {
self.log?.error(
"failed to relay between vsock \(self.port) and \(hostPath.path): \(error)")
}
}
} catch {
self.log?.error("failed to setup relay between vsock \(self.port) and \(hostPath.path): \(error)")
}
}
}
@@ -176,6 +200,10 @@ extension UnixSocketRelay {
)
} catch {
log?.error("failed to relay between vsock \(port) and \(hostConn)")
// [Nucleic vendored patch] Close the accepted client connection on failure so the peer
// sees a prompt EOF (fail fast, retryable) instead of a half-open socket, and its fd
// isn't leaked (acceptStream vends closeOnDeinit: false).
try? hostConn.close()
throw error
}
}
@@ -186,12 +214,22 @@ extension UnixSocketRelay {
port: UInt32,
log: Logger?
) async throws {
// [Nucleic vendored patch] Any failure before the relay owns the fds must close BOTH ends:
// the guest connection so the in-guest client sees a prompt EOF (fail fast, retryable — not
// a half-open socket it waits on forever), and the freshly-made host socket so its fd isn't
// leaked (closeOnDeinit is false).
let hostPath = hostConnectionPath.path
let socketType = try UnixType(path: hostPath)
let hostSocket = try Socket(
type: socketType,
closeOnDeinit: false
)
let hostSocket: Socket
do {
let socketType = try UnixType(path: hostPath)
hostSocket = try Socket(
type: socketType,
closeOnDeinit: false
)
} catch {
try? vsockConn.close()
throw error
}
log?.debug(
"initiating connection from guest to host",
metadata: [
@@ -199,7 +237,13 @@ extension UnixSocketRelay {
"hostFd": "\(hostSocket.fileDescriptor)",
"guestFd": "\(vsockConn.fileDescriptor)",
])
try hostSocket.connect()
do {
try hostSocket.connect()
} catch {
try? hostSocket.close()
try? vsockConn.close()
throw error
}
do {
try await self.relay(
@@ -208,6 +252,8 @@ extension UnixSocketRelay {
)
} catch {
log?.error("failed to relay between vsock \(port) and \(hostPath)")
try? hostSocket.close()
try? vsockConn.close()
}
}
@@ -130,6 +130,21 @@ extension Socket {
SocketError.withErrno("\(msg) (\(_errnoString(errno)))", errno: errno)
}
/// [Nucleic vendored patch] Whether an `accept(2)` failure is transient — the listener is still
/// healthy and later accepts can succeed — as opposed to a dead listener. ECONNABORTED/ECONNRESET
/// are a single queued connection dying before accept; EMFILE/ENFILE/ENOBUFS/ENOMEM are resource
/// pressure that clears; EINTR/EAGAIN are spurious wakeups. See `acceptStream`.
static func isTransientAcceptError(_ error: Swift.Error) -> Bool {
guard case SocketError.withErrno(_, let code) = error else { return false }
switch code {
case ECONNABORTED, ECONNRESET, EINTR, EAGAIN, EWOULDBLOCK, EMFILE, ENFILE, ENOBUFS, ENOMEM,
EPROTO:
return true
default:
return false
}
}
public func connect() throws {
try state.withLock { currentState in
guard currentState.socketState == .created else {
@@ -277,6 +292,17 @@ extension Socket {
} catch SocketError.closed {
source.cancel()
} catch {
// [Nucleic vendored patch] One-strike accepting is a control-plane brick: a
// single transient accept(2) failure — ECONNABORTED (peer aborted while queued,
// routine under connection churn), EMFILE/ENFILE (fd pressure), ENOBUFS/ENOMEM,
// EINTR — used to cancel the source FOREVER while the socket stayed bound and
// listening. Every later client then connect(2)ed into the kernel backlog and
// hung unanswered (a silent black hole — the "agent produced no output within
// 60s" stall when this socket is a relayed control plane). Skip the failed
// accept and keep listening; only a genuinely dead listener ends the stream.
if Self.isTransientAcceptError(error) {
return
}
cont.yield(with: .failure(error))
source.cancel()
}