Merge nucleic/tidy-north-gecko-6wqv into dev
This commit is contained in:
@@ -175,6 +175,11 @@ extension VsockProxy {
|
||||
)
|
||||
} catch {
|
||||
self.log?.error("failed to handle connection: \(error)")
|
||||
// [Nucleic vendored patch] A connection that failed before the relay
|
||||
// owned it must be closed, or its fd leaks for the proxy's lifetime
|
||||
// (accept vends closeOnDeinit: true, but the Socket is retained by the
|
||||
// stream's yielded value until then — close deterministically).
|
||||
try? conn.close()
|
||||
}
|
||||
}
|
||||
// Safe: actor serialization ensures this runs before connTask can execute its defer.
|
||||
@@ -183,10 +188,33 @@ extension VsockProxy {
|
||||
} catch {
|
||||
self.log?.error("failed to accept connection: \(error)")
|
||||
}
|
||||
// [Nucleic vendored patch] If this loop ever ends while the proxy is still nominally
|
||||
// running (fatal accept error), the listening socket MUST come down with it. Leaving it
|
||||
// bound-but-unaccepted turned the relayed control socket into a silent black hole: every
|
||||
// later client connect(2) SUCCEEDED into the kernel backlog and hung forever unanswered
|
||||
// — for Nucleic, every session in the container stalling with "produced no output within
|
||||
// 60s" until the VM was recreated. Closing the listener makes later connects fail fast
|
||||
// (ECONNREFUSED/ENOENT), which callers surface and retry.
|
||||
self.listenerLoopEnded()
|
||||
}
|
||||
self.task = task
|
||||
}
|
||||
|
||||
/// [Nucleic vendored patch] The accept loop ended. If `close()` already ran (normal teardown)
|
||||
/// this is a no-op; otherwise the listener died unexpectedly — tear it down so peers get
|
||||
/// fail-fast refusals instead of connecting into a never-accepted backlog.
|
||||
private func listenerLoopEnded() {
|
||||
guard listener != nil else { return }
|
||||
log?.error(
|
||||
"proxy accept loop ended unexpectedly; closing listener",
|
||||
metadata: [
|
||||
"vport": "\(port)",
|
||||
"uds": "\(path)",
|
||||
"action": "\(action)",
|
||||
])
|
||||
try? close()
|
||||
}
|
||||
|
||||
private func handleConn(
|
||||
conn: ContainerizationOS.Socket,
|
||||
connType: SocketType
|
||||
@@ -229,7 +257,18 @@ extension VsockProxy {
|
||||
// - both the client and server have half closed via:
|
||||
// - read hangup on epoll
|
||||
// - EOF on splice
|
||||
//
|
||||
// [Nucleic vendored patch] Hardened: (1) runs at most once — both fds' epoll
|
||||
// handlers can reach the cleanup condition, and a second entry after a failed
|
||||
// unregister would double-resume the continuation (a fatal trap in the guest's
|
||||
// PID-1 agent); (2) every step is attempted independently — a thrown unregister
|
||||
// used to SKIP the close(2)s, leaking both connection fds. Under control-plane
|
||||
// connection churn those leaks accumulated until vminitd hit EMFILE, its control-
|
||||
// socket accept loop died, and every session in the container stalled.
|
||||
nonisolated(unsafe) var cleanedUp = false
|
||||
let cleanup = { @Sendable [log, port, path, action] in
|
||||
guard !cleanedUp else { return }
|
||||
cleanedUp = true
|
||||
log?.debug(
|
||||
"cleaning up",
|
||||
metadata: [
|
||||
@@ -245,16 +284,34 @@ extension VsockProxy {
|
||||
|
||||
do {
|
||||
try ProcessSupervisor.default.unregisterFd(clientFile.fileDescriptor)
|
||||
} catch {
|
||||
self.log?.error("Failed to unregister vsock proxy client fd: \(error)")
|
||||
}
|
||||
do {
|
||||
try ProcessSupervisor.default.unregisterFd(serverFile.fileDescriptor)
|
||||
} catch {
|
||||
self.log?.error("Failed to unregister vsock proxy server fd: \(error)")
|
||||
}
|
||||
do {
|
||||
try conn.close()
|
||||
} catch {
|
||||
self.log?.error("Failed to close vsock proxy client: \(error)")
|
||||
}
|
||||
do {
|
||||
try relayTo.close()
|
||||
} catch {
|
||||
self.log?.error("Failed to clean up vsock proxy: \(error)")
|
||||
self.log?.error("Failed to close vsock proxy server: \(error)")
|
||||
}
|
||||
c.resume()
|
||||
}
|
||||
|
||||
try! ProcessSupervisor.default.registerFd(clientFile.fileDescriptor, mask: [.input, .output]) { mask in
|
||||
// [Nucleic vendored patch] These registrations were `try!` — an epoll_ctl failure
|
||||
// (fd pressure, a stale registration) crashed vminitd, the VM's PID-1 agent,
|
||||
// taking every session in the container down. Fail the one connection instead,
|
||||
// releasing whatever was already set up so nothing leaks (the caller closes `conn`;
|
||||
// `relayTo` and the first registration are released in the catch blocks below).
|
||||
do {
|
||||
try ProcessSupervisor.default.registerFd(clientFile.fileDescriptor, mask: [.input, .output]) { mask in
|
||||
if mask.readyToRead && !eofFromClient {
|
||||
let (fromEof, toEof) = Self.transferData(
|
||||
fromFile: &clientFile,
|
||||
@@ -305,8 +362,13 @@ extension VsockProxy {
|
||||
return cleanup()
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
try? relayTo.close()
|
||||
throw error
|
||||
}
|
||||
|
||||
try! ProcessSupervisor.default.registerFd(serverFile.fileDescriptor, mask: [.input, .output]) { mask in
|
||||
do {
|
||||
try ProcessSupervisor.default.registerFd(serverFile.fileDescriptor, mask: [.input, .output]) { mask in
|
||||
if mask.readyToRead && !eofFromServer {
|
||||
let (fromEof, toEof) = Self.transferData(
|
||||
fromFile: &serverFile,
|
||||
@@ -357,6 +419,11 @@ extension VsockProxy {
|
||||
return cleanup()
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
try? ProcessSupervisor.default.unregisterFd(clientFile.fileDescriptor)
|
||||
try? relayTo.close()
|
||||
throw error
|
||||
}
|
||||
} catch {
|
||||
c.resume(throwing: error)
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user