Files
containerization/vminitd/Sources/VminitdCore/VsockProxy.swift
T

498 lines
20 KiB
Swift
Raw Normal View History

//===----------------------------------------------------------------------===//
// Copyright © 2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
#if os(Linux)
import ContainerizationIO
import ContainerizationOS
import Foundation
import LCShim
import Logging
actor VsockProxy {
enum Action {
case listen
case dial
}
private enum SocketType {
case unix
case vsock
}
let id: String
private let path: URL
private let action: Action
private let port: UInt32
private let udsPerms: UInt32?
private let log: Logger?
private var listener: Socket?
private var task: Task<(), Never>?
private var connectionTasks: [UUID: Task<(), Never>] = [:]
init(
id: String,
action: Action,
port: UInt32,
path: URL,
udsPerms: UInt32?,
log: Logger? = nil
) {
self.id = id
self.action = action
self.port = port
self.path = path
self.udsPerms = udsPerms
self.log = log
}
}
extension VsockProxy {
func start() throws {
guard listener == nil else {
return
}
log?.debug(
"starting proxy",
metadata: [
"vport": "\(port)",
"uds": "\(path)",
"action": "\(action)",
])
switch action {
case .dial:
try dialHost()
case .listen:
try dialGuest()
}
}
func close() throws {
guard let listener else {
return
}
log?.debug(
"stopping proxy",
metadata: [
"vport": "\(port)",
"uds": "\(path)",
"action": "\(action)",
])
try listener.close()
for (_, t) in connectionTasks { t.cancel() }
connectionTasks.removeAll()
if action == .dial {
let fm = FileManager.default
if fm.fileExists(atPath: path.path) {
try fm.removeItem(at: path)
}
}
task?.cancel()
self.listener = nil
}
private func dialHost() throws {
let fm = FileManager.default
let parentDir = path.deletingLastPathComponent()
try fm.createDirectory(
at: parentDir,
withIntermediateDirectories: true
)
let type = try UnixType(
path: path.path,
perms: udsPerms,
unlinkExisting: true
)
let oldMask = umask(0)
defer { umask(oldMask) }
let uds = try Socket(type: type)
try uds.listen()
listener = uds
try acceptLoop(socketType: .unix)
}
private func dialGuest() throws {
let type = VsockType(
port: port,
cid: VsockType.anyCID
)
let vsock = try Socket(type: type)
try vsock.listen()
listener = vsock
try acceptLoop(socketType: .vsock)
}
private func acceptLoop(socketType: SocketType) throws {
guard let listener else {
return
}
let stream = try listener.acceptStream()
let task = Task {
do {
for try await conn in stream {
let connID = UUID()
let connTask = Task {
defer { self.connectionTasks[connID] = nil }
log?.debug(
"accepting connection",
metadata: [
"vport": "\(port)",
"uds": "\(path)",
"action": "\(action)",
"socketType": "\(socketType)",
])
do {
try await handleConn(
conn: conn,
connType: socketType
)
} catch {
self.log?.error("failed to handle connection: \(error)")
// [Nucleic vendored patch] A connection that failed before the relay
// owned it must be closed, or its fd leaks for the proxy's lifetime
// (accept vends closeOnDeinit: true, but the Socket is retained by the
// stream's yielded value until then — close deterministically).
try? conn.close()
}
}
// Safe: actor serialization ensures this runs before connTask can execute its defer.
connectionTasks[connID] = connTask
}
} catch {
self.log?.error("failed to accept connection: \(error)")
}
// [Nucleic vendored patch] If this loop ever ends while the proxy is still nominally
// running (fatal accept error), the listening socket MUST come down with it. Leaving it
// bound-but-unaccepted turned the relayed control socket into a silent black hole: every
// later client connect(2) SUCCEEDED into the kernel backlog and hung forever unanswered
// — for Nucleic, every session in the container stalling with "produced no output within
// 60s" until the VM was recreated. Closing the listener makes later connects fail fast
// (ECONNREFUSED/ENOENT), which callers surface and retry.
self.listenerLoopEnded()
}
self.task = task
}
/// [Nucleic vendored patch] The accept loop ended. If `close()` already ran (normal teardown)
/// this is a no-op; otherwise the listener died unexpectedly — tear it down so peers get
/// fail-fast refusals instead of connecting into a never-accepted backlog.
private func listenerLoopEnded() {
guard listener != nil else { return }
log?.error(
"proxy accept loop ended unexpectedly; closing listener",
metadata: [
"vport": "\(port)",
"uds": "\(path)",
"action": "\(action)",
])
try? close()
}
private func handleConn(
conn: ContainerizationOS.Socket,
connType: SocketType
) async throws {
try await withCheckedThrowingContinuation { (c: CheckedContinuation<Void, Error>) in
do {
// `relayTo` isn't used concurrently.
nonisolated(unsafe) var relayTo: ContainerizationOS.Socket
switch connType {
case .unix:
let type = VsockType(
port: port,
cid: VsockType.hostCID
)
relayTo = try Socket(
type: type,
closeOnDeinit: false
)
case .vsock:
let type = try UnixType(path: path.path)
relayTo = try Socket(
type: type,
closeOnDeinit: false
)
}
try relayTo.connect()
2026-07-17 23:48:59 -07:00
// [Nucleic vendored patch] BOTH fds must be non-blocking BEFORE either is
// registered. `Epoll.add` sets O_NONBLOCK only at registration time, and the first
// (client) registration's handler can fire — and splice toward the server fd —
// before the second (server) registration has made that fd non-blocking. A full
// destination then turned the splice into a genuinely BLOCKING call on vminitd's
// single ProcessSupervisor poller thread, freezing every exec's stdio and every
// control-plane relay in the container until the peer drained.
for fd in [conn.fileDescriptor, relayTo.fileDescriptor] {
let flags = fcntl(fd, F_GETFL)
if flags == -1 || fcntl(fd, F_SETFL, flags | O_NONBLOCK) == -1 {
self.log?.error(
"failed to set proxy fd non-blocking",
metadata: ["fd": "\(fd)", "errno": "\(errno)"])
}
}
// `clientFile` isn't used concurrently.
nonisolated(unsafe) var clientFile = OSFile.SpliceFile(fd: conn.fileDescriptor)
nonisolated(unsafe) var eofFromClient = false
// `serverFile` isn't used concurrently.
nonisolated(unsafe) var serverFile = OSFile.SpliceFile(fd: relayTo.fileDescriptor)
nonisolated(unsafe) var eofFromServer = false
// clean up when any of these conditions apply:
// - the client has completely hung up or errored
// - the server has completely hung up or errored
// - both the client and server have half closed via:
// - read hangup on epoll
// - EOF on splice
//
// [Nucleic vendored patch] Hardened: (1) runs at most once — both fds' epoll
// handlers can reach the cleanup condition, and a second entry after a failed
// unregister would double-resume the continuation (a fatal trap in the guest's
// PID-1 agent); (2) every step is attempted independently — a thrown unregister
// used to SKIP the close(2)s, leaking both connection fds. Under control-plane
// connection churn those leaks accumulated until vminitd hit EMFILE, its control-
// socket accept loop died, and every session in the container stalled.
nonisolated(unsafe) var cleanedUp = false
let cleanup = { @Sendable [log, port, path, action] in
guard !cleanedUp else { return }
cleanedUp = true
log?.debug(
"cleaning up",
metadata: [
"vport": "\(port)",
"uds": "\(path)",
"action": "\(action)",
"eofFromClient": "\(eofFromClient)",
"eofFromServer": "\(eofFromServer)",
"clientFd": "\(clientFile.fileDescriptor)",
"serverFd": "\(serverFile.fileDescriptor)",
]
)
do {
try ProcessSupervisor.default.unregisterFd(clientFile.fileDescriptor)
} catch {
self.log?.error("Failed to unregister vsock proxy client fd: \(error)")
}
do {
try ProcessSupervisor.default.unregisterFd(serverFile.fileDescriptor)
} catch {
self.log?.error("Failed to unregister vsock proxy server fd: \(error)")
}
do {
try conn.close()
} catch {
self.log?.error("Failed to close vsock proxy client: \(error)")
}
do {
try relayTo.close()
} catch {
self.log?.error("Failed to close vsock proxy server: \(error)")
}
c.resume()
}
// [Nucleic vendored patch] These registrations were `try!` — an epoll_ctl failure
// (fd pressure, a stale registration) crashed vminitd, the VM's PID-1 agent,
// taking every session in the container down. Fail the one connection instead,
// releasing whatever was already set up so nothing leaks (the caller closes `conn`;
// `relayTo` and the first registration are released in the catch blocks below).
do {
try ProcessSupervisor.default.registerFd(clientFile.fileDescriptor, mask: [.input, .output]) { mask in
if mask.readyToRead && !eofFromClient {
let (fromEof, toEof) = Self.transferData(
fromFile: &clientFile,
toFile: &serverFile,
description: "readyToRead:toServer",
log: self.log
)
eofFromClient = eofFromClient || fromEof
eofFromServer = eofFromServer || toEof
}
if mask.readyToWrite && !eofFromServer {
let (fromEof, toEof) = Self.transferData(
fromFile: &serverFile,
toFile: &clientFile,
description: "readyToWrite:toClient",
log: self.log
)
eofFromClient = eofFromClient || toEof
eofFromServer = eofFromServer || fromEof
}
if mask.isHangup {
eofFromClient = true
eofFromServer = true
} else if mask.isRemoteHangup && !eofFromClient {
// half close, shut down client to server transfer
// we should see no more EPOLLIN events on the client fd
// and no more EPOLLOUT events on the server fd
eofFromClient = true
if shutdown(serverFile.fileDescriptor, Int32(SHUT_WR)) != 0 {
self.log?.warning(
"failed to shut down client reads",
metadata: [
"vport": "\(self.port)",
"uds": "\(self.path)",
"errno": "\(errno)",
"eofFromClient": "\(eofFromClient)",
"eofFromServer": "\(eofFromServer)",
"clientFd": "\(clientFile.fileDescriptor)",
"serverFd": "\(serverFile.fileDescriptor)",
]
)
}
}
if eofFromClient && eofFromServer {
return cleanup()
}
}
} catch {
try? relayTo.close()
throw error
}
do {
try ProcessSupervisor.default.registerFd(serverFile.fileDescriptor, mask: [.input, .output]) { mask in
if mask.readyToRead && !eofFromServer {
let (fromEof, toEof) = Self.transferData(
fromFile: &serverFile,
toFile: &clientFile,
description: "readyToRead:toClient",
log: self.log
)
eofFromClient = eofFromClient || toEof
eofFromServer = eofFromServer || fromEof
}
if mask.readyToWrite && !eofFromClient {
let (fromEof, toEof) = Self.transferData(
fromFile: &clientFile,
toFile: &serverFile,
description: "readyToWrite:toServer",
log: self.log
)
eofFromClient = eofFromClient || fromEof
eofFromServer = eofFromServer || toEof
}
if mask.isHangup {
eofFromClient = true
eofFromServer = true
} else if mask.isRemoteHangup && !eofFromServer {
// half close, shut down server to client transfer
// we should see no more EPOLLIN events on the server fd
// and no more EPOLLOUT events on the client fd
eofFromServer = true
if shutdown(clientFile.fileDescriptor, Int32(SHUT_WR)) != 0 {
self.log?.warning(
"failed to shut down server reads",
metadata: [
"vport": "\(self.port)",
"uds": "\(self.path)",
"errno": "\(errno)",
"eofFromClient": "\(eofFromClient)",
"eofFromServer": "\(eofFromServer)",
"clientFd": "\(clientFile.fileDescriptor)",
"serverFd": "\(serverFile.fileDescriptor)",
]
)
}
}
if eofFromClient && eofFromServer {
return cleanup()
}
}
} catch {
try? ProcessSupervisor.default.unregisterFd(clientFile.fileDescriptor)
try? relayTo.close()
throw error
}
} catch {
c.resume(throwing: error)
}
}
}
private static func transferData(
fromFile: inout OSFile.SpliceFile,
toFile: inout OSFile.SpliceFile,
description: String,
log: Logger?
) -> (Bool, Bool) {
do {
let (readBytes, writeBytes, action) = try OSFile.splice(from: &fromFile, to: &toFile)
log?.trace(
"transferred data",
metadata: [
"description": "\(description)",
"action": "\(action)",
"readBytes": "\(readBytes)",
"writeBytes": "\(writeBytes)",
"fromFd": "\(fromFile.fileDescriptor)",
"toFd": "\(toFile.fileDescriptor)",
]
)
if action == .eof {
// half close, shut down client to server transfer
// we should see no more EPOLLIN events on the client fd
// and no more EPOLLOUT events on the server fd
if shutdown(toFile.fileDescriptor, Int32(SHUT_WR)) != 0 {
log?.warning(
"failed to shut down reads",
metadata: [
"description": "\(description)",
"errno": "\(errno)",
"action": "\(action)",
"readBytes": "\(readBytes)",
"writeBytes": "\(writeBytes)",
"fromFd": "\(fromFile.fileDescriptor)",
"toFd": "\(toFile.fileDescriptor)",
]
)
}
return (true, false)
} else if action == .brokenPipe {
return (true, true)
}
return (false, false)
} catch {
return (true, true)
}
}
}
#endif