Fix the recurring session-stall pair: MainActor lock-reconcile hang + vminitd relay spin

Host (dominant): AppStore.reconcileLocks polls every 3s on the MainActor
while any lock is held — effectively forever, since interrupted/errored
sessions deliberately retain locks. Each pass ran heldPathDisposition's
diverges() as three held×unmerged scans with two split-allocations per
pathsOverlap call, pinning the main thread for tens of seconds per pass
on a diverged trunk (hang-reports 2026-07-28: 100% of samples in
reconcileLocks→heldPathDisposition→pathsOverlap). That froze running
sessions' transcripts and starved the spawn path into the 60s
"produced no output" watchdog. pathsOverlap is now allocation-free
bytewise comparison with identical semantics, and divergentHeldPaths
answers all three questions from one O((held+unmerged)·depth) set.

Guest (persistence): VsockProxy threaded ONE offset pair through BOTH
relay directions; once the EAGAIN-return backpressure patch let pending
bytes persist, traffic in the other direction skewed the shared
counters, made the write leg unreachable, and spun the single
ProcessSupervisor poller thread forever — container-wide dead control
plane until VM recreation, triggered by exactly the backpressure the
host hang created. Each direction now owns its own pipe and counters
(OSFile.RelayDirection), and source EOF is only surfaced after the
pipe drains so SHUT_WR can't truncate a parked tail. Vendored patch
docs updated (#15); inert until the initfs image is rebuilt+repointed.

Co-Authored-By: Claude Fable 5 <[email protected]>
This commit is contained in:
2026-07-28 15:05:37 -07:00
co-authored by Claude Fable 5
parent 5d00d7b39c
commit 6f950d859b
3 changed files with 238 additions and 207 deletions
+127 -129
View File
@@ -260,12 +260,19 @@ extension VsockProxy {
}
}
// `clientFile` isn't used concurrently.
nonisolated(unsafe) var clientFile = OSFile.SpliceFile(fd: conn.fileDescriptor)
nonisolated(unsafe) var eofFromClient = false
// `serverFile` isn't used concurrently.
nonisolated(unsafe) var serverFile = OSFile.SpliceFile(fd: relayTo.fileDescriptor)
nonisolated(unsafe) var eofFromServer = false
// [Nucleic vendored patch] Each relay direction owns its own pipe and byte
// counters (see RelayDirection) — the previous shared-offset SpliceFile pair
// let one parked direction corrupt the other's accounting and spin the poller
// thread. Neither is used concurrently (all access is on the single
// ProcessSupervisor poller thread).
nonisolated(unsafe) var toServer = OSFile.RelayDirection(
from: conn.fileDescriptor, to: relayTo.fileDescriptor)
nonisolated(unsafe) var toClient = OSFile.RelayDirection(
from: relayTo.fileDescriptor, to: conn.fileDescriptor)
// A direction is DONE once its source EOF fully flushed (SHUT_WR sent), its
// destination broke, or a full hangup ended the connection.
nonisolated(unsafe) var toServerDone = false
nonisolated(unsafe) var toClientDone = false
// clean up when any of these conditions apply:
// - the client has completely hung up or errored
@@ -291,20 +298,20 @@ extension VsockProxy {
"vport": "\(port)",
"uds": "\(path)",
"action": "\(action)",
"eofFromClient": "\(eofFromClient)",
"eofFromServer": "\(eofFromServer)",
"clientFd": "\(clientFile.fileDescriptor)",
"serverFd": "\(serverFile.fileDescriptor)",
"toServerDone": "\(toServerDone)",
"toClientDone": "\(toClientDone)",
"clientFd": "\(conn.fileDescriptor)",
"serverFd": "\(relayTo.fileDescriptor)",
]
)
do {
try ProcessSupervisor.default.unregisterFd(clientFile.fileDescriptor)
try ProcessSupervisor.default.unregisterFd(conn.fileDescriptor)
} catch {
self.log?.error("Failed to unregister vsock proxy client fd: \(error)")
}
do {
try ProcessSupervisor.default.unregisterFd(serverFile.fileDescriptor)
try ProcessSupervisor.default.unregisterFd(relayTo.fileDescriptor)
} catch {
self.log?.error("Failed to unregister vsock proxy server fd: \(error)")
}
@@ -326,55 +333,52 @@ extension VsockProxy {
// taking every session in the container down. Fail the one connection instead,
// releasing whatever was already set up so nothing leaks (the caller closes `conn`;
// `relayTo` and the first registration are released in the catch blocks below).
// [Nucleic vendored patch] Interpret one relay-step outcome: returns whether
// the stepped direction is now done; a broken destination ends BOTH directions
// (the peer is gone). Returns rather than writing the stepped flag itself so no
// captured var is ever aliased by an inout parameter (exclusivity).
let apply = { @Sendable (outcome: RelayStepOutcome) -> Bool in
switch outcome {
case .open: return false
case .finished: return true
case .broken:
toServerDone = true
toClientDone = true
return true
}
}
do {
try ProcessSupervisor.default.registerFd(clientFile.fileDescriptor, mask: [.input, .output]) { mask in
if mask.readyToRead && !eofFromClient {
let (fromEof, toEof) = Self.transferData(
fromFile: &clientFile,
toFile: &serverFile,
description: "readyToRead:toServer",
log: self.log
)
eofFromClient = eofFromClient || fromEof
eofFromServer = eofFromServer || toEof
}
if mask.readyToWrite && !eofFromServer {
let (fromEof, toEof) = Self.transferData(
fromFile: &serverFile,
toFile: &clientFile,
description: "readyToWrite:toClient",
log: self.log
)
eofFromClient = eofFromClient || toEof
eofFromServer = eofFromServer || fromEof
}
if mask.isHangup {
eofFromClient = true
eofFromServer = true
} else if mask.isRemoteHangup && !eofFromClient {
// half close, shut down client to server transfer
// we should see no more EPOLLIN events on the client fd
// and no more EPOLLOUT events on the server fd
eofFromClient = true
if shutdown(serverFile.fileDescriptor, Int32(SHUT_WR)) != 0 {
self.log?.warning(
"failed to shut down client reads",
metadata: [
"vport": "\(self.port)",
"uds": "\(self.path)",
"errno": "\(errno)",
"eofFromClient": "\(eofFromClient)",
"eofFromServer": "\(eofFromServer)",
"clientFd": "\(clientFile.fileDescriptor)",
"serverFd": "\(serverFile.fileDescriptor)",
]
)
try ProcessSupervisor.default.registerFd(conn.fileDescriptor, mask: [.input, .output]) { mask in
if mask.readyToRead && !toServerDone {
if apply(Self.relayStep(&toServer, description: "client:readyToRead:toServer", log: self.log)) {
toServerDone = true
}
}
if eofFromClient && eofFromServer {
// The client drained: flush toClient's parked bytes (and whatever more the
// server has ready).
if mask.readyToWrite && !toClientDone {
if apply(Self.relayStep(&toClient, description: "client:readyToWrite:toClient", log: self.log)) {
toClientDone = true
}
}
if mask.isHangup {
toServerDone = true
toClientDone = true
} else if mask.isRemoteHangup && !toServerDone {
// Half close: the client sends no more. Drain the tail — relayStep
// observes the real EOF after the last buffered bytes and only then
// SHUT_WRs the server, so a parked backlog is never dropped. If the
// server is full right now the direction stays open and its EPOLLOUT
// edge finishes the flush.
if apply(Self.relayStep(&toServer, description: "client:remoteHangup:toServer", log: self.log)) {
toServerDone = true
}
}
if toServerDone && toClientDone {
return cleanup()
}
}
@@ -384,59 +388,38 @@ extension VsockProxy {
}
do {
try ProcessSupervisor.default.registerFd(serverFile.fileDescriptor, mask: [.input, .output]) { mask in
if mask.readyToRead && !eofFromServer {
let (fromEof, toEof) = Self.transferData(
fromFile: &serverFile,
toFile: &clientFile,
description: "readyToRead:toClient",
log: self.log
)
eofFromClient = eofFromClient || toEof
eofFromServer = eofFromServer || fromEof
}
if mask.readyToWrite && !eofFromClient {
let (fromEof, toEof) = Self.transferData(
fromFile: &clientFile,
toFile: &serverFile,
description: "readyToWrite:toServer",
log: self.log
)
eofFromClient = eofFromClient || fromEof
eofFromServer = eofFromServer || toEof
}
if mask.isHangup {
eofFromClient = true
eofFromServer = true
} else if mask.isRemoteHangup && !eofFromServer {
// half close, shut down server to client transfer
// we should see no more EPOLLIN events on the server fd
// and no more EPOLLOUT events on the client fd
eofFromServer = true
if shutdown(clientFile.fileDescriptor, Int32(SHUT_WR)) != 0 {
self.log?.warning(
"failed to shut down server reads",
metadata: [
"vport": "\(self.port)",
"uds": "\(self.path)",
"errno": "\(errno)",
"eofFromClient": "\(eofFromClient)",
"eofFromServer": "\(eofFromServer)",
"clientFd": "\(clientFile.fileDescriptor)",
"serverFd": "\(serverFile.fileDescriptor)",
]
)
try ProcessSupervisor.default.registerFd(relayTo.fileDescriptor, mask: [.input, .output]) { mask in
if mask.readyToRead && !toClientDone {
if apply(Self.relayStep(&toClient, description: "server:readyToRead:toClient", log: self.log)) {
toClientDone = true
}
}
if eofFromClient && eofFromServer {
// The server drained: flush toServer's parked bytes (and whatever more the
// client has ready).
if mask.readyToWrite && !toServerDone {
if apply(Self.relayStep(&toServer, description: "server:readyToWrite:toServer", log: self.log)) {
toServerDone = true
}
}
if mask.isHangup {
toServerDone = true
toClientDone = true
} else if mask.isRemoteHangup && !toClientDone {
// Half close: the server sends no more — drain the tail toward the
// client (see the client handler's mirror-image comment).
if apply(Self.relayStep(&toClient, description: "server:remoteHangup:toClient", log: self.log)) {
toClientDone = true
}
}
if toServerDone && toClientDone {
return cleanup()
}
}
} catch {
try? ProcessSupervisor.default.unregisterFd(clientFile.fileDescriptor)
try? ProcessSupervisor.default.unregisterFd(conn.fileDescriptor)
try? relayTo.close()
throw error
}
@@ -446,50 +429,65 @@ extension VsockProxy {
}
}
private static func transferData(
fromFile: inout OSFile.SpliceFile,
toFile: inout OSFile.SpliceFile,
/// [Nucleic vendored patch] Outcome of one non-blocking relay pass over a direction.
enum RelayStepOutcome {
/// More may come (source dry, or destination full with bytes parked in the pipe).
case open
/// Source EOF fully flushed; the destination has been SHUT_WR'd.
case finished
/// The destination hung up (or the splice failed) — the connection is over.
case broken
}
/// [Nucleic vendored patch] Run one non-blocking relay pass over `direction`. `.eof` is
/// only reported by `OSFile.relay` once the transfer pipe has fully drained, so the
/// SHUT_WR here can never truncate a parked tail.
private static func relayStep(
_ direction: inout OSFile.RelayDirection,
description: String,
log: Logger?
) -> (Bool, Bool) {
) -> RelayStepOutcome {
do {
let (readBytes, writeBytes, action) = try OSFile.splice(from: &fromFile, to: &toFile)
let result = try OSFile.relay(&direction)
log?.trace(
"transferred data",
metadata: [
"description": "\(description)",
"action": "\(action)",
"readBytes": "\(readBytes)",
"writeBytes": "\(writeBytes)",
"fromFd": "\(fromFile.fileDescriptor)",
"toFd": "\(toFile.fileDescriptor)",
"result": "\(result)",
"pendingBytes": "\(direction.pendingBytes)",
"fromFd": "\(direction.from)",
"toFd": "\(direction.to)",
]
)
if action == .eof {
// half close, shut down client to server transfer
// we should see no more EPOLLIN events on the client fd
// and no more EPOLLOUT events on the server fd
if shutdown(toFile.fileDescriptor, Int32(SHUT_WR)) != 0 {
switch result {
case .idle:
return .open
case .eof:
if shutdown(direction.to, Int32(SHUT_WR)) != 0 {
log?.warning(
"failed to shut down reads",
"failed to shut down destination writes",
metadata: [
"description": "\(description)",
"errno": "\(errno)",
"action": "\(action)",
"readBytes": "\(readBytes)",
"writeBytes": "\(writeBytes)",
"fromFd": "\(fromFile.fileDescriptor)",
"toFd": "\(toFile.fileDescriptor)",
"fromFd": "\(direction.from)",
"toFd": "\(direction.to)",
]
)
}
return (true, false)
} else if action == .brokenPipe {
return (true, true)
return .finished
case .brokenPipe:
return .broken
}
return (false, false)
} catch {
return (true, true)
log?.error(
"relay failed: \(error)",
metadata: [
"description": "\(description)",
"fromFd": "\(direction.from)",
"toFd": "\(direction.to)",
]
)
return .broken
}
}
}