From 831d19c3a671b29c9a72f842658e0f79e39398e2 Mon Sep 17 00:00:00 2001 From: Andrew Blakeslee Moore Date: Mon, 13 Jul 2026 20:12:42 -0700 Subject: [PATCH] Per-exec cgroups follow-up: host-configured hard memory.max (no protobuf) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds an opt-in hard per-session memory ceiling on top of patch #9's scoped-OOM. The exec already ships the full OCI Spec, so the limit rides spec.linux.resources.memory.limit — no RPC/protobuf change: - host framework: LinuxProcessConfiguration.memoryLimitInBytes; LinuxContainer.exec stamps it onto the exec spec. - guest: Server+GRPC.createProcess reads it back and applies it as the exec cgroup's memory.max (new Cgroup2Manager.setMemoryMax) via createExec/ManagedProcess. - Nucleic: ContainerServiceSettings.controlPerSessionMemoryGiB (default 0 = off), applied only to the shared control container (ContainerManager.exec); wired through ContainerEngine.exec. So one session can't consume the whole shared container's memory before its own (oom.group-scoped) OOM. Default off preserves #9's behavior. Compile-verified host + musl guest; rides the pending -nucleic2 image, still runtime-pending. Co-Authored-By: Claude Opus 4.8 --- PATCHES.md | 20 +++++++++++++------ Sources/Containerization/LinuxContainer.swift | 9 +++++++++ .../LinuxProcessConfiguration.swift | 5 +++++ vminitd/Sources/Cgroup/Cgroup2Manager.swift | 6 ++++++ .../VminitdCore/ManagedContainer.swift | 4 +++- .../Sources/VminitdCore/ManagedProcess.swift | 7 +++++++ vminitd/Sources/VminitdCore/Server+GRPC.swift | 10 +++++++++- 7 files changed, 53 insertions(+), 8 deletions(-) diff --git a/PATCHES.md b/PATCHES.md index d00d757..b34a7f0 100644 --- a/PATCHES.md +++ b/PATCHES.md @@ -109,12 +109,20 @@ rebuild whenever a guest patch changes. Built locally, not in CI: the host frame `setPidsMax`/`remove`. **Best-effort with a graceful fallback**: if any step of the per-exec setup fails it wipes the partial state and reverts to the flat layout, and `ManagedProcess` falls back to the container cgroup per exec — so a cgroup hiccup degrades to today's behavior, never a failed - start. `ManagedContainer.execCgroupParent == nil` marks flat mode. NOTE: this delivers *scoped-OOM* - containment without host-configured limits; a hard per-exec `memory.max` (host-chosen, so a session - can't consume the whole box before its own OOM) still wants the exec-RPC resources field - (protobuf + `Vminitd.createProcess`/`ContainerEngine.exec` plumbing) — a follow-up. Marked - `[Nucleic vendored patch]` across `Cgroup2Manager.swift`, `ManagedContainer.swift`, - `ManagedProcess.swift`. **COMPILE-VERIFIED ONLY (musl cross-build); NOT yet runtime-validated** — a + start. `ManagedContainer.execCgroupParent == nil` marks flat mode. + + Beyond scoped-OOM, a **hard host-configured per-exec `memory.max`** is also wired — WITHOUT a + protobuf change, because the exec already ships the full OCI `Spec` and the guest just ignored + `linux.resources`. Host: `LinuxProcessConfiguration.memoryLimitInBytes` → `LinuxContainer.exec` + stamps it onto `spec.linux.resources.memory.limit`. Guest: `Server+GRPC.createProcess` reads that + back and passes it to `createExec`/`ManagedProcess`, which sets `memory.max` (new + `Cgroup2Manager.setMemoryMax`) on the exec's cgroup — so a session can't consume the whole box + before its own OOM. Driven by Nucleic's `ContainerServiceSettings.controlPerSessionMemoryGiB` + (default 0 = off; applied only to the shared control container, via `ContainerManager.exec`), so the + default stays scoped-OOM-only. Marked `[Nucleic vendored patch]` across `Cgroup2Manager.swift`, + `ManagedContainer.swift`, `ManagedProcess.swift`, `Server+GRPC.swift` (guest) and + `LinuxProcessConfiguration.swift`, `LinuxContainer.swift` (host). **COMPILE-VERIFIED (host + musl + cross-build); NOT yet runtime-validated** — a wrong cgroup-v2 hierarchy fails at runtime, so boot a container with the new image and confirm sessions start, `/sys/fs/cgroup/container//` exists per session, and a hog is contained, before pointing a shipping build at it. `vmexec/RunCommand` is unchanged: it still applies diff --git a/Sources/Containerization/LinuxContainer.swift b/Sources/Containerization/LinuxContainer.swift index 0b057f0..5c609b9 100644 --- a/Sources/Containerization/LinuxContainer.swift +++ b/Sources/Containerization/LinuxContainer.swift @@ -911,6 +911,11 @@ extension LinuxContainer { var config = LinuxProcessConfiguration() try configuration(&config) spec.process = config.toOCI() + // [Nucleic vendored patch] Per-exec memory ceiling → the exec's OCI resources, which the + // guest applies as memory.max on this exec's own cgroup (patch #9). + if let limit = config.memoryLimitInBytes { + spec.linux?.resources?.memory?.limit = Int64(limit) + } let stdio = IOUtil.setup( portAllocator: self.hostVsockPorts, @@ -948,6 +953,10 @@ extension LinuxContainer { var spec = self.generateRuntimeSpec() spec.process = configuration.toOCI() + // [Nucleic vendored patch] Per-exec memory ceiling → the exec's OCI resources (see above). + if let limit = configuration.memoryLimitInBytes { + spec.linux?.resources?.memory?.limit = Int64(limit) + } let stdio = IOUtil.setup( portAllocator: self.hostVsockPorts, diff --git a/Sources/Containerization/LinuxProcessConfiguration.swift b/Sources/Containerization/LinuxProcessConfiguration.swift index 1eef8b0..62da875 100644 --- a/Sources/Containerization/LinuxProcessConfiguration.swift +++ b/Sources/Containerization/LinuxProcessConfiguration.swift @@ -388,6 +388,11 @@ public struct LinuxProcessConfiguration: Sendable { public var stdout: Writer? /// The stderr for the process. public var stderr: Writer? + /// [Nucleic vendored patch] A hard per-exec memory ceiling in bytes. When set, `LinuxContainer.exec` + /// stamps it onto the exec's `spec.linux.resources.memory.limit`, which the guest applies as + /// `memory.max` on this exec's own cgroup (see patch #9) — so one session can't consume the whole + /// shared container's memory before *its own* OOM. nil → no per-exec ceiling (inherit the container). + public var memoryLimitInBytes: UInt64? public init() {} diff --git a/vminitd/Sources/Cgroup/Cgroup2Manager.swift b/vminitd/Sources/Cgroup/Cgroup2Manager.swift index 02151f6..c5c94df 100644 --- a/vminitd/Sources/Cgroup/Cgroup2Manager.swift +++ b/vminitd/Sources/Cgroup/Cgroup2Manager.swift @@ -288,6 +288,12 @@ public struct Cgroup2Manager: Sendable { try Self.writeValue(path: self.path, value: String(max), fileName: "pids.max") } + /// [Nucleic vendored patch] Hard memory ceiling (`memory.max`) — a host-configured per-exec cap so + /// one session can't consume the whole container's memory before its own (oom.group-scoped) OOM. + package func setMemoryMax(bytes: UInt64) throws { + try Self.writeValue(path: self.path, value: String(bytes), fileName: "memory.max") + } + /// [Nucleic vendored patch] Remove this cgroup directory (rmdir). The cgroup must already be empty /// of processes and child cgroups. Best-effort partial-setup cleanup for the per-exec layout. package func remove() throws { diff --git a/vminitd/Sources/VminitdCore/ManagedContainer.swift b/vminitd/Sources/VminitdCore/ManagedContainer.swift index 4d02c0f..8ace02f 100644 --- a/vminitd/Sources/VminitdCore/ManagedContainer.swift +++ b/vminitd/Sources/VminitdCore/ManagedContainer.swift @@ -201,7 +201,8 @@ extension ManagedContainer { func createExec( id: String, stdio: HostStdio, - process: ContainerizationOCI.Process + process: ContainerizationOCI.Process, + memoryLimitBytes: UInt64? = nil // [Nucleic vendored patch] hard per-exec memory.max ) throws { log.debug("creating exec process with \(process)") @@ -217,6 +218,7 @@ extension ManagedContainer { bundle: self.bundle, owningPid: self.initProcess.pid, execCgroupParent: self.execCgroupParent, // [Nucleic vendored patch] per-exec cgroup + execMemoryLimitBytes: memoryLimitBytes, // [Nucleic vendored patch] log: self.log ) self.execs[id] = process diff --git a/vminitd/Sources/VminitdCore/ManagedProcess.swift b/vminitd/Sources/VminitdCore/ManagedProcess.swift index e90560d..3e91308 100644 --- a/vminitd/Sources/VminitdCore/ManagedProcess.swift +++ b/vminitd/Sources/VminitdCore/ManagedProcess.swift @@ -60,6 +60,8 @@ final class ManagedProcess: ContainerProcess, Sendable { // [Nucleic vendored patch] Parent cgroup for this exec's OWN per-exec child (`/`); // nil means the legacy flat layout (join the container/init cgroup via `owningPid`). private let execCgroupParent: String? + // [Nucleic vendored patch] Hard per-exec memory.max (bytes) for this exec's cgroup; nil = none. + private let execMemoryLimitBytes: UInt64? private let ackPipe: Pipe private let syncPipe: Pipe private let errorPipe: Pipe @@ -78,6 +80,7 @@ final class ManagedProcess: ContainerProcess, Sendable { bundle: ContainerizationOCI.Bundle, owningPid: Int32? = nil, execCgroupParent: String? = nil, // [Nucleic vendored patch] + execMemoryLimitBytes: UInt64? = nil, // [Nucleic vendored patch] log: Logger ) throws { self.id = id @@ -86,6 +89,7 @@ final class ManagedProcess: ContainerProcess, Sendable { self.log = log self.owningPid = owningPid self.execCgroupParent = execCgroupParent + self.execMemoryLimitBytes = execMemoryLimitBytes let syncPipe = Pipe() try syncPipe.setCloexec() @@ -233,6 +237,9 @@ extension ManagedProcess { try? execCg.setOomGroup(true) try? execCg.setCpuWeight(100) try? execCg.setPidsMax(4096) + if let limit = execMemoryLimitBytes, limit > 0 { + try? execCg.setMemoryMax(bytes: limit) // host-configured hard per-exec ceiling + } try execCg.addProcess(pid: pid) } catch { log.error("per-exec cgroup for \(id) failed; joining the container cgroup: \(error)") diff --git a/vminitd/Sources/VminitdCore/Server+GRPC.swift b/vminitd/Sources/VminitdCore/Server+GRPC.swift index 6659199..67315ac 100644 --- a/vminitd/Sources/VminitdCore/Server+GRPC.swift +++ b/vminitd/Sources/VminitdCore/Server+GRPC.swift @@ -895,10 +895,18 @@ extension Initd: Com_Apple_Containerization_Sandbox_V3_SandboxContext.SimpleServ // This is an exec. if let container = await self.state.containers[request.containerID] { + // [Nucleic vendored patch] A per-exec memory ceiling rides the exec's OCI + // resources (set host-side by LinuxContainer.exec); apply it as this exec's + // own memory.max (patch #9). Only positive limits count. + let execMemoryLimit: UInt64? = { + guard let limit = ociSpec.linux?.resources?.memory?.limit, limit > 0 else { return nil } + return UInt64(limit) + }() try await container.createExec( id: request.id, stdio: stdioPorts, - process: process + process: process, + memoryLimitBytes: execMemoryLimit ) } else { // We need to make our new fangled container.