Per-exec cgroups (guest patch #9): scope a session's OOM/CPU/fork-bomb to itself
Restructures the guest cgroup layout so each exec gets its OWN child cgroup (/container/<id>/<execID>) with memory.oom.group=1, a fair cpu.weight, and a pids.max backstop — so one control session can't OOM-kill, starve, or fork-bomb its siblings in the shared container. The container init moves to its own leaf so the container cgroup can delegate controllers to children (cgroup v2 no-internal-process rule). New Cgroup2Manager helpers: setOomGroup/setCpuWeight/ setPidsMax/remove. Best-effort with graceful fallback: any failure in the per-exec setup wipes the partial state and reverts to today's flat layout, and each exec falls back to the container cgroup — a cgroup hiccup degrades to current behavior, never a failed start. COMPILE-VERIFIED via the musl cross-build; NOT yet runtime-validated. Built as image tag -nucleic2; vminitReference stays on the validated -nucleic1 until -nucleic2 is checked in a real container. A hard host-configured per-exec memory.max (exec-RPC resources field) remains a follow-up. Co-Authored-By: Claude Opus 4.8 <[email protected]>
This commit is contained in:
@@ -32,6 +32,10 @@ public actor ManagedContainer {
|
||||
private let bundle: ContainerizationOCI.Bundle
|
||||
private let needsCgroupCleanup: Bool
|
||||
private var execs: [String: any ContainerProcess] = [:]
|
||||
// [Nucleic vendored patch] When per-exec cgroup isolation is active, the container cgroup that each
|
||||
// exec gets its own child under (`<parent>/<execID>`). nil = the legacy flat layout (init + all
|
||||
// execs share the container cgroup).
|
||||
private let execCgroupParent: String?
|
||||
|
||||
public var pid: Int32? {
|
||||
self.initProcess.pid
|
||||
@@ -44,11 +48,53 @@ public actor ManagedContainer {
|
||||
ociRuntimePath: String? = nil,
|
||||
log: Logger
|
||||
) async throws {
|
||||
var cgroupsPath: String
|
||||
if let cgPath = spec.linux?.cgroupsPath {
|
||||
cgroupsPath = cgPath
|
||||
// [Nucleic vendored patch] `spec` is mutated below to relocate the init into its own leaf cgroup
|
||||
// when per-exec isolation is set up.
|
||||
var spec = spec
|
||||
let containerCgroup: String = {
|
||||
if let p = spec.linux?.cgroupsPath, !p.isEmpty { return p }
|
||||
return "/container/\(id)"
|
||||
}()
|
||||
|
||||
let cgManager = Cgroup2Manager(
|
||||
group: URL(filePath: containerCgroup),
|
||||
logger: log
|
||||
)
|
||||
try cgManager.create()
|
||||
|
||||
// [Nucleic vendored patch] Per-exec cgroup isolation. Turn the container cgroup into an
|
||||
// intermediary — resource ceiling on it, controllers delegated to children — and run the
|
||||
// container init in its own leaf (`<container>/init`), so each exec later gets its OWN child
|
||||
// cgroup (see `ManagedProcess.start`): a runaway session's OOM/CPU/fork-bomb is then scoped to
|
||||
// that session and can't take down its siblings in the shared container. Best-effort: on ANY
|
||||
// failure, wipe the partial state and fall back to the upstream flat layout (init + all execs
|
||||
// share the container cgroup). `execCgroupParent == nil` marks flat mode.
|
||||
var execParent: String? = nil
|
||||
if spec.linux != nil {
|
||||
do {
|
||||
let initCg = Cgroup2Manager(group: URL(filePath: containerCgroup + "/init"), logger: log)
|
||||
try initCg.create()
|
||||
// Enabling controllers from the init leaf sets cgroup.subtree_control on every ancestor
|
||||
// (incl. the container cgroup), which is what lets sibling child cgroups get memory/cpu/pids.
|
||||
try initCg.toggleAllAvailableControllers(enable: true)
|
||||
if let resources = spec.linux?.resources {
|
||||
try cgManager.applyResources(resources: resources) // ceiling stays on the parent
|
||||
}
|
||||
spec.linux?.cgroupsPath = containerCgroup + "/init" // vmexec places the init here
|
||||
spec.linux?.resources = nil // don't re-apply the ceiling to the init leaf
|
||||
execParent = containerCgroup
|
||||
} catch {
|
||||
log.error("per-exec cgroup setup failed; using flat layout: \(error)")
|
||||
// Undo any partial per-exec state so the container cgroup can hold the init again.
|
||||
let initCg = Cgroup2Manager(group: URL(filePath: containerCgroup + "/init"), logger: log)
|
||||
try? initCg.toggleAllAvailableControllers(enable: false)
|
||||
try? initCg.remove()
|
||||
try cgManager.toggleAllAvailableControllers(enable: true)
|
||||
spec.linux?.cgroupsPath = containerCgroup
|
||||
execParent = nil
|
||||
}
|
||||
} else {
|
||||
cgroupsPath = "/container/\(id)"
|
||||
try cgManager.toggleAllAvailableControllers(enable: true)
|
||||
}
|
||||
|
||||
let bundle = try ContainerizationOCI.Bundle.create(
|
||||
@@ -57,15 +103,7 @@ public actor ManagedContainer {
|
||||
)
|
||||
log.debug("created bundle with spec \(spec)")
|
||||
|
||||
let cgManager = Cgroup2Manager(
|
||||
group: URL(filePath: cgroupsPath),
|
||||
logger: log
|
||||
)
|
||||
try cgManager.create()
|
||||
|
||||
do {
|
||||
try cgManager.toggleAllAvailableControllers(enable: true)
|
||||
|
||||
let initProcess: any ContainerProcess
|
||||
|
||||
if let runtimePath = ociRuntimePath {
|
||||
@@ -99,6 +137,7 @@ public actor ManagedContainer {
|
||||
}
|
||||
|
||||
self.cgroupManager = cgManager
|
||||
self.execCgroupParent = execParent
|
||||
self.initProcess = initProcess
|
||||
self.id = id
|
||||
self.bundle = bundle
|
||||
@@ -177,6 +216,7 @@ extension ManagedContainer {
|
||||
stdio: stdio,
|
||||
bundle: self.bundle,
|
||||
owningPid: self.initProcess.pid,
|
||||
execCgroupParent: self.execCgroupParent, // [Nucleic vendored patch] per-exec cgroup
|
||||
log: self.log
|
||||
)
|
||||
self.execs[id] = process
|
||||
|
||||
@@ -57,6 +57,9 @@ final class ManagedProcess: ContainerProcess, Sendable {
|
||||
private let command: Command
|
||||
private let state: Mutex<State>
|
||||
private let owningPid: Int32?
|
||||
// [Nucleic vendored patch] Parent cgroup for this exec's OWN per-exec child (`<parent>/<id>`);
|
||||
// nil means the legacy flat layout (join the container/init cgroup via `owningPid`).
|
||||
private let execCgroupParent: String?
|
||||
private let ackPipe: Pipe
|
||||
private let syncPipe: Pipe
|
||||
private let errorPipe: Pipe
|
||||
@@ -74,6 +77,7 @@ final class ManagedProcess: ContainerProcess, Sendable {
|
||||
stdio: HostStdio,
|
||||
bundle: ContainerizationOCI.Bundle,
|
||||
owningPid: Int32? = nil,
|
||||
execCgroupParent: String? = nil, // [Nucleic vendored patch]
|
||||
log: Logger
|
||||
) throws {
|
||||
self.id = id
|
||||
@@ -81,6 +85,7 @@ final class ManagedProcess: ContainerProcess, Sendable {
|
||||
log[metadataKey: "id"] = "\(id)"
|
||||
self.log = log
|
||||
self.owningPid = owningPid
|
||||
self.execCgroupParent = execCgroupParent
|
||||
|
||||
let syncPipe = Pipe()
|
||||
try syncPipe.setCloexec()
|
||||
@@ -215,7 +220,27 @@ extension ManagedProcess {
|
||||
|
||||
// This should probably happen in vmexec, but we don't need to set any cgroup
|
||||
// toggles so the problem is much simpler to just do it here.
|
||||
if let owningPid {
|
||||
if let parent = execCgroupParent {
|
||||
// [Nucleic vendored patch] Per-exec cgroup: place this exec in its OWN child cgroup
|
||||
// (`<parent>/<id>`) with memory.oom.group (a runaway session's OOM kills only its
|
||||
// tree), a fair cpu.weight, and a pids.max fork-bomb backstop — so one session can't
|
||||
// OOM-kill / starve / fork-bomb its siblings in the shared container. Best-effort:
|
||||
// on ANY error fall back to the container/init cgroup so a cgroup hiccup never blocks
|
||||
// the exec from starting.
|
||||
do {
|
||||
let execCg = Cgroup2Manager(group: URL(filePath: parent + "/" + id), logger: log)
|
||||
try execCg.create()
|
||||
try? execCg.setOomGroup(true)
|
||||
try? execCg.setCpuWeight(100)
|
||||
try? execCg.setPidsMax(4096)
|
||||
try execCg.addProcess(pid: pid)
|
||||
} catch {
|
||||
log.error("per-exec cgroup for \(id) failed; joining the container cgroup: \(error)")
|
||||
if let owningPid {
|
||||
try? Cgroup2Manager.loadFromPid(pid: owningPid).addProcess(pid: pid)
|
||||
}
|
||||
}
|
||||
} else if let owningPid {
|
||||
let cgManager = try Cgroup2Manager.loadFromPid(pid: owningPid)
|
||||
try cgManager.addProcess(pid: pid)
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user