Adds an opt-in hard per-session memory ceiling on top of patch #9's scoped-OOM. The exec already ships the full OCI Spec, so the limit rides spec.linux.resources.memory.limit — no RPC/protobuf change: - host framework: LinuxProcessConfiguration.memoryLimitInBytes; LinuxContainer.exec stamps it onto the exec spec. - guest: Server+GRPC.createProcess reads it back and applies it as the exec cgroup's memory.max (new Cgroup2Manager.setMemoryMax) via createExec/ManagedProcess. - Nucleic: ContainerServiceSettings.controlPerSessionMemoryGiB (default 0 = off), applied only to the shared control container (ContainerManager.exec); wired through ContainerEngine.exec. So one session can't consume the whole shared container's memory before its own (oom.group-scoped) OOM. Default off preserves #9's behavior. Compile-verified host + musl guest; rides the pending -nucleic2 image, still runtime-pending. Co-Authored-By: Claude Opus 4.8 <[email protected]>
329 lines
12 KiB
Swift
329 lines
12 KiB
Swift
//===----------------------------------------------------------------------===//
|
|
// Copyright © 2026 Apple Inc. and the Containerization project authors.
|
|
//
|
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
// you may not use this file except in compliance with the License.
|
|
// You may obtain a copy of the License at
|
|
//
|
|
// https://www.apache.org/licenses/LICENSE-2.0
|
|
//
|
|
// Unless required by applicable law or agreed to in writing, software
|
|
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
// See the License for the specific language governing permissions and
|
|
// limitations under the License.
|
|
//===----------------------------------------------------------------------===//
|
|
|
|
#if os(Linux)
|
|
|
|
import Cgroup
|
|
import ContainerizationError
|
|
import ContainerizationOCI
|
|
import ContainerizationOS
|
|
import Foundation
|
|
import Logging
|
|
|
|
public actor ManagedContainer {
|
|
public let id: String
|
|
let initProcess: any ContainerProcess
|
|
|
|
private let cgroupManager: Cgroup2Manager
|
|
private let log: Logger
|
|
private let bundle: ContainerizationOCI.Bundle
|
|
private let needsCgroupCleanup: Bool
|
|
private var execs: [String: any ContainerProcess] = [:]
|
|
// [Nucleic vendored patch] When per-exec cgroup isolation is active, the container cgroup that each
|
|
// exec gets its own child under (`<parent>/<execID>`). nil = the legacy flat layout (init + all
|
|
// execs share the container cgroup).
|
|
private let execCgroupParent: String?
|
|
|
|
public var pid: Int32? {
|
|
self.initProcess.pid
|
|
}
|
|
|
|
init(
|
|
id: String,
|
|
stdio: HostStdio,
|
|
spec: ContainerizationOCI.Spec,
|
|
ociRuntimePath: String? = nil,
|
|
log: Logger
|
|
) async throws {
|
|
// [Nucleic vendored patch] `spec` is mutated below to relocate the init into its own leaf cgroup
|
|
// when per-exec isolation is set up.
|
|
var spec = spec
|
|
let containerCgroup: String = {
|
|
if let p = spec.linux?.cgroupsPath, !p.isEmpty { return p }
|
|
return "/container/\(id)"
|
|
}()
|
|
|
|
let cgManager = Cgroup2Manager(
|
|
group: URL(filePath: containerCgroup),
|
|
logger: log
|
|
)
|
|
try cgManager.create()
|
|
|
|
// [Nucleic vendored patch] Per-exec cgroup isolation. Turn the container cgroup into an
|
|
// intermediary — resource ceiling on it, controllers delegated to children — and run the
|
|
// container init in its own leaf (`<container>/init`), so each exec later gets its OWN child
|
|
// cgroup (see `ManagedProcess.start`): a runaway session's OOM/CPU/fork-bomb is then scoped to
|
|
// that session and can't take down its siblings in the shared container. Best-effort: on ANY
|
|
// failure, wipe the partial state and fall back to the upstream flat layout (init + all execs
|
|
// share the container cgroup). `execCgroupParent == nil` marks flat mode.
|
|
var execParent: String? = nil
|
|
if spec.linux != nil {
|
|
do {
|
|
let initCg = Cgroup2Manager(group: URL(filePath: containerCgroup + "/init"), logger: log)
|
|
try initCg.create()
|
|
// Enabling controllers from the init leaf sets cgroup.subtree_control on every ancestor
|
|
// (incl. the container cgroup), which is what lets sibling child cgroups get memory/cpu/pids.
|
|
try initCg.toggleAllAvailableControllers(enable: true)
|
|
if let resources = spec.linux?.resources {
|
|
try cgManager.applyResources(resources: resources) // ceiling stays on the parent
|
|
}
|
|
spec.linux?.cgroupsPath = containerCgroup + "/init" // vmexec places the init here
|
|
spec.linux?.resources = nil // don't re-apply the ceiling to the init leaf
|
|
execParent = containerCgroup
|
|
} catch {
|
|
log.error("per-exec cgroup setup failed; using flat layout: \(error)")
|
|
// Undo any partial per-exec state so the container cgroup can hold the init again.
|
|
let initCg = Cgroup2Manager(group: URL(filePath: containerCgroup + "/init"), logger: log)
|
|
try? initCg.toggleAllAvailableControllers(enable: false)
|
|
try? initCg.remove()
|
|
try cgManager.toggleAllAvailableControllers(enable: true)
|
|
spec.linux?.cgroupsPath = containerCgroup
|
|
execParent = nil
|
|
}
|
|
} else {
|
|
try cgManager.toggleAllAvailableControllers(enable: true)
|
|
}
|
|
|
|
let bundle = try ContainerizationOCI.Bundle.create(
|
|
path: Self.craftBundlePath(id: id),
|
|
spec: spec
|
|
)
|
|
log.debug("created bundle with spec \(spec)")
|
|
|
|
do {
|
|
let initProcess: any ContainerProcess
|
|
|
|
if let runtimePath = ociRuntimePath {
|
|
// Use runc runtime
|
|
let runc = ProcessSupervisor.default.getRuncWithReaper(
|
|
Runc(
|
|
command: runtimePath,
|
|
root: "/run/runc"
|
|
)
|
|
)
|
|
initProcess = try RuncProcess(
|
|
id: id,
|
|
stdio: stdio,
|
|
bundle: bundle,
|
|
runc: runc,
|
|
log: log
|
|
)
|
|
self.needsCgroupCleanup = false
|
|
log.info("created runc init process with runtime: \(runtimePath)")
|
|
} else {
|
|
// Use vmexec runtime
|
|
initProcess = try ManagedProcess(
|
|
id: id,
|
|
stdio: stdio,
|
|
bundle: bundle,
|
|
owningPid: nil,
|
|
log: log
|
|
)
|
|
self.needsCgroupCleanup = true
|
|
log.info("created vmexec init process")
|
|
}
|
|
|
|
self.cgroupManager = cgManager
|
|
self.execCgroupParent = execParent
|
|
self.initProcess = initProcess
|
|
self.id = id
|
|
self.bundle = bundle
|
|
self.log = log
|
|
} catch {
|
|
try? cgManager.delete()
|
|
throw error
|
|
}
|
|
}
|
|
}
|
|
|
|
extension ManagedContainer {
|
|
// removeCgroupWithRetry will remove a cgroup path handling EAGAIN and EBUSY errors and
|
|
// retrying the remove after an exponential timeout
|
|
private func removeCgroupWithRetry() async throws {
|
|
var delay = 10 // 10ms
|
|
let maxRetries = 5
|
|
|
|
for i in 0..<maxRetries {
|
|
if i != 0 {
|
|
try await Task.sleep(for: .milliseconds(delay))
|
|
delay *= 2
|
|
}
|
|
|
|
do {
|
|
try self.cgroupManager.delete(force: true)
|
|
return
|
|
} catch let error as Cgroup2Manager.Error {
|
|
guard case .errno(let errnoValue, let message) = error,
|
|
errnoValue == EBUSY || errnoValue == EAGAIN
|
|
else {
|
|
throw error
|
|
}
|
|
self.log.warning(
|
|
"cgroup deletion failed with EBUSY/EAGAIN, retrying",
|
|
metadata: [
|
|
"attempt": "\(i + 1)",
|
|
"delay": "\(delay)",
|
|
"errno": "\(errnoValue)",
|
|
"context": "\(message)",
|
|
])
|
|
continue
|
|
}
|
|
}
|
|
|
|
throw ContainerizationError(
|
|
.internalError,
|
|
message: "cgroups: unable to remove cgroup after \(maxRetries) retries"
|
|
)
|
|
}
|
|
|
|
private func ensureExecExists(_ id: String) throws {
|
|
if self.execs[id] == nil {
|
|
throw ContainerizationError(
|
|
.invalidState,
|
|
message: "exec \(id) does not exist in container \(self.id)"
|
|
)
|
|
}
|
|
}
|
|
|
|
func createExec(
|
|
id: String,
|
|
stdio: HostStdio,
|
|
process: ContainerizationOCI.Process,
|
|
memoryLimitBytes: UInt64? = nil // [Nucleic vendored patch] hard per-exec memory.max
|
|
) throws {
|
|
log.debug("creating exec process with \(process)")
|
|
|
|
// Write the process config to the bundle, and pass this on
|
|
// over to ManagedProcess to deal with.
|
|
try self.bundle.createExecSpec(
|
|
id: id,
|
|
process: process
|
|
)
|
|
let process = try ManagedProcess(
|
|
id: id,
|
|
stdio: stdio,
|
|
bundle: self.bundle,
|
|
owningPid: self.initProcess.pid,
|
|
execCgroupParent: self.execCgroupParent, // [Nucleic vendored patch] per-exec cgroup
|
|
execMemoryLimitBytes: memoryLimitBytes, // [Nucleic vendored patch]
|
|
log: self.log
|
|
)
|
|
self.execs[id] = process
|
|
}
|
|
|
|
func start(execID: String) async throws -> Int32 {
|
|
let proc = try self.getExecOrInit(execID: execID)
|
|
return try await ProcessSupervisor.default.start(process: proc)
|
|
}
|
|
|
|
func wait(execID: String) async throws -> ContainerExitStatus {
|
|
let proc = try self.getExecOrInit(execID: execID)
|
|
return await proc.wait()
|
|
}
|
|
|
|
func kill(execID: String, _ signal: Int32) async throws {
|
|
let proc = try self.getExecOrInit(execID: execID)
|
|
try await proc.kill(signal)
|
|
}
|
|
|
|
func resize(execID: String, size: Terminal.Size) throws {
|
|
let proc = try self.getExecOrInit(execID: execID)
|
|
try proc.resize(size: size)
|
|
}
|
|
|
|
func closeStdin(execID: String) throws {
|
|
let proc = try self.getExecOrInit(execID: execID)
|
|
try proc.closeStdin()
|
|
}
|
|
|
|
func deleteExec(id: String) throws {
|
|
try ensureExecExists(id)
|
|
do {
|
|
try self.bundle.deleteExecSpec(id: id)
|
|
} catch {
|
|
self.log.error("failed to remove exec spec from filesystem: \(error)")
|
|
}
|
|
self.execs.removeValue(forKey: id)
|
|
}
|
|
|
|
func delete() async throws {
|
|
// Delete the init process if it's a RuncProcess
|
|
try await self.initProcess.delete()
|
|
|
|
// Delete the bundle and cgroup
|
|
try self.bundle.delete()
|
|
if self.needsCgroupCleanup {
|
|
try await self.removeCgroupWithRetry()
|
|
}
|
|
}
|
|
|
|
func stats(_ categories: Cgroup2StatsCategory = .all) throws -> Cgroup2Stats {
|
|
try self.cgroupManager.stats(categories)
|
|
}
|
|
|
|
func getMemoryEvents() throws -> MemoryEvents {
|
|
try self.cgroupManager.getMemoryEvents()
|
|
}
|
|
|
|
func getExecOrInit(execID: String) throws -> any ContainerProcess {
|
|
if execID == self.id {
|
|
return self.initProcess
|
|
}
|
|
guard let proc = self.execs[execID] else {
|
|
throw ContainerizationError(
|
|
.invalidState,
|
|
message: "exec \(execID) does not exist in container \(self.id)"
|
|
)
|
|
}
|
|
return proc
|
|
}
|
|
}
|
|
|
|
extension ContainerizationOCI.Bundle {
|
|
func createExecSpec(id: String, process: ContainerizationOCI.Process) throws {
|
|
let specDir = self.path.appending(path: "execs/\(id)")
|
|
|
|
let fm = FileManager.default
|
|
try fm.createDirectory(
|
|
atPath: specDir.path,
|
|
withIntermediateDirectories: true
|
|
)
|
|
|
|
let specData = try JSONEncoder().encode(process)
|
|
let processConfigPath = specDir.appending(path: "process.json")
|
|
try specData.write(to: processConfigPath)
|
|
}
|
|
|
|
func getExecSpecPath(id: String) -> URL {
|
|
self.path.appending(path: "execs/\(id)/process.json")
|
|
}
|
|
|
|
func deleteExecSpec(id: String) throws {
|
|
let specDir = self.path.appending(path: "execs/\(id)")
|
|
|
|
let fm = FileManager.default
|
|
try fm.removeItem(at: specDir)
|
|
}
|
|
}
|
|
|
|
extension ManagedContainer {
|
|
static func craftBundlePath(id: String) -> URL {
|
|
URL(fileURLWithPath: "/run/container").appending(path: id)
|
|
}
|
|
}
|
|
|
|
#endif
|