Files
containerization/vminitd/Sources/VminitdCore/ManagedContainer.swift
T

329 lines
12 KiB
Swift
Raw Normal View History

//===----------------------------------------------------------------------===//
// Copyright © 2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
#if os(Linux)
import Cgroup
import ContainerizationError
import ContainerizationOCI
import ContainerizationOS
import Foundation
import Logging
public actor ManagedContainer {
public let id: String
let initProcess: any ContainerProcess
private let cgroupManager: Cgroup2Manager
private let log: Logger
private let bundle: ContainerizationOCI.Bundle
private let needsCgroupCleanup: Bool
private var execs: [String: any ContainerProcess] = [:]
// [Nucleic vendored patch] When per-exec cgroup isolation is active, the container cgroup that each
// exec gets its own child under (`<parent>/<execID>`). nil = the legacy flat layout (init + all
// execs share the container cgroup).
private let execCgroupParent: String?
public var pid: Int32? {
self.initProcess.pid
}
init(
id: String,
stdio: HostStdio,
spec: ContainerizationOCI.Spec,
ociRuntimePath: String? = nil,
log: Logger
) async throws {
// [Nucleic vendored patch] `spec` is mutated below to relocate the init into its own leaf cgroup
// when per-exec isolation is set up.
var spec = spec
let containerCgroup: String = {
if let p = spec.linux?.cgroupsPath, !p.isEmpty { return p }
return "/container/\(id)"
}()
let cgManager = Cgroup2Manager(
group: URL(filePath: containerCgroup),
logger: log
)
try cgManager.create()
// [Nucleic vendored patch] Per-exec cgroup isolation. Turn the container cgroup into an
// intermediary — resource ceiling on it, controllers delegated to children — and run the
// container init in its own leaf (`<container>/init`), so each exec later gets its OWN child
// cgroup (see `ManagedProcess.start`): a runaway session's OOM/CPU/fork-bomb is then scoped to
// that session and can't take down its siblings in the shared container. Best-effort: on ANY
// failure, wipe the partial state and fall back to the upstream flat layout (init + all execs
// share the container cgroup). `execCgroupParent == nil` marks flat mode.
var execParent: String? = nil
if spec.linux != nil {
do {
let initCg = Cgroup2Manager(group: URL(filePath: containerCgroup + "/init"), logger: log)
try initCg.create()
// Enabling controllers from the init leaf sets cgroup.subtree_control on every ancestor
// (incl. the container cgroup), which is what lets sibling child cgroups get memory/cpu/pids.
try initCg.toggleAllAvailableControllers(enable: true)
if let resources = spec.linux?.resources {
try cgManager.applyResources(resources: resources) // ceiling stays on the parent
}
spec.linux?.cgroupsPath = containerCgroup + "/init" // vmexec places the init here
spec.linux?.resources = nil // don't re-apply the ceiling to the init leaf
execParent = containerCgroup
} catch {
log.error("per-exec cgroup setup failed; using flat layout: \(error)")
// Undo any partial per-exec state so the container cgroup can hold the init again.
let initCg = Cgroup2Manager(group: URL(filePath: containerCgroup + "/init"), logger: log)
try? initCg.toggleAllAvailableControllers(enable: false)
try? initCg.remove()
try cgManager.toggleAllAvailableControllers(enable: true)
spec.linux?.cgroupsPath = containerCgroup
execParent = nil
}
} else {
try cgManager.toggleAllAvailableControllers(enable: true)
}
let bundle = try ContainerizationOCI.Bundle.create(
path: Self.craftBundlePath(id: id),
spec: spec
)
log.debug("created bundle with spec \(spec)")
do {
let initProcess: any ContainerProcess
if let runtimePath = ociRuntimePath {
// Use runc runtime
let runc = ProcessSupervisor.default.getRuncWithReaper(
Runc(
command: runtimePath,
root: "/run/runc"
)
)
initProcess = try RuncProcess(
id: id,
stdio: stdio,
bundle: bundle,
runc: runc,
log: log
)
self.needsCgroupCleanup = false
log.info("created runc init process with runtime: \(runtimePath)")
} else {
// Use vmexec runtime
initProcess = try ManagedProcess(
id: id,
stdio: stdio,
bundle: bundle,
owningPid: nil,
log: log
)
self.needsCgroupCleanup = true
log.info("created vmexec init process")
}
self.cgroupManager = cgManager
self.execCgroupParent = execParent
self.initProcess = initProcess
self.id = id
self.bundle = bundle
self.log = log
} catch {
try? cgManager.delete()
throw error
}
}
}
extension ManagedContainer {
// removeCgroupWithRetry will remove a cgroup path handling EAGAIN and EBUSY errors and
// retrying the remove after an exponential timeout
private func removeCgroupWithRetry() async throws {
var delay = 10 // 10ms
let maxRetries = 5
for i in 0..<maxRetries {
if i != 0 {
try await Task.sleep(for: .milliseconds(delay))
delay *= 2
}
do {
try self.cgroupManager.delete(force: true)
return
} catch let error as Cgroup2Manager.Error {
guard case .errno(let errnoValue, let message) = error,
errnoValue == EBUSY || errnoValue == EAGAIN
else {
throw error
}
self.log.warning(
"cgroup deletion failed with EBUSY/EAGAIN, retrying",
metadata: [
"attempt": "\(i + 1)",
"delay": "\(delay)",
"errno": "\(errnoValue)",
"context": "\(message)",
])
continue
}
}
throw ContainerizationError(
.internalError,
message: "cgroups: unable to remove cgroup after \(maxRetries) retries"
)
}
private func ensureExecExists(_ id: String) throws {
if self.execs[id] == nil {
throw ContainerizationError(
.invalidState,
message: "exec \(id) does not exist in container \(self.id)"
)
}
}
func createExec(
id: String,
stdio: HostStdio,
process: ContainerizationOCI.Process,
memoryLimitBytes: UInt64? = nil // [Nucleic vendored patch] hard per-exec memory.max
) throws {
log.debug("creating exec process with \(process)")
// Write the process config to the bundle, and pass this on
// over to ManagedProcess to deal with.
try self.bundle.createExecSpec(
id: id,
process: process
)
let process = try ManagedProcess(
id: id,
stdio: stdio,
bundle: self.bundle,
owningPid: self.initProcess.pid,
execCgroupParent: self.execCgroupParent, // [Nucleic vendored patch] per-exec cgroup
execMemoryLimitBytes: memoryLimitBytes, // [Nucleic vendored patch]
log: self.log
)
self.execs[id] = process
}
func start(execID: String) async throws -> Int32 {
let proc = try self.getExecOrInit(execID: execID)
return try await ProcessSupervisor.default.start(process: proc)
}
func wait(execID: String) async throws -> ContainerExitStatus {
let proc = try self.getExecOrInit(execID: execID)
return await proc.wait()
}
func kill(execID: String, _ signal: Int32) async throws {
let proc = try self.getExecOrInit(execID: execID)
try await proc.kill(signal)
}
func resize(execID: String, size: Terminal.Size) throws {
let proc = try self.getExecOrInit(execID: execID)
try proc.resize(size: size)
}
func closeStdin(execID: String) throws {
let proc = try self.getExecOrInit(execID: execID)
try proc.closeStdin()
}
func deleteExec(id: String) throws {
try ensureExecExists(id)
do {
try self.bundle.deleteExecSpec(id: id)
} catch {
self.log.error("failed to remove exec spec from filesystem: \(error)")
}
self.execs.removeValue(forKey: id)
}
func delete() async throws {
// Delete the init process if it's a RuncProcess
try await self.initProcess.delete()
// Delete the bundle and cgroup
try self.bundle.delete()
if self.needsCgroupCleanup {
try await self.removeCgroupWithRetry()
}
}
func stats(_ categories: Cgroup2StatsCategory = .all) throws -> Cgroup2Stats {
try self.cgroupManager.stats(categories)
}
func getMemoryEvents() throws -> MemoryEvents {
try self.cgroupManager.getMemoryEvents()
}
func getExecOrInit(execID: String) throws -> any ContainerProcess {
if execID == self.id {
return self.initProcess
}
guard let proc = self.execs[execID] else {
throw ContainerizationError(
.invalidState,
message: "exec \(execID) does not exist in container \(self.id)"
)
}
return proc
}
}
extension ContainerizationOCI.Bundle {
func createExecSpec(id: String, process: ContainerizationOCI.Process) throws {
let specDir = self.path.appending(path: "execs/\(id)")
let fm = FileManager.default
try fm.createDirectory(
atPath: specDir.path,
withIntermediateDirectories: true
)
let specData = try JSONEncoder().encode(process)
let processConfigPath = specDir.appending(path: "process.json")
try specData.write(to: processConfigPath)
}
func getExecSpecPath(id: String) -> URL {
self.path.appending(path: "execs/\(id)/process.json")
}
func deleteExecSpec(id: String) throws {
let specDir = self.path.appending(path: "execs/\(id)")
let fm = FileManager.default
try fm.removeItem(at: specDir)
}
}
extension ManagedContainer {
static func craftBundlePath(id: String) -> URL {
URL(fileURLWithPath: "/run/container").appending(path: id)
}
}
#endif