Per-exec cgroups (guest patch #9): scope a session's OOM/CPU/fork-bomb to itself
Restructures the guest cgroup layout so each exec gets its OWN child cgroup (/container/<id>/<execID>) with memory.oom.group=1, a fair cpu.weight, and a pids.max backstop — so one control session can't OOM-kill, starve, or fork-bomb its siblings in the shared container. The container init moves to its own leaf so the container cgroup can delegate controllers to children (cgroup v2 no-internal-process rule). New Cgroup2Manager helpers: setOomGroup/setCpuWeight/ setPidsMax/remove. Best-effort with graceful fallback: any failure in the per-exec setup wipes the partial state and reverts to today's flat layout, and each exec falls back to the container cgroup — a cgroup hiccup degrades to current behavior, never a failed start. COMPILE-VERIFIED via the musl cross-build; NOT yet runtime-validated. Built as image tag -nucleic2; vminitReference stays on the validated -nucleic1 until -nucleic2 is checked in a real container. A hard host-configured per-exec memory.max (exec-RPC resources field) remains a follow-up. Co-Authored-By: Claude Opus 4.8 <[email protected]>
This commit is contained in:
@@ -32,6 +32,10 @@ public actor ManagedContainer {
|
||||
private let bundle: ContainerizationOCI.Bundle
|
||||
private let needsCgroupCleanup: Bool
|
||||
private var execs: [String: any ContainerProcess] = [:]
|
||||
// [Nucleic vendored patch] When per-exec cgroup isolation is active, the container cgroup that each
|
||||
// exec gets its own child under (`<parent>/<execID>`). nil = the legacy flat layout (init + all
|
||||
// execs share the container cgroup).
|
||||
private let execCgroupParent: String?
|
||||
|
||||
public var pid: Int32? {
|
||||
self.initProcess.pid
|
||||
@@ -44,11 +48,53 @@ public actor ManagedContainer {
|
||||
ociRuntimePath: String? = nil,
|
||||
log: Logger
|
||||
) async throws {
|
||||
var cgroupsPath: String
|
||||
if let cgPath = spec.linux?.cgroupsPath {
|
||||
cgroupsPath = cgPath
|
||||
// [Nucleic vendored patch] `spec` is mutated below to relocate the init into its own leaf cgroup
|
||||
// when per-exec isolation is set up.
|
||||
var spec = spec
|
||||
let containerCgroup: String = {
|
||||
if let p = spec.linux?.cgroupsPath, !p.isEmpty { return p }
|
||||
return "/container/\(id)"
|
||||
}()
|
||||
|
||||
let cgManager = Cgroup2Manager(
|
||||
group: URL(filePath: containerCgroup),
|
||||
logger: log
|
||||
)
|
||||
try cgManager.create()
|
||||
|
||||
// [Nucleic vendored patch] Per-exec cgroup isolation. Turn the container cgroup into an
|
||||
// intermediary — resource ceiling on it, controllers delegated to children — and run the
|
||||
// container init in its own leaf (`<container>/init`), so each exec later gets its OWN child
|
||||
// cgroup (see `ManagedProcess.start`): a runaway session's OOM/CPU/fork-bomb is then scoped to
|
||||
// that session and can't take down its siblings in the shared container. Best-effort: on ANY
|
||||
// failure, wipe the partial state and fall back to the upstream flat layout (init + all execs
|
||||
// share the container cgroup). `execCgroupParent == nil` marks flat mode.
|
||||
var execParent: String? = nil
|
||||
if spec.linux != nil {
|
||||
do {
|
||||
let initCg = Cgroup2Manager(group: URL(filePath: containerCgroup + "/init"), logger: log)
|
||||
try initCg.create()
|
||||
// Enabling controllers from the init leaf sets cgroup.subtree_control on every ancestor
|
||||
// (incl. the container cgroup), which is what lets sibling child cgroups get memory/cpu/pids.
|
||||
try initCg.toggleAllAvailableControllers(enable: true)
|
||||
if let resources = spec.linux?.resources {
|
||||
try cgManager.applyResources(resources: resources) // ceiling stays on the parent
|
||||
}
|
||||
spec.linux?.cgroupsPath = containerCgroup + "/init" // vmexec places the init here
|
||||
spec.linux?.resources = nil // don't re-apply the ceiling to the init leaf
|
||||
execParent = containerCgroup
|
||||
} catch {
|
||||
log.error("per-exec cgroup setup failed; using flat layout: \(error)")
|
||||
// Undo any partial per-exec state so the container cgroup can hold the init again.
|
||||
let initCg = Cgroup2Manager(group: URL(filePath: containerCgroup + "/init"), logger: log)
|
||||
try? initCg.toggleAllAvailableControllers(enable: false)
|
||||
try? initCg.remove()
|
||||
try cgManager.toggleAllAvailableControllers(enable: true)
|
||||
spec.linux?.cgroupsPath = containerCgroup
|
||||
execParent = nil
|
||||
}
|
||||
} else {
|
||||
cgroupsPath = "/container/\(id)"
|
||||
try cgManager.toggleAllAvailableControllers(enable: true)
|
||||
}
|
||||
|
||||
let bundle = try ContainerizationOCI.Bundle.create(
|
||||
@@ -57,15 +103,7 @@ public actor ManagedContainer {
|
||||
)
|
||||
log.debug("created bundle with spec \(spec)")
|
||||
|
||||
let cgManager = Cgroup2Manager(
|
||||
group: URL(filePath: cgroupsPath),
|
||||
logger: log
|
||||
)
|
||||
try cgManager.create()
|
||||
|
||||
do {
|
||||
try cgManager.toggleAllAvailableControllers(enable: true)
|
||||
|
||||
let initProcess: any ContainerProcess
|
||||
|
||||
if let runtimePath = ociRuntimePath {
|
||||
@@ -99,6 +137,7 @@ public actor ManagedContainer {
|
||||
}
|
||||
|
||||
self.cgroupManager = cgManager
|
||||
self.execCgroupParent = execParent
|
||||
self.initProcess = initProcess
|
||||
self.id = id
|
||||
self.bundle = bundle
|
||||
@@ -177,6 +216,7 @@ extension ManagedContainer {
|
||||
stdio: stdio,
|
||||
bundle: self.bundle,
|
||||
owningPid: self.initProcess.pid,
|
||||
execCgroupParent: self.execCgroupParent, // [Nucleic vendored patch] per-exec cgroup
|
||||
log: self.log
|
||||
)
|
||||
self.execs[id] = process
|
||||
|
||||
Reference in New Issue
Block a user