Per-exec cgroups (guest patch #9): scope a session's OOM/CPU/fork-bomb to itself

Restructures the guest cgroup layout so each exec gets its OWN child cgroup
(/container/<id>/<execID>) with memory.oom.group=1, a fair cpu.weight, and a
pids.max backstop — so one control session can't OOM-kill, starve, or fork-bomb
its siblings in the shared container. The container init moves to its own leaf
so the container cgroup can delegate controllers to children (cgroup v2
no-internal-process rule). New Cgroup2Manager helpers: setOomGroup/setCpuWeight/
setPidsMax/remove.

Best-effort with graceful fallback: any failure in the per-exec setup wipes the
partial state and reverts to today's flat layout, and each exec falls back to the
container cgroup — a cgroup hiccup degrades to current behavior, never a failed
start.

COMPILE-VERIFIED via the musl cross-build; NOT yet runtime-validated. Built as
image tag -nucleic2; vminitReference stays on the validated -nucleic1 until
-nucleic2 is checked in a real container. A hard host-configured per-exec
memory.max (exec-RPC resources field) remains a follow-up.

Co-Authored-By: Claude Opus 4.8 <[email protected]>
This commit is contained in:
2026-07-13 20:03:20 -07:00
co-authored by Claude Opus 4.8
parent 0f9957d5e5
commit 2eb563c90c
4 changed files with 130 additions and 32 deletions
@@ -267,6 +267,33 @@ public struct Cgroup2Manager: Sendable {
fileName: "memory.low")
}
/// [Nucleic vendored patch] Make the kernel OOM-killer treat this cgroup as an atomic unit: when a
/// memory limit (this cgroup's or an ancestor's) forces an OOM, the whole cgroup's process tree is
/// killed together rather than one victim. Used to scope a runaway exec's OOM to that exec so
/// sibling execs in the same container survive.
package func setOomGroup(_ enabled: Bool) throws {
try Self.writeValue(path: self.path, value: enabled ? "1" : "0", fileName: "memory.oom.group")
}
/// [Nucleic vendored patch] Relative CPU share under contention (cgroup v2 `cpu.weight`, 1…10000,
/// default 100). Equal weights give each exec a fair slice so one busy session can't starve
/// siblings of CPU.
package func setCpuWeight(_ weight: UInt64) throws {
try Self.writeValue(path: self.path, value: String(weight), fileName: "cpu.weight")
}
/// [Nucleic vendored patch] Cap the pids in this cgroup (`pids.max`) — a fork-bomb backstop so one
/// exec can't exhaust the pid space and wedge its siblings.
package func setPidsMax(_ max: UInt64) throws {
try Self.writeValue(path: self.path, value: String(max), fileName: "pids.max")
}
/// [Nucleic vendored patch] Remove this cgroup directory (rmdir). The cgroup must already be empty
/// of processes and child cgroups. Best-effort partial-setup cleanup for the per-exec layout.
package func remove() throws {
try FileManager.default.removeItem(at: self.path)
}
package func getMemoryEvents() throws -> MemoryEvents {
let content = try readFileContent(fileName: "memory.events")
let values = parseKeyValuePairs(content)
@@ -32,6 +32,10 @@ public actor ManagedContainer {
private let bundle: ContainerizationOCI.Bundle
private let needsCgroupCleanup: Bool
private var execs: [String: any ContainerProcess] = [:]
// [Nucleic vendored patch] When per-exec cgroup isolation is active, the container cgroup that each
// exec gets its own child under (`<parent>/<execID>`). nil = the legacy flat layout (init + all
// execs share the container cgroup).
private let execCgroupParent: String?
public var pid: Int32? {
self.initProcess.pid
@@ -44,11 +48,53 @@ public actor ManagedContainer {
ociRuntimePath: String? = nil,
log: Logger
) async throws {
var cgroupsPath: String
if let cgPath = spec.linux?.cgroupsPath {
cgroupsPath = cgPath
// [Nucleic vendored patch] `spec` is mutated below to relocate the init into its own leaf cgroup
// when per-exec isolation is set up.
var spec = spec
let containerCgroup: String = {
if let p = spec.linux?.cgroupsPath, !p.isEmpty { return p }
return "/container/\(id)"
}()
let cgManager = Cgroup2Manager(
group: URL(filePath: containerCgroup),
logger: log
)
try cgManager.create()
// [Nucleic vendored patch] Per-exec cgroup isolation. Turn the container cgroup into an
// intermediary — resource ceiling on it, controllers delegated to children — and run the
// container init in its own leaf (`<container>/init`), so each exec later gets its OWN child
// cgroup (see `ManagedProcess.start`): a runaway session's OOM/CPU/fork-bomb is then scoped to
// that session and can't take down its siblings in the shared container. Best-effort: on ANY
// failure, wipe the partial state and fall back to the upstream flat layout (init + all execs
// share the container cgroup). `execCgroupParent == nil` marks flat mode.
var execParent: String? = nil
if spec.linux != nil {
do {
let initCg = Cgroup2Manager(group: URL(filePath: containerCgroup + "/init"), logger: log)
try initCg.create()
// Enabling controllers from the init leaf sets cgroup.subtree_control on every ancestor
// (incl. the container cgroup), which is what lets sibling child cgroups get memory/cpu/pids.
try initCg.toggleAllAvailableControllers(enable: true)
if let resources = spec.linux?.resources {
try cgManager.applyResources(resources: resources) // ceiling stays on the parent
}
spec.linux?.cgroupsPath = containerCgroup + "/init" // vmexec places the init here
spec.linux?.resources = nil // don't re-apply the ceiling to the init leaf
execParent = containerCgroup
} catch {
log.error("per-exec cgroup setup failed; using flat layout: \(error)")
// Undo any partial per-exec state so the container cgroup can hold the init again.
let initCg = Cgroup2Manager(group: URL(filePath: containerCgroup + "/init"), logger: log)
try? initCg.toggleAllAvailableControllers(enable: false)
try? initCg.remove()
try cgManager.toggleAllAvailableControllers(enable: true)
spec.linux?.cgroupsPath = containerCgroup
execParent = nil
}
} else {
cgroupsPath = "/container/\(id)"
try cgManager.toggleAllAvailableControllers(enable: true)
}
let bundle = try ContainerizationOCI.Bundle.create(
@@ -57,15 +103,7 @@ public actor ManagedContainer {
)
log.debug("created bundle with spec \(spec)")
let cgManager = Cgroup2Manager(
group: URL(filePath: cgroupsPath),
logger: log
)
try cgManager.create()
do {
try cgManager.toggleAllAvailableControllers(enable: true)
let initProcess: any ContainerProcess
if let runtimePath = ociRuntimePath {
@@ -99,6 +137,7 @@ public actor ManagedContainer {
}
self.cgroupManager = cgManager
self.execCgroupParent = execParent
self.initProcess = initProcess
self.id = id
self.bundle = bundle
@@ -177,6 +216,7 @@ extension ManagedContainer {
stdio: stdio,
bundle: self.bundle,
owningPid: self.initProcess.pid,
execCgroupParent: self.execCgroupParent, // [Nucleic vendored patch] per-exec cgroup
log: self.log
)
self.execs[id] = process
@@ -57,6 +57,9 @@ final class ManagedProcess: ContainerProcess, Sendable {
private let command: Command
private let state: Mutex<State>
private let owningPid: Int32?
// [Nucleic vendored patch] Parent cgroup for this exec's OWN per-exec child (`<parent>/<id>`);
// nil means the legacy flat layout (join the container/init cgroup via `owningPid`).
private let execCgroupParent: String?
private let ackPipe: Pipe
private let syncPipe: Pipe
private let errorPipe: Pipe
@@ -74,6 +77,7 @@ final class ManagedProcess: ContainerProcess, Sendable {
stdio: HostStdio,
bundle: ContainerizationOCI.Bundle,
owningPid: Int32? = nil,
execCgroupParent: String? = nil, // [Nucleic vendored patch]
log: Logger
) throws {
self.id = id
@@ -81,6 +85,7 @@ final class ManagedProcess: ContainerProcess, Sendable {
log[metadataKey: "id"] = "\(id)"
self.log = log
self.owningPid = owningPid
self.execCgroupParent = execCgroupParent
let syncPipe = Pipe()
try syncPipe.setCloexec()
@@ -215,7 +220,27 @@ extension ManagedProcess {
// This should probably happen in vmexec, but we don't need to set any cgroup
// toggles so the problem is much simpler to just do it here.
if let owningPid {
if let parent = execCgroupParent {
// [Nucleic vendored patch] Per-exec cgroup: place this exec in its OWN child cgroup
// (`<parent>/<id>`) with memory.oom.group (a runaway session's OOM kills only its
// tree), a fair cpu.weight, and a pids.max fork-bomb backstop — so one session can't
// OOM-kill / starve / fork-bomb its siblings in the shared container. Best-effort:
// on ANY error fall back to the container/init cgroup so a cgroup hiccup never blocks
// the exec from starting.
do {
let execCg = Cgroup2Manager(group: URL(filePath: parent + "/" + id), logger: log)
try execCg.create()
try? execCg.setOomGroup(true)
try? execCg.setCpuWeight(100)
try? execCg.setPidsMax(4096)
try execCg.addProcess(pid: pid)
} catch {
log.error("per-exec cgroup for \(id) failed; joining the container cgroup: \(error)")
if let owningPid {
try? Cgroup2Manager.loadFromPid(pid: owningPid).addProcess(pid: pid)
}
}
} else if let owningPid {
let cgManager = try Cgroup2Manager.loadFromPid(pid: owningPid)
try cgManager.addProcess(pid: pid)
}