Per-exec cgroups (guest patch #9): scope a session's OOM/CPU/fork-bomb to itself
Restructures the guest cgroup layout so each exec gets its OWN child cgroup (/container/<id>/<execID>) with memory.oom.group=1, a fair cpu.weight, and a pids.max backstop — so one control session can't OOM-kill, starve, or fork-bomb its siblings in the shared container. The container init moves to its own leaf so the container cgroup can delegate controllers to children (cgroup v2 no-internal-process rule). New Cgroup2Manager helpers: setOomGroup/setCpuWeight/ setPidsMax/remove. Best-effort with graceful fallback: any failure in the per-exec setup wipes the partial state and reverts to today's flat layout, and each exec falls back to the container cgroup — a cgroup hiccup degrades to current behavior, never a failed start. COMPILE-VERIFIED via the musl cross-build; NOT yet runtime-validated. Built as image tag -nucleic2; vminitReference stays on the validated -nucleic1 until -nucleic2 is checked in a real container. A hard host-configured per-exec memory.max (exec-RPC resources field) remains a follow-up. Co-Authored-By: Claude Opus 4.8 <[email protected]>
This commit is contained in:
@@ -57,6 +57,9 @@ final class ManagedProcess: ContainerProcess, Sendable {
|
||||
private let command: Command
|
||||
private let state: Mutex<State>
|
||||
private let owningPid: Int32?
|
||||
// [Nucleic vendored patch] Parent cgroup for this exec's OWN per-exec child (`<parent>/<id>`);
|
||||
// nil means the legacy flat layout (join the container/init cgroup via `owningPid`).
|
||||
private let execCgroupParent: String?
|
||||
private let ackPipe: Pipe
|
||||
private let syncPipe: Pipe
|
||||
private let errorPipe: Pipe
|
||||
@@ -74,6 +77,7 @@ final class ManagedProcess: ContainerProcess, Sendable {
|
||||
stdio: HostStdio,
|
||||
bundle: ContainerizationOCI.Bundle,
|
||||
owningPid: Int32? = nil,
|
||||
execCgroupParent: String? = nil, // [Nucleic vendored patch]
|
||||
log: Logger
|
||||
) throws {
|
||||
self.id = id
|
||||
@@ -81,6 +85,7 @@ final class ManagedProcess: ContainerProcess, Sendable {
|
||||
log[metadataKey: "id"] = "\(id)"
|
||||
self.log = log
|
||||
self.owningPid = owningPid
|
||||
self.execCgroupParent = execCgroupParent
|
||||
|
||||
let syncPipe = Pipe()
|
||||
try syncPipe.setCloexec()
|
||||
@@ -215,7 +220,27 @@ extension ManagedProcess {
|
||||
|
||||
// This should probably happen in vmexec, but we don't need to set any cgroup
|
||||
// toggles so the problem is much simpler to just do it here.
|
||||
if let owningPid {
|
||||
if let parent = execCgroupParent {
|
||||
// [Nucleic vendored patch] Per-exec cgroup: place this exec in its OWN child cgroup
|
||||
// (`<parent>/<id>`) with memory.oom.group (a runaway session's OOM kills only its
|
||||
// tree), a fair cpu.weight, and a pids.max fork-bomb backstop — so one session can't
|
||||
// OOM-kill / starve / fork-bomb its siblings in the shared container. Best-effort:
|
||||
// on ANY error fall back to the container/init cgroup so a cgroup hiccup never blocks
|
||||
// the exec from starting.
|
||||
do {
|
||||
let execCg = Cgroup2Manager(group: URL(filePath: parent + "/" + id), logger: log)
|
||||
try execCg.create()
|
||||
try? execCg.setOomGroup(true)
|
||||
try? execCg.setCpuWeight(100)
|
||||
try? execCg.setPidsMax(4096)
|
||||
try execCg.addProcess(pid: pid)
|
||||
} catch {
|
||||
log.error("per-exec cgroup for \(id) failed; joining the container cgroup: \(error)")
|
||||
if let owningPid {
|
||||
try? Cgroup2Manager.loadFromPid(pid: owningPid).addProcess(pid: pid)
|
||||
}
|
||||
}
|
||||
} else if let owningPid {
|
||||
let cgManager = try Cgroup2Manager.loadFromPid(pid: owningPid)
|
||||
try cgManager.addProcess(pid: pid)
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user