Vendor apple/containerization with a VM-extensions forwarding patch

Switch the containerization dependency from the github URL to a vendored copy
(third_party/containerization, upstream commit 6b7b42ca) referenced by path, so
we can carry a small local patch that upstream lacks: LinuxContainer.Configuration
gains a `vmExtensions` field forwarded into VMConfiguration.extensions. Upstream
already supports VMConfiguration.extensions + the VZInstanceExtension hook, but
LinuxContainer — our only entry point — never forwarded them, so there was no way
to attach a device (e.g. a memory balloon) to a container's VM.

Tests/, docs/, examples/, images/ and the corresponding test targets are trimmed
for footprint (we never build the dependency's tests). See PATCHES.md for the full
diff vs. upstream and the re-vendoring procedure. Also adds the ContainerizationExtras
product to NucleicCore (AddressAllocator, named in the configureVZ signature).

Co-Authored-By: Claude Opus 4.8 <[email protected]>
This commit is contained in:
Nucleic
2026-06-21 20:22:21 -07:00
co-authored by Claude Opus 4.8
commit 11b9825e09
280 changed files with 75699 additions and 0 deletions
+32
View File
@@ -0,0 +1,32 @@
/*
* Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* https://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#if defined(__linux__)
#include <sys/syscall.h>
#include <unistd.h>
#include "capability.h"
// Capability syscall wrappers
int CZ_capget(void *header, void *data) {
return syscall(SYS_capget, header, data);
}
int CZ_capset(void *header, void *data) {
return syscall(SYS_capset, header, data);
}
#endif
+393
View File
@@ -0,0 +1,393 @@
/*
* Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* https://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#if defined(__linux__) || defined(__APPLE__)
#include <errno.h>
#include <fcntl.h>
#include <dirent.h>
#include <limits.h>
#include <pthread.h>
#include <signal.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <sys/ioctl.h>
#include <sys/resource.h>
#include <sys/syscall.h>
#include <sys/time.h>
#include <sys/types.h>
#include <sys/wait.h>
#include <unistd.h>
#if defined(__linux__)
#include <sys/prctl.h>
#endif
#include "exec_command.h"
#ifndef SYS_close_range
#define SYS_close_range 436
#endif
#ifndef CLOSE_RANGE_CLOEXEC
#define CLOSE_RANGE_CLOEXEC 0x4
#endif
static int mark_cloexec(int fd) {
int flags = fcntl(fd, F_GETFD);
if (flags == -1) return flags;
if (flags & FD_CLOEXEC) return 0;
return fcntl(fd, F_SETFD, flags | FD_CLOEXEC);
}
static int cloexec_from(int min_fd) {
#if defined(__linux__)
// First try close_range.
long ret = syscall(SYS_close_range, min_fd, ~0U, CLOSE_RANGE_CLOEXEC);
if (ret == 0) {
return 0;
}
const char* dirpath = "/proc/self/fd";
#elif defined(__APPLE__)
const char* dirpath = "/dev/fd";
#endif
DIR *dp = opendir(dirpath);
if (!dp) return -1;
int dp_fd = dirfd(dp);
struct dirent *de;
while ((de = readdir(dp))) {
if (de->d_name[0] == '.') continue;
char *end;
long val = strtol(de->d_name, &end, 10);
if (*end || val < 0 || val > INT_MAX) continue;
int fd = (int)val;
if (fd < min_fd || fd == dp_fd) continue;
int ret = mark_cloexec(fd);
if (ret != 0) {
return ret;
}
}
close(dp_fd);
closedir(dp);
return 0;
}
void exec_command_attrs_init(struct exec_command_attrs *attrs) {
attrs->setpgid = 0;
attrs->pgid = 0;
attrs->setsid = 0;
attrs->setctty = 0;
attrs->ctty = 0;
attrs->mask = 0;
attrs->uid = -1;
attrs->gid = -1;
attrs->pdeathSignal = 0;
attrs->setfgpgrp = 0;
}
static void child_handler(const int sync_pipes[2], const char *executable,
char *const args[], char *const environment[],
const int file_handles[], const int file_handle_count,
const char *cwd, const sigset_t old_mask,
const struct exec_command_attrs attrs) {
int i = 0;
int err = 0;
int fd_index = 0;
int fd_table[file_handle_count];
struct rlimit limits = {0};
int syncfd = sync_pipes[1];
struct sigaction action = {0};
// Closing our parent's side of the pipe
if (close(sync_pipes[0]) < 0) {
goto fail;
}
// Setup process group and foreground before clearing signal mask.
if (attrs.setpgid) {
if (setpgid(0, attrs.pgid) < 0) {
goto fail;
}
}
// Make the new process group the foreground process group so it can read from the TTY.
if (attrs.setfgpgrp) {
if (tcsetpgrp(STDIN_FILENO, getpgrp()) < 0) {
if (errno != ENOTTY && errno != ENXIO) {
goto fail;
}
}
}
// clear sighandlers
action.sa_flags = 0;
action.sa_handler = SIG_DFL;
sigemptyset(&action.sa_mask);
for (i = 0; i < NSIG; i++) {
sigaction(i, &action, 0);
}
sigset_t local_mask;
sigemptyset(&local_mask);
if (pthread_sigmask(SIG_SETMASK, &local_mask, NULL) < 0) {
goto fail;
}
// start shuffling fds.
// look at all the file handles and find the highest one,
// use that for our pipe,
//
// Then, we need to start dup2 the fds starting for the final process
// at 0-n.
// as an example we have this list of FDs that should be passed to the
// process:
//
/*
The index of this list is the final result that the new process expects.
The values are open fds provided from the parent process.
[0] == 12
[1] == 7
[2] == 9
[3] == 0
We also have a pipe to sync the child and parent so that adds an additional
parameter to consider.
So we start by finding the highest open fd in the list, then move our pipe to
the next.
i.e. fd12 is highest so move our pipe to fd13
Now start moving all the fds above our pipe as we will need to start placing
the fds in the child process into the right order. Make sure they are all
marked cloexec.
pipe == 13
[0] == 12 dup2 14
[1] == 7 dup2 15
[2] == 9 dup2 16
[3] == 0 dup2 17
Now overwrite the fd table for the child with the current index.
Make index == fd.
pipe == 13
[0] == 14 dup2 0
[1] == 15 dup2 1
[2] == 16 dup2 2
[3] == 17 dup2 3
Clear cloexec on this new fds.
*/
// find the highest fd value in our list.
for (i = 0; i < file_handle_count; i++) {
if (file_handles[i] > fd_index) {
fd_index = file_handles[i];
}
fd_table[i] = file_handles[i];
}
// now fd_index is == to the highest fd in our list of handles.
// Increment it and set our pipe to it.
fd_index++;
if (syncfd != fd_index) {
if (dup2(syncfd, fd_index) < 0) {
goto fail;
}
if (close(syncfd) < 0) {
goto fail;
}
syncfd = fd_index;
}
fd_index++;
// make sure our syncfd retains its cloexec
if (fcntl(syncfd, F_SETFD, FD_CLOEXEC) == -1) {
goto fail;
}
// move the rest of the fds up above our index if they don't match the index.
for (i = 0; i < file_handle_count; i++) {
if (fd_table[i] == i) {
continue;
}
if (dup2(fd_table[i], fd_index) < 0) {
goto fail;
}
if (fcntl(fd_index, F_SETFD, FD_CLOEXEC) == -1) {
goto fail;
}
fd_table[i] = fd_index;
fd_index++;
}
// now create the child process's final fd table. where i == i
for (i = 0; i < file_handle_count; i++) {
if (fd_table[i] != i) {
if (dup2(fd_table[i], i) < 0) {
goto fail;
}
}
// now fd[i] should == i
// clear cloexec as this fd is where we want it.
if (fcntl(i, F_SETFD, 0) == -1) {
goto fail;
}
}
if (attrs.setsid) {
if (setsid() == -1) {
goto fail;
}
}
if (attrs.setctty) {
if (ioctl(attrs.ctty, TIOCSCTTY, 0)) {
goto fail;
}
}
#if defined(__linux__)
// Set parent death signal if specified
if (attrs.pdeathSignal != 0) {
if (prctl(PR_SET_PDEATHSIG, attrs.pdeathSignal) != 0) {
goto fail;
}
}
#endif
// close exec everything outside of our child's fd_table.
if (cloexec_from(file_handle_count) != 0) {
goto fail;
}
// set gid
if (attrs.gid != -1) {
if (setgid(attrs.gid) != 0) {
goto fail;
}
}
// set uid
if (attrs.uid != -1) {
if (setreuid(attrs.uid, attrs.uid) != 0) {
goto fail;
}
}
if (cwd != NULL) {
if (chdir(cwd)) {
goto fail;
}
}
execve(executable, args, environment);
fail:
err = errno;
if (err) {
// send our error to the parent
while (write(syncfd, &err, sizeof(err)) < 0)
;
}
exit(127);
}
int exec_command(pid_t *result, const char *executable, char *const args[],
char *const envp[], const int file_handles[],
const int file_handle_count, const char *working_directory,
struct exec_command_attrs *attrs) {
pid_t pid = 0;
int err = 0;
int sync_pipe[2];
sigset_t old_mask;
sigset_t all;
sigfillset(&all);
if (pipe(sync_pipe)) {
goto fail;
}
if (pthread_sigmask(SIG_SETMASK, &all, &old_mask) < 0) {
goto fail;
}
pid = fork();
if (pid == -1) {
close(sync_pipe[0]);
close(sync_pipe[1]);
goto fail;
}
if (pid == 0) {
// hand off to child
child_handler(sync_pipe, executable, args, envp, file_handles,
file_handle_count, working_directory, old_mask, *attrs);
exit(EXIT_FAILURE);
}
// handle parent operations
if (close(sync_pipe[1]) < 0) {
goto fail;
}
// sync with our child process
err = 0;
ssize_t size = read(sync_pipe[0], &err, sizeof(err));
// -- we didn't get an errno back
if (size != sizeof(err)) {
// will be used as return result
err = 0;
} else {
// we did get an errno back from the child process and our
// err var is set to that errno
// lets set our errno and then reap the process
errno = err;
int status = 0;
waitpid(pid, &status, 0);
// lets continue our journey below
}
if (close(sync_pipe[0]) < 0) {
goto fail;
}
if (err) {
goto fail;
}
(*result) = pid;
err = 0;
fail:
if (pthread_sigmask(SIG_SETMASK, &old_mask, 0) < 0) {
printf("restoring signal mask: %s\n", strerror(errno));
}
if (err) {
printf("exec_command execve: %s\n", strerror(err));
return -1;
}
return 0;
}
#endif
+28
View File
@@ -0,0 +1,28 @@
/*
* Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* https://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#ifndef __CAPABILITY_H
#define __CAPABILITY_H
#if defined(__linux__)
// Capability syscall wrappers
int CZ_capget(void *header, void *data);
int CZ_capset(void *header, void *data);
#endif
#endif
+56
View File
@@ -0,0 +1,56 @@
/*
* Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* https://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#ifndef exec_command_h
#define exec_command_h
#if defined(__linux__) || defined(__APPLE__)
#include <sys/types.h>
#include <unistd.h>
struct exec_command_attrs {
int setpgid;
/// parent group id
pid_t pgid;
/// set the controlling terminal
int setctty;
/// controlling terminal fd
int ctty;
/// set the process as session leader
int setsid;
/// set the process user id
uid_t uid;
/// set the process group id
gid_t gid;
/// signal mask for the child process
int mask;
/// parent death signal (Linux only, 0 to disable)
int pdeathSignal;
/// make the new process group the foreground process group
int setfgpgrp;
};
void exec_command_attrs_init(struct exec_command_attrs *attrs);
/// spawn a new child process with the provided attrs
int exec_command(pid_t *result, const char *executable, char *const argv[],
char *const envp[], const int file_handles[],
const int file_handle_count, const char *working_directory,
struct exec_command_attrs *attrs);
#endif /* defined(__linux__) || defined(__APPLE__) */
#endif /* exec_command_h */
+32
View File
@@ -0,0 +1,32 @@
/*
* Copyright © 2026 Apple Inc. and the Containerization project authors.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* https://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
// The below fall into two main categories:
// 1. Aren't exposed by Swifts glibc modulemap.
// 2. Don't have syscall wrappers/definitions in glibc/musl.
#ifndef __LINUX_SHIM_H
#define __LINUX_SHIM_H
#if defined(__linux__)
#include <sys/epoll.h>
#include <sys/eventfd.h>
#include <sys/vfs.h>
#endif /* __linux__ */
#endif /* __LINUX_SHIM_H */
+37
View File
@@ -0,0 +1,37 @@
/*
* Copyright © 2026 Apple Inc. and the Containerization project authors.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* https://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#ifndef __OPENAT2_H
#define __OPENAT2_H
#include <sys/types.h>
#ifndef RESOLVE_IN_ROOT
#define RESOLVE_IN_ROOT 0x10
#endif
struct cz_open_how {
unsigned long long flags;
unsigned long long mode;
unsigned long long resolve;
};
/// openat2(2) wrapper. Musl does not provide openat2 so we invoke the syscall
/// directly. Requires Linux 5.6+.
int CZ_openat2(int dirfd, const char *pathname, struct cz_open_how *how,
size_t size);
#endif
+33
View File
@@ -0,0 +1,33 @@
/*
* Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* https://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#ifndef __PRCTL_H
#define __PRCTL_H
#if defined(__linux__)
#include <sys/types.h>
// Capability management prctl wrappers
int CZ_prctl_set_keepcaps();
int CZ_prctl_clear_keepcaps();
int CZ_prctl_capbset_drop(unsigned int capability);
int CZ_prctl_cap_ambient_clear_all();
int CZ_prctl_cap_ambient_raise(unsigned int capability);
#endif
#endif
+29
View File
@@ -0,0 +1,29 @@
/*
* Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* https://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#ifndef socket_helpers_h
#define socket_helpers_h
#include <sys/socket.h>
#include <stdint.h>
// Helper functions to access CMSG macros from Swift
struct cmsghdr* CZ_CMSG_FIRSTHDR(struct msghdr *msg);
void* CZ_CMSG_DATA(struct cmsghdr *cmsg);
size_t CZ_CMSG_SPACE(size_t length);
size_t CZ_CMSG_LEN(size_t length);
#endif /* socket_helpers_h */
+33
View File
@@ -0,0 +1,33 @@
/*
* Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* https://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
//
#ifndef vsock_h
#define vsock_h
#include <sys/ioctl.h>
#ifdef __APPLE__
#include <sys/vsock.h>
#else
#include <sys/socket.h>
#include <linux/vm_sockets.h>
#endif /* __APPLE__ */
extern const unsigned long VsockLocalCIDIoctl;
#endif /* vsock_h */
+33
View File
@@ -0,0 +1,33 @@
/*
* Copyright © 2026 Apple Inc. and the Containerization project authors.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* https://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#if defined(__linux__)
#include <sys/syscall.h>
#include <unistd.h>
#include "openat2.h"
#ifndef SYS_openat2
#define SYS_openat2 437
#endif
int CZ_openat2(int dirfd, const char *pathname, struct cz_open_how *how,
size_t size) {
return syscall(SYS_openat2, dirfd, pathname, how, size);
}
#endif
+47
View File
@@ -0,0 +1,47 @@
/*
* Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* https://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#if defined(__linux__)
#include <sys/prctl.h>
#include "prctl.h"
// Set keep caps to preserve capabilities across setuid()
int CZ_prctl_set_keepcaps() {
return prctl(PR_SET_KEEPCAPS, 1, 0, 0, 0);
}
// Clear keep caps after user change
int CZ_prctl_clear_keepcaps() {
return prctl(PR_SET_KEEPCAPS, 0, 0, 0, 0);
}
// Drop capability from bounding set
int CZ_prctl_capbset_drop(unsigned int capability) {
return prctl(PR_CAPBSET_DROP, capability, 0, 0, 0);
}
// Clear all ambient capabilities
int CZ_prctl_cap_ambient_clear_all() {
return prctl(PR_CAP_AMBIENT, PR_CAP_AMBIENT_CLEAR_ALL, 0, 0, 0);
}
// Raise ambient capability
int CZ_prctl_cap_ambient_raise(unsigned int capability) {
return prctl(PR_CAP_AMBIENT, PR_CAP_AMBIENT_RAISE, capability, 0, 0);
}
#endif
+33
View File
@@ -0,0 +1,33 @@
/*
* Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* https://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#include "socket_helpers.h"
struct cmsghdr* CZ_CMSG_FIRSTHDR(struct msghdr *msg) {
return CMSG_FIRSTHDR(msg);
}
void* CZ_CMSG_DATA(struct cmsghdr *cmsg) {
return CMSG_DATA(cmsg);
}
size_t CZ_CMSG_SPACE(size_t length) {
return CMSG_SPACE(length);
}
size_t CZ_CMSG_LEN(size_t length) {
return CMSG_LEN(length);
}
+19
View File
@@ -0,0 +1,19 @@
/*
* Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* https://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#include "vsock.h"
const unsigned long VsockLocalCIDIoctl = IOCTL_VM_SOCKETS_GET_LOCAL_CID;
@@ -0,0 +1,53 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationExtras
import ContainerizationOCI
/// A filesystem that was attached and able to be mounted inside the runtime environment.
public struct AttachedFilesystem: Sendable {
/// The type of the filesystem.
public var type: String
/// The path to the filesystem within a sandbox.
public var source: String
/// Destination when mounting the filesystem inside a sandbox.
public var destination: String
/// The options to use when mounting the filesystem.
public var options: [String]
public init(mount: Mount, allocator: any AddressAllocator<Character>) throws {
switch mount.runtimeOptions {
case .virtiofs:
let name = try hashFilePath(path: mount.source)
self.source = name
case .virtioblk:
let char = try allocator.allocate()
self.source = "/dev/vd\(char)"
case .shared, .any:
self.source = mount.source
}
self.type = mount.type
self.options = mount.options
self.destination = mount.destination
}
public init(type: String, source: String, destination: String, options: [String]) {
self.type = type
self.source = source
self.destination = destination
self.options = options
}
}
+27
View File
@@ -0,0 +1,27 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
/// The core protocol container implementations must implement.
public protocol Container {
/// ID for the container.
var id: String { get }
/// The amount of cpus assigned to the container.
var cpus: Int { get }
/// The memory in bytes assigned to the container.
var memoryInBytes: UInt64 { get }
/// The network interfaces assigned to the container.
var interfaces: [any Interface] { get }
}
@@ -0,0 +1,390 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
#if os(macOS)
import ContainerizationError
import ContainerizationEXT4
import ContainerizationOCI
import ContainerizationOS
import Foundation
import ContainerizationExtras
import SystemPackage
import Virtualization
/// A manager for creating and running containers.
/// Supports container networking options.
public struct ContainerManager: Sendable {
public let imageStore: ImageStore
private let vmm: VirtualMachineManager
private var network: Network?
private var containerRoot: URL {
self.imageStore.path.appendingPathComponent("containers")
}
/// Create a new manager with the provided kernel, initfs mount, image store
/// and optional network implementation. This will use a Virtualization.framework
/// backed VMM implicitly.
public init(
kernel: Kernel,
initfs: Mount,
imageStore: ImageStore,
network: Network? = nil,
rosetta: Bool = false,
nestedVirtualization: Bool = false
) throws {
self.imageStore = imageStore
self.network = network
try Self.createRootDirectory(path: self.imageStore.path)
self.vmm = VZVirtualMachineManager(
kernel: kernel,
initialFilesystem: initfs,
rosetta: rosetta,
nestedVirtualization: nestedVirtualization
)
}
/// Create a new manager with the provided kernel, initfs mount, root state
/// directory and optional network implementation. This will use a Virtualization.framework
/// backed VMM implicitly.
public init(
kernel: Kernel,
initfs: Mount,
root: URL? = nil,
network: Network? = nil,
rosetta: Bool = false,
nestedVirtualization: Bool = false
) throws {
if let root {
self.imageStore = try ImageStore(path: root)
} else {
self.imageStore = ImageStore.default
}
self.network = network
try Self.createRootDirectory(path: self.imageStore.path)
self.vmm = VZVirtualMachineManager(
kernel: kernel,
initialFilesystem: initfs,
rosetta: rosetta,
nestedVirtualization: nestedVirtualization
)
}
/// Create a new manager with the provided kernel, initfs reference, image store
/// and optional network implementation. This will use a Virtualization.framework
/// backed VMM implicitly.
public init(
kernel: Kernel,
initfsReference: String,
imageStore: ImageStore,
network: Network? = nil,
rosetta: Bool = false,
nestedVirtualization: Bool = false
) async throws {
self.imageStore = imageStore
self.network = network
try Self.createRootDirectory(path: self.imageStore.path)
let initPath = self.imageStore.path.appendingPathComponent("initfs.ext4")
let initImage = try await self.imageStore.getInitImage(reference: initfsReference)
let initfs = try await {
do {
return try await initImage.initBlock(at: initPath, for: .linuxArm)
} catch let err as ContainerizationError {
guard err.code == .exists else {
throw err
}
return .block(
format: "ext4",
source: initPath.absolutePath(),
destination: "/",
options: ["ro"]
)
}
}()
self.vmm = VZVirtualMachineManager(
kernel: kernel,
initialFilesystem: initfs,
rosetta: rosetta,
nestedVirtualization: nestedVirtualization
)
}
/// Create a new manager with the provided kernel and image reference for the initfs.
/// This will use a Virtualization.framework backed VMM implicitly.
public init(
kernel: Kernel,
initfsReference: String,
root: URL? = nil,
network: Network? = nil,
rosetta: Bool = false,
nestedVirtualization: Bool = false
) async throws {
if let root {
self.imageStore = try ImageStore(path: root)
} else {
self.imageStore = ImageStore.default
}
self.network = network
try Self.createRootDirectory(path: self.imageStore.path)
let initPath = self.imageStore.path.appendingPathComponent("initfs.ext4")
let initImage = try await self.imageStore.getInitImage(reference: initfsReference)
let initfs = try await {
do {
return try await initImage.initBlock(at: initPath, for: .linuxArm)
} catch let err as ContainerizationError {
guard err.code == .exists else {
throw err
}
return .block(
format: "ext4",
source: initPath.absolutePath(),
destination: "/",
options: ["ro"]
)
}
}()
self.vmm = VZVirtualMachineManager(
kernel: kernel,
initialFilesystem: initfs,
rosetta: rosetta,
nestedVirtualization: nestedVirtualization
)
}
/// Create a new manager with the provided vmm and network.
public init(
vmm: any VirtualMachineManager,
network: Network? = nil
) throws {
self.imageStore = ImageStore.default
try Self.createRootDirectory(path: self.imageStore.path)
self.network = network
self.vmm = vmm
}
private static func createRootDirectory(path: URL) throws {
try FileManager.default.createDirectory(
at: path.appendingPathComponent("containers"),
withIntermediateDirectories: true
)
}
/// Returns a new container from the provided image reference.
/// - Parameters:
/// - id: The container ID.
/// - reference: The image reference.
/// - rootfsSizeInBytes: The size of the root filesystem in bytes. Defaults to 8 GiB.
/// - writableLayerSizeInBytes: Optional size for a separate writable layer. When provided,
/// the rootfs becomes read-only and an overlayfs is used with a separate writable layer of this size.
/// - readOnly: Whether to mount the root filesystem as read-only.
/// - networking: Whether to create a network interface for this container. Defaults to `true`.
/// When `false`, no network resources are allocated and `releaseNetwork`/`delete` remain safe to call.
/// - progress: Optional handler for tracking rootfs unpacking progress.
public mutating func create(
_ id: String,
reference: String,
rootfsSizeInBytes: UInt64 = 8.gib(),
writableLayerSizeInBytes: UInt64? = nil,
readOnly: Bool = false,
networking: Bool = true,
progress: ProgressHandler? = nil,
configuration: (inout LinuxContainer.Configuration) throws -> Void
) async throws -> LinuxContainer {
let image = try await imageStore.get(reference: reference, pull: true)
return try await create(
id,
image: image,
rootfsSizeInBytes: rootfsSizeInBytes,
writableLayerSizeInBytes: writableLayerSizeInBytes,
readOnly: readOnly,
networking: networking,
progress: progress,
configuration: configuration
)
}
/// Returns a new container from the provided image.
/// - Parameters:
/// - id: The container ID.
/// - image: The image.
/// - rootfsSizeInBytes: The size of the root filesystem in bytes. Defaults to 8 GiB.
/// - writableLayerSizeInBytes: Optional size for a separate writable layer. When provided,
/// the rootfs becomes read-only and an overlayfs is used with a separate writable layer of this size.
/// - readOnly: Whether to mount the root filesystem as read-only.
/// - networking: Whether to create a network interface for this container. Defaults to `true`.
/// When `false`, no network resources are allocated and `releaseNetwork`/`delete` remain safe to call.
/// - progress: Optional handler for tracking rootfs unpacking progress.
public mutating func create(
_ id: String,
image: Image,
rootfsSizeInBytes: UInt64 = 8.gib(),
writableLayerSizeInBytes: UInt64? = nil,
readOnly: Bool = false,
networking: Bool = true,
progress: ProgressHandler? = nil,
configuration: (inout LinuxContainer.Configuration) throws -> Void
) async throws -> LinuxContainer {
let path = try createContainerRoot(id)
var rootfs = try await unpack(
image: image,
destination: path.appendingPathComponent("rootfs.ext4"),
size: rootfsSizeInBytes,
progress: progress
)
if readOnly {
rootfs.options.append("ro")
}
// Create writable layer if size is specified.
var writableLayer: Mount? = nil
if let writableLayerSize = writableLayerSizeInBytes {
writableLayer = try createEmptyFilesystem(
at: path.appendingPathComponent("writable.ext4"),
size: writableLayerSize
)
}
return try await create(
id,
image: image,
rootfs: rootfs,
writableLayer: writableLayer,
networking: networking,
configuration: configuration
)
}
/// Returns a new container from the provided image and root filesystem mount.
/// - Parameters:
/// - id: The container ID.
/// - image: The image.
/// - rootfs: The root filesystem mount pointing to an existing block file.
/// The `destination` field is ignored as mounting is handled internally.
/// - writableLayer: Optional writable layer mount. When provided, an overlayfs is used with
/// rootfs as the lower layer and this as the upper layer.
/// The `destination` field is ignored as mounting is handled internally.
/// - networking: Whether to create a network interface for this container. Defaults to `true`.
/// When `false`, no network resources are allocated and `releaseNetwork`/`delete` remain safe to call.
public mutating func create(
_ id: String,
image: Image,
rootfs: Mount,
writableLayer: Mount? = nil,
networking: Bool = true,
configuration: (inout LinuxContainer.Configuration) throws -> Void
) async throws -> LinuxContainer {
let imageConfig = try await image.config(for: .current).config
return try LinuxContainer(
id,
rootfs: rootfs,
writableLayer: writableLayer,
vmm: self.vmm
) { config in
if let imageConfig {
config.process = .init(from: imageConfig)
}
if networking {
if let interface = try self.network?.createInterface(id) {
config.interfaces = [interface]
guard let gateway = interface.ipv4Gateway else {
throw ContainerizationError(
.invalidState,
message: "missing ipv4 gateway for container \(id)"
)
}
config.dns = .init(nameservers: [gateway.description])
}
}
config.bootLog = BootLog.file(path: self.containerRoot.appendingPathComponent(id).appendingPathComponent("bootlog.log"))
try configuration(&config)
}
}
/// Releases network resources for a container.
///
/// - Parameter id: The container ID.
public mutating func releaseNetwork(_ id: String) throws {
try self.network?.releaseInterface(id)
}
/// Releases network resources and removes all files for a container.
/// - Parameter id: The container ID.
public mutating func delete(_ id: String) throws {
try self.releaseNetwork(id)
let path = containerRoot.appendingPathComponent(id)
try FileManager.default.removeItem(at: path)
}
private func createContainerRoot(_ id: String) throws -> URL {
let path = containerRoot.appendingPathComponent(id)
try FileManager.default.createDirectory(at: path, withIntermediateDirectories: false)
return path
}
private func unpack(image: Image, destination: URL, size: UInt64, progress: ProgressHandler? = nil) async throws -> Mount {
do {
let unpacker = EXT4Unpacker(blockSizeInBytes: size)
return try await unpacker.unpack(image, for: .current, at: destination, progress: progress)
} catch let err as ContainerizationError {
if err.code == .exists {
return .block(
format: "ext4",
source: destination.absolutePath(),
destination: "/",
options: []
)
}
throw err
}
}
private func createEmptyFilesystem(at destination: URL, size: UInt64) throws -> Mount {
let path = destination.absolutePath()
guard !FileManager.default.fileExists(atPath: path) else {
throw ContainerizationError(.exists, message: "filesystem already exists at \(path)")
}
let filesystem = try EXT4.Formatter(FilePath(path), minDiskSize: size)
try filesystem.close()
return .block(
format: "ext4",
source: path,
destination: "/",
options: []
)
}
}
extension CIDRv4 {
/// The gateway address of the network.
public var gateway: IPv4Address {
IPv4Address(self.lower.value + 1)
}
}
extension CIDRv6 {
/// The gateway address of the network.
public var gateway: IPv6Address {
IPv6Address(self.lower.value + 1)
}
}
#endif
@@ -0,0 +1,248 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
/// Statistics for a container.
public struct ContainerStatistics: Sendable {
public var id: String
public var process: ProcessStatistics?
public var memory: MemoryStatistics?
public var cpu: CPUStatistics?
public var blockIO: BlockIOStatistics?
public var networks: [NetworkStatistics]?
public var memoryEvents: MemoryEventStatistics?
public init(
id: String,
process: ProcessStatistics? = nil,
memory: MemoryStatistics? = nil,
cpu: CPUStatistics? = nil,
blockIO: BlockIOStatistics? = nil,
networks: [NetworkStatistics]? = nil,
memoryEvents: MemoryEventStatistics? = nil
) {
self.id = id
self.process = process
self.memory = memory
self.cpu = cpu
self.blockIO = blockIO
self.networks = networks
self.memoryEvents = memoryEvents
}
/// Process statistics for a container.
public struct ProcessStatistics: Sendable {
public var current: UInt64
public var limit: UInt64
public init(current: UInt64, limit: UInt64) {
self.current = current
self.limit = limit
}
}
/// Memory statistics for a container.
public struct MemoryStatistics: Sendable {
public var usageBytes: UInt64
public var limitBytes: UInt64
public var swapUsageBytes: UInt64
public var swapLimitBytes: UInt64
public var cacheBytes: UInt64
public var kernelStackBytes: UInt64
public var slabBytes: UInt64
public var pageFaults: UInt64
public var majorPageFaults: UInt64
public var inactiveFile: UInt64
public var anon: UInt64
public var workingsetRefaultAnon: UInt64
public var workingsetRefaultFile: UInt64
public var pgstealKswapd: UInt64
public var pgstealDirect: UInt64
public var pgstealKhugepaged: UInt64
public init(
usageBytes: UInt64,
limitBytes: UInt64,
swapUsageBytes: UInt64,
swapLimitBytes: UInt64,
cacheBytes: UInt64,
kernelStackBytes: UInt64,
slabBytes: UInt64,
pageFaults: UInt64,
majorPageFaults: UInt64,
inactiveFile: UInt64,
anon: UInt64,
workingsetRefaultAnon: UInt64 = 0,
workingsetRefaultFile: UInt64 = 0,
pgstealKswapd: UInt64 = 0,
pgstealDirect: UInt64 = 0,
pgstealKhugepaged: UInt64 = 0
) {
self.usageBytes = usageBytes
self.limitBytes = limitBytes
self.swapUsageBytes = swapUsageBytes
self.swapLimitBytes = swapLimitBytes
self.cacheBytes = cacheBytes
self.kernelStackBytes = kernelStackBytes
self.slabBytes = slabBytes
self.pageFaults = pageFaults
self.majorPageFaults = majorPageFaults
self.inactiveFile = inactiveFile
self.anon = anon
self.workingsetRefaultAnon = workingsetRefaultAnon
self.workingsetRefaultFile = workingsetRefaultFile
self.pgstealKswapd = pgstealKswapd
self.pgstealDirect = pgstealDirect
self.pgstealKhugepaged = pgstealKhugepaged
}
}
/// CPU statistics for a container.
public struct CPUStatistics: Sendable {
public var usageUsec: UInt64
public var userUsec: UInt64
public var systemUsec: UInt64
public var throttlingPeriods: UInt64
public var throttledPeriods: UInt64
public var throttledTimeUsec: UInt64
public init(
usageUsec: UInt64,
userUsec: UInt64,
systemUsec: UInt64,
throttlingPeriods: UInt64,
throttledPeriods: UInt64,
throttledTimeUsec: UInt64
) {
self.usageUsec = usageUsec
self.userUsec = userUsec
self.systemUsec = systemUsec
self.throttlingPeriods = throttlingPeriods
self.throttledPeriods = throttledPeriods
self.throttledTimeUsec = throttledTimeUsec
}
}
/// Block I/O statistics for a container.
public struct BlockIOStatistics: Sendable {
public var devices: [BlockIODevice]
public init(devices: [BlockIODevice]) {
self.devices = devices
}
}
/// Block I/O statistics for a specific device.
public struct BlockIODevice: Sendable {
public var major: UInt64
public var minor: UInt64
public var readBytes: UInt64
public var writeBytes: UInt64
public var readOperations: UInt64
public var writeOperations: UInt64
public init(
major: UInt64,
minor: UInt64,
readBytes: UInt64,
writeBytes: UInt64,
readOperations: UInt64,
writeOperations: UInt64
) {
self.major = major
self.minor = minor
self.readBytes = readBytes
self.writeBytes = writeBytes
self.readOperations = readOperations
self.writeOperations = writeOperations
}
}
/// Statistics for a network interface.
public struct NetworkStatistics: Sendable {
public var interface: String
public var receivedPackets: UInt64
public var transmittedPackets: UInt64
public var receivedBytes: UInt64
public var transmittedBytes: UInt64
public var receivedErrors: UInt64
public var transmittedErrors: UInt64
public init(
interface: String,
receivedPackets: UInt64,
transmittedPackets: UInt64,
receivedBytes: UInt64,
transmittedBytes: UInt64,
receivedErrors: UInt64,
transmittedErrors: UInt64
) {
self.interface = interface
self.receivedPackets = receivedPackets
self.transmittedPackets = transmittedPackets
self.receivedBytes = receivedBytes
self.transmittedBytes = transmittedBytes
self.receivedErrors = receivedErrors
self.transmittedErrors = transmittedErrors
}
}
/// Memory event counters from cgroup2's memory.events file.
public struct MemoryEventStatistics: Sendable {
/// Number of times the cgroup was reclaimed due to low memory.
public var low: UInt64
/// Number of times the cgroup exceeded its high memory limit.
public var high: UInt64
/// Number of times the cgroup hit its max memory limit.
public var max: UInt64
/// Number of times the cgroup triggered OOM.
public var oom: UInt64
/// Number of processes killed by OOM killer.
public var oomKill: UInt64
public init(low: UInt64, high: UInt64, max: UInt64, oom: UInt64, oomKill: UInt64) {
self.low = low
self.high = high
self.max = max
self.oom = oom
self.oomKill = oomKill
}
}
}
/// Categories of statistics that can be requested.
public struct StatCategory: OptionSet, Sendable {
public let rawValue: Int
public init(rawValue: Int) {
self.rawValue = rawValue
}
/// Process statistics (pids.current, pids.max).
public static let process = StatCategory(rawValue: 1 << 0)
/// Memory usage statistics.
public static let memory = StatCategory(rawValue: 1 << 1)
/// CPU usage statistics.
public static let cpu = StatCategory(rawValue: 1 << 2)
/// Block I/O statistics.
public static let blockIO = StatCategory(rawValue: 1 << 3)
/// Network interface statistics.
public static let network = StatCategory(rawValue: 1 << 4)
/// Memory event counters (OOM kills, pressure events, etc.).
public static let memoryEvents = StatCategory(rawValue: 1 << 5)
/// All available statistics categories.
public static let all: StatCategory = [.process, .memory, .cpu, .blockIO, .network, .memoryEvents]
}
@@ -0,0 +1,91 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationError
import ContainerizationExtras
/// DNS configuration for a container. The values will be used to
/// construct /etc/resolv.conf for a given container.
public struct DNS: Sendable {
/// The set of default nameservers to use if none are provided
/// in the constructor.
public static let defaultNameservers = ["1.1.1.1"]
/// The nameservers a container should use.
public var nameservers: [String]
/// The DNS domain to use.
public var domain: String?
/// The DNS search domains to use.
public var searchDomains: [String]
/// The DNS options to use.
public var options: [String]
public init(
nameservers: [String] = defaultNameservers,
domain: String? = nil,
searchDomains: [String] = [],
options: [String] = []
) {
self.nameservers = nameservers
self.domain = domain
self.searchDomains = searchDomains
self.options = options
}
/// Validates the DNS configuration.
///
/// Ensures that all nameserver entries are valid IPv4 or IPv6 addresses.
/// Arbitrary hostnames are not permitted as nameservers.
///
/// - Throws: ``ContainerizationError`` with code `.invalidArgument` if
/// any nameserver is not a valid IP address.
public func validate() throws {
for nameserver in nameservers {
let isValidIPv4 = (try? IPv4Address(nameserver)) != nil
let isValidIPv6 = (try? IPv6Address(nameserver)) != nil
if !isValidIPv4 && !isValidIPv6 {
throw ContainerizationError(
.invalidArgument,
message: "nameserver '\(nameserver)' is not a valid IPv4 or IPv6 address"
)
}
}
}
}
extension DNS {
public var resolvConf: String {
var text = ""
if !nameservers.isEmpty {
text += nameservers.map { "nameserver \($0)" }.joined(separator: "\n") + "\n"
}
if let domain {
text += "domain \(domain)\n"
}
if !searchDomains.isEmpty {
text += "search \(searchDomains.joined(separator: " "))\n"
}
if !options.isEmpty {
text += "options \(options.joined(separator: " "))\n"
}
return text
}
}
+36
View File
@@ -0,0 +1,36 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import Foundation
/// ExitStatus contains the exit code for a given container process,
/// as well as the timestamp at which it exited.
public struct ExitStatus: Sendable {
/// The exit code for the process.
public var exitCode: Int32
/// The timestamp when the process exited.
public var exitedAt: Date
public init(exitCode: Int32) {
self.exitCode = exitCode
self.exitedAt = .now
}
public init(exitCode: Int32, exitedAt: Date) {
self.exitCode = exitCode
self.exitedAt = exitedAt
}
}
+202
View File
@@ -0,0 +1,202 @@
//===----------------------------------------------------------------------===//
// Copyright © 2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationError
import ContainerizationOCI
import Foundation
/// Manages single-file mounts by transforming them into virtiofs directory shares
/// plus bind mounts.
///
/// Since virtiofs only supports sharing directories, mounting a single file requires
/// sharing the file's parent directory via virtiofs and then bind mounting the specific
/// file from that share to the final destination in the container.
struct FileMountContext: Sendable {
/// Metadata for a single prepared file mount.
struct PreparedMount: Sendable {
/// Original file path on host
let hostFilePath: String
/// Where the user wants the file in the container
let containerDestination: String
/// Just the filename (after resolving symlinks)
let filename: String
/// The parent directory containing the file (after resolving symlinks)
let parentDirectory: URL
/// The virtiofs tag (hash of parent dir path). Used to find the AttachedFilesystem
let tag: String
/// Mount options from the original mount
let options: [String]
/// Where we mounted the share in the guest (set after mountHoldingDirectories)
var guestHoldingPath: String?
}
/// Prepared file mounts for this context
var preparedMounts: [PreparedMount]
/// The transformed mounts to pass to the VM (files replaced with directory shares)
private(set) var transformedMounts: [Mount]
private init() {
self.preparedMounts = []
self.transformedMounts = []
}
/// Returns true if there are any file mounts that need handling.
var hasFileMounts: Bool {
!preparedMounts.isEmpty
}
/// Returns the set of virtiofs tags for file mount holding directories.
/// These should be filtered out from OCI spec mounts since we mount them
/// separately under /run.
var holdingDirectoryTags: Set<String> {
Set(preparedMounts.map { $0.tag })
}
}
extension FileMountContext {
/// Prepare mounts for a container, detecting file mounts and transforming them.
///
/// This method stats each virtiofs mount source. If it's a regular file rather than
/// a directory, it shares the file's parent directory via virtiofs and records the
/// metadata needed to bind mount the specific file later.
///
/// - Parameter mounts: The original mounts from the container config
/// - Returns: A FileMountContext containing transformed mounts and tracking info
static func prepare(mounts: [Mount]) throws -> FileMountContext {
var context = FileMountContext()
var transformed: [Mount] = []
// Track parent directories we've already added a share for to avoid duplicates.
var sharedParentTags: Set<String> = []
for mount in mounts {
// Only virtiofs mounts can be files
guard case .virtiofs(let runtimeOpts) = mount.runtimeOptions else {
transformed.append(mount)
continue
}
// Stat the source to see if it's a file
let fm = FileManager.default
var isDirectory: ObjCBool = false
guard fm.fileExists(atPath: mount.source, isDirectory: &isDirectory) else {
// Doesn't exist. Let the normal flow handle the error
transformed.append(mount)
continue
}
if isDirectory.boolValue {
// It's a directory, pass through unchanged
transformed.append(mount)
continue
}
// It's a file, so prepare it.
let prepared = try context.prepareFileMount(mount: mount, runtimeOptions: runtimeOpts)
// Only add the directory share once per unique parent directory.
if !sharedParentTags.contains(prepared.tag) {
sharedParentTags.insert(prepared.tag)
// The destination here is unused. We mount the share ourselves
// to a location under /run in mountHoldingDirectories.
let directoryShare = Mount.share(
source: prepared.parentDirectory.path,
destination: "/.file-mount-holding",
options: mount.options.filter { $0 != "bind" },
runtimeOptions: runtimeOpts
)
transformed.append(directoryShare)
}
}
context.transformedMounts = transformed
return context
}
private mutating func prepareFileMount(
mount: Mount,
runtimeOptions: [String]
) throws -> PreparedMount {
let resolvedSource = URL(fileURLWithPath: mount.source).resolvingSymlinksInPath()
let filename = resolvedSource.lastPathComponent
let parentDirectory = resolvedSource.deletingLastPathComponent()
let tag = try hashFilePath(path: parentDirectory.path)
let prepared = PreparedMount(
hostFilePath: mount.source,
containerDestination: mount.destination,
filename: filename,
parentDirectory: parentDirectory,
tag: tag,
options: mount.options,
guestHoldingPath: nil
)
preparedMounts.append(prepared)
return prepared
}
}
extension FileMountContext {
/// Set up the holding directory paths for all file mounts.
/// Since virtiofs shares are now mounted once at /run/virtiofs, the holding
/// directories appear as subdirectories there automatically.
/// - Parameters:
/// - vmMounts: The AttachedFilesystem array from the VM for this container
/// - agent: The VM agent for RPCs (unused, kept for API compatibility)
mutating func mountHoldingDirectories(
vmMounts: [AttachedFilesystem],
agent: any VirtualMachineAgent
) async throws {
for i in preparedMounts.indices {
let prepared = preparedMounts[i]
// Verify the attached filesystem exists
guard
vmMounts.first(where: {
$0.type == "virtiofs" && $0.source == prepared.tag
}) != nil
else {
throw ContainerizationError(
.notFound,
message: "could not find attached filesystem for file mount \(prepared.hostFilePath)"
)
}
// With unified virtiofs, holding directories are subdirectories under /run/virtiofs
let guestPath = "/run/virtiofs/\(prepared.tag)"
preparedMounts[i].guestHoldingPath = guestPath
}
}
}
extension FileMountContext {
/// Get the bind mounts to append to the OCI spec.
func ociBindMounts() -> [ContainerizationOCI.Mount] {
preparedMounts.compactMap { prepared in
guard let guestPath = prepared.guestHoldingPath else {
return nil
}
return ContainerizationOCI.Mount(
type: "none",
source: "\(guestPath)/\(prepared.filename)",
destination: prepared.containerDestination,
options: ["bind"] + prepared.options
)
}
}
}
@@ -0,0 +1,90 @@
//===----------------------------------------------------------------------===//
// Copyright © 2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationError
import GRPCCore
import GRPCNIOTransportCore
import NIOCore
import NIOPosix
/// Buffers incoming bytes until the full gRPC HTTP/2 pipeline is configured, then replays them.
///
/// This prevents the race condition where the vminitd server's initial HTTP/2 SETTINGS frame
/// arrives and is discarded before `configureGRPCClientPipeline` has finished installing
/// `ClientConnectionHandler`.
///
/// The handler is added via `ClientBootstrap.channelInitializer`, which runs before
/// `registerAlreadyConfigured0` adds the fd to epoll/kqueue — guaranteeing it is in place
/// before any bytes can arrive on the socket.
///
/// When `NIOHTTP2Handler` is added to the pipeline (inside `configureGRPCClientPipeline`), its
/// `handlerAdded` fires an outbound flush (the HTTP/2 client preface). We intercept that flush
/// and schedule a deferred removal via the event loop. Because `configureGRPCClientPipeline` runs
/// as a single synchronous event loop task, the deferred removal is guaranteed to run after that
/// entire task completes — i.e., after `ClientConnectionHandler` is also in the pipeline.
/// Buffered bytes are replayed atomically as part of the pipeline removal.
// FIXME: This handler is needed until the swift GRPC libraries offers us a way to create a
// client transport from an existing fd. Remove this type when such an API exists.
public final class HTTP2ConnectBufferingHandler: ChannelDuplexHandler, RemovableChannelHandler {
public typealias InboundIn = ByteBuffer
public typealias InboundOut = ByteBuffer
public typealias OutboundIn = ByteBuffer
public typealias OutboundOut = ByteBuffer
private var removalScheduled = false
private var bufferedReads: [NIOAny] = []
public init() {}
public func channelRead(context: ChannelHandlerContext, data: NIOAny) {
bufferedReads.append(data)
}
public func channelReadComplete(context: ChannelHandlerContext) {
// Suppress while buffering; a single readComplete is emitted after replay.
}
public func flush(context: ChannelHandlerContext) {
if !removalScheduled {
removalScheduled = true
// Defer removal to the next event loop task. configureGRPCClientPipeline runs as a
// single synchronous event loop task, so this deferred task is guaranteed to run
// after that whole task completes (including ClientConnectionHandler being added).
context.eventLoop.assumeIsolatedUnsafeUnchecked().execute {
context.pipeline.syncOperations.removeHandler(self, promise: nil)
}
}
context.flush()
}
public func removeHandler(context: ChannelHandlerContext, removalToken: ChannelHandlerContext.RemovalToken) {
var didRead = false
while !bufferedReads.isEmpty {
context.fireChannelRead(bufferedReads.removeFirst())
didRead = true
}
if didRead {
context.fireChannelReadComplete()
}
context.leavePipeline(removalToken: removalToken)
}
public func channelInactive(context: ChannelHandlerContext) {
bufferedReads.removeAll()
context.fireChannelInactive()
}
}
+40
View File
@@ -0,0 +1,40 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationError
import Crypto
import Foundation
extension Mount {
/// A deterministic hash of the mount's source path, used as the virtiofs tag.
///
/// Resolves symlinks before hashing so that different paths to the same
/// directory produce an identical tag.
public var tagHash: String {
get throws {
try hashFilePath(path: self.source)
}
}
}
func hashFilePath(path: String) throws -> String {
// Resolve symlinks so different paths to the same directory get the same hash.
let resolvedSource = URL(fileURLWithPath: path).resolvingSymlinksInPath().path
guard let data = resolvedSource.data(using: .utf8) else {
throw ContainerizationError(.invalidArgument, message: "\(path) could not be converted to Data")
}
return String(SHA256.hash(data: data).encoded.prefix(36))
}
@@ -0,0 +1,140 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
/// Static table lookups for a container. The values will be used to
/// construct /etc/hosts for a given container.
public struct Hosts: Sendable {
/// Represents one entry in an /etc/hosts file.
public struct Entry: Sendable {
/// The IPV4 or IPV6 address in String form.
public var ipAddress: String
/// The hostname(s) for the entry.
public var hostnames: [String]
/// An optional comment to be placed to the right side of the entry.
public var comment: String?
public init(ipAddress: String, hostnames: [String], comment: String? = nil) {
self.comment = comment
self.hostnames = hostnames
self.ipAddress = ipAddress
}
/// The information in the structure rendered to a String representation
/// that matches the format /etc/hosts expects.
public var rendered: String {
var line = ipAddress
if !hostnames.isEmpty {
line += " " + hostnames.joined(separator: " ")
}
if let comment {
line += " # \(comment) "
}
return line
}
public static func localHostIPV4(comment: String? = nil) -> Self {
Self(
ipAddress: "127.0.0.1",
hostnames: ["localhost"],
comment: comment
)
}
public static func localHostIPV6(comment: String? = nil) -> Self {
Self(
ipAddress: "::1",
hostnames: ["localhost", "ip6-localhost", "ip6-loopback"],
comment: comment
)
}
public static func ipv6LocalNet(comment: String? = nil) -> Self {
Self(
ipAddress: "fe00::",
hostnames: ["ip6-localnet"],
comment: comment
)
}
public static func ipv6MulticastPrefix(comment: String? = nil) -> Self {
Self(
ipAddress: "ff00::",
hostnames: ["ip6-mcastprefix"],
comment: comment
)
}
public static func ipv6AllNodes(comment: String? = nil) -> Self {
Self(
ipAddress: "ff02::1",
hostnames: ["ip6-allnodes"],
comment: comment
)
}
public static func ipv6AllRouters(comment: String? = nil) -> Self {
Self(
ipAddress: "ff02::2",
hostnames: ["ip6-allrouters"],
comment: comment
)
}
}
/// The entries to be written to /etc/hosts.
public var entries: [Entry]
/// A comment to render at the top of the file.
public var comment: String?
public init(
entries: [Entry],
comment: String? = nil
) {
self.entries = entries
self.comment = comment
}
}
extension Hosts {
/// A default entry that can be used for convenience. It contains a IPV4
/// and IPV6 localhost entry, as well as ipv6 localnet, ipv6 mcastprefix,
/// ipv6 allnodes, and ipv6 allrouters.
public static let `default` = Hosts(entries: [
Entry.localHostIPV4(),
Entry.localHostIPV6(),
Entry.ipv6LocalNet(),
Entry.ipv6MulticastPrefix(),
Entry.ipv6AllNodes(),
Entry.ipv6AllRouters(),
])
/// Returns a string variant of the data that can be written to
/// /etc/hosts directly.
public var hostsFile: String {
var lines: [String] = []
if let comment {
lines.append("# \(comment)")
}
for entry in entries {
lines.append(entry.rendered)
}
return lines.joined(separator: "\n") + "\n"
}
}
@@ -0,0 +1,56 @@
//===----------------------------------------------------------------------===//
// Copyright © 2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
/// A provider that manages hotplug operations for a virtual machine instance.
///
/// Conforming types implement the mechanics of hotplugging block devices and
/// virtiofs shares into a running VM.
public protocol HotplugProvider: Sendable {
/// Hotplug a block device into the running VM.
/// - Parameters:
/// - block: The mount configuration for the block device
/// - id: The container ID to associate with this device
/// - Returns: The attached filesystem with the device path in the guest
func hotplug(_ block: Mount, id: String) async throws -> AttachedFilesystem
/// Register mounts for a container in the VM's mount registry.
/// - Parameters:
/// - id: The container ID
/// - rootfs: The rootfs attachment from hotplug
/// - additionalMounts: Additional mounts to register
func registerMounts(id: String, rootfs: AttachedFilesystem, additionalMounts: [Mount]) throws
/// Release a hotplug device.
/// - Parameter id: The container ID who should be released
func releaseHotplug(id: String) async throws
/// Hotplug virtiofs directories into the running VM.
/// - Parameters:
/// - mounts: The virtiofs mounts to add
/// - id: The container ID that owns these mounts
func hotplugVirtioFS(_ mounts: [Mount], id: String) async throws
/// Release virtiofs shares for a container.
/// - Parameter id: The container ID whose shares should be released
func releaseVirtioFS(id: String) async throws
/// Clean up resources held by the provider.
func cleanup()
}
extension HotplugProvider {
public func cleanup() {}
}
@@ -0,0 +1,22 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import Foundation
/// A type that returns a stream of Data.
public protocol ReaderStream: Sendable {
func stream() -> AsyncStream<Data>
}
@@ -0,0 +1,36 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationOS
import Foundation
extension Terminal: ReaderStream {
public func stream() -> AsyncStream<Data> {
.init { cont in
self.handle.readabilityHandler = { handle in
let data = handle.availableData
if data.isEmpty {
self.handle.readabilityHandler = nil
cont.finish()
return
}
cont.yield(data)
}
}
}
}
extension Terminal: Writer {}
+23
View File
@@ -0,0 +1,23 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import Foundation
/// A type that writes the provided Data.
public protocol Writer: Sendable {
func write(_ data: Data) throws
func close() throws
}
+130
View File
@@ -0,0 +1,130 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationError
import ContainerizationOCI
import ContainerizationOS
import Foundation
/// Type representing an OCI container image.
public struct Image: Sendable {
private let contentStore: ContentStore
/// The description for the image that comprises of its name and a reference to its root descriptor.
public let description: Description
/// A description of the OCI image.
public struct Description: Sendable {
/// The string reference of the image.
public let reference: String
/// The descriptor identifying the image.
public let descriptor: Descriptor
/// The digest for the image.
public var digest: String { descriptor.digest }
/// The media type of the image.
public var mediaType: String { descriptor.mediaType }
public init(reference: String, descriptor: Descriptor) {
self.reference = reference
self.descriptor = descriptor
}
}
/// The descriptor for the image.
public var descriptor: Descriptor { description.descriptor }
/// The digest of the image.
public var digest: String { description.digest }
/// The media type of the image.
public var mediaType: String { description.mediaType }
/// The string reference for the image.
public var reference: String { description.reference }
public init(description: Description, contentStore: ContentStore) {
self.description = description
self.contentStore = contentStore
}
/// Returns the underlying OCI index for the image.
public func index() async throws -> Index {
guard let content: Content = try await contentStore.get(digest: digest) else {
throw ContainerizationError(.notFound, message: "content with digest \(digest)")
}
return try content.decode()
}
/// Returns the manifest for the specified platform.
public func manifest(for platform: Platform) async throws -> Manifest {
let index = try await self.index()
let desc = index.manifests.first { desc in
desc.platform == platform
}
guard let desc else {
throw ContainerizationError(.unsupported, message: "platform \(platform.description)")
}
guard let content: Content = try await contentStore.get(digest: desc.digest) else {
throw ContainerizationError(.notFound, message: "content with digest \(digest)")
}
return try content.decode()
}
/// Returns the descriptor for the given platform. If it does not exist
/// will throw a ContainerizationError with the code set to .invalidArgument.
public func descriptor(for platform: Platform) async throws -> Descriptor {
let index = try await self.index()
let desc = index.manifests.first { $0.platform == platform }
guard let desc else {
throw ContainerizationError(.invalidArgument, message: "unsupported platform \(platform)")
}
return desc
}
/// Returns the OCI config for the specified platform.
public func config(for platform: Platform) async throws -> ContainerizationOCI.Image {
let manifest = try await self.manifest(for: platform)
let desc = manifest.config
guard let content: Content = try await contentStore.get(digest: desc.digest) else {
throw ContainerizationError(.notFound, message: "content with digest \(digest)")
}
return try content.decode()
}
/// Returns a list of digests to all the referenced OCI objects.
public func referencedDigests() async throws -> [String] {
var referenced: [String] = [self.digest.trimmingDigestPrefix]
let index = try await self.index()
for manifest in index.manifests {
referenced.append(manifest.digest.trimmingDigestPrefix)
guard let m: Manifest = try? await contentStore.get(digest: manifest.digest) else {
// If the requested digest does not exist or is not a manifest. Skip.
// It's safe to skip processing this digest as it won't have any child layers.
continue
}
let descs = m.layers + [m.config]
referenced.append(contentsOf: descs.map { $0.digest.trimmingDigestPrefix })
}
return referenced
}
/// Returns a reference to the content blob for the image. The specified digest must be referenced by the image in one of its layers.
public func getContent(digest: String) async throws -> Content {
guard try await self.referencedDigests().contains(digest.trimmingDigestPrefix) else {
throw ContainerizationError(.internalError, message: "image \(self.reference) does not reference digest \(digest)")
}
guard let content: Content = try await contentStore.get(digest: digest) else {
throw ContainerizationError(.notFound, message: "content with digest \(digest)")
}
return content
}
}
@@ -0,0 +1,179 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
//
import ContainerizationError
import ContainerizationExtras
import ContainerizationIO
import ContainerizationOCI
import Crypto
import Foundation
extension ImageStore {
public struct ExportOperation: Sendable {
let name: String
let tag: String
let contentStore: ContentStore
let client: ContentClient
let progress: ProgressHandler?
public init(name: String, tag: String, contentStore: ContentStore, client: ContentClient, progress: ProgressHandler? = nil) {
self.contentStore = contentStore
self.client = client
self.progress = progress
self.name = name
self.tag = tag
}
@discardableResult
public func export(index: Descriptor, platforms: (Platform) -> Bool, filter: (Descriptor) -> Bool = { _ in true }) async throws -> Descriptor {
var pushQueue: [[Descriptor]] = []
var current: [Descriptor] = [index]
while !current.isEmpty {
let children = try await self.getChildren(descs: current)
let matches = try filterPlatforms(matcher: platforms, children).uniqued { $0.digest }
pushQueue.append(matches)
current = matches
}
let localIndexData = try await self.createIndex(from: index, matching: platforms)
await updatePushProgress(pushQueue: pushQueue, localIndexData: localIndexData)
// We need to work bottom up when pushing an image.
// First, the tar blobs / config layers, then, the manifests and so on...
// When processing a given "level", the requests maybe made in parallel.
// We need to ensure that the child level has been uploaded fully
// before uploading the parent level.
try await withThrowingTaskGroup(of: Void.self) { group in
for layerGroup in pushQueue.reversed() {
for chunk in layerGroup.chunks(ofCount: 8) {
for desc in chunk.filter(filter) {
guard let content = try await self.contentStore.get(digest: desc.digest) else {
throw ContainerizationError(.notFound, message: "content with digest \(desc.digest)")
}
group.addTask {
let readStream = try ReadStream(url: content.path)
try await self.pushContent(descriptor: desc, stream: readStream)
}
}
try await group.waitForAll()
}
}
}
// Lastly, we need to construct and push a new index, since we may
// have pushed content only for specific platforms.
let digest = SHA256.hash(data: localIndexData)
// The descriptor's mediaType becomes the HTTP Content-Type in
// RegistryClient.push and must match the mediaType field inside
// localIndexData. Registries reject mismatches with MANIFEST_INVALID.
let descriptor = Descriptor(
mediaType: index.mediaType,
digest: digest.digestString,
size: Int64(localIndexData.count))
let stream = ReadStream(data: localIndexData)
try await self.pushContent(descriptor: descriptor, stream: stream)
return descriptor
}
private func updatePushProgress(pushQueue: [[Descriptor]], localIndexData: Data) async {
for layerGroup in pushQueue {
for desc in layerGroup {
await progress?([
.addTotalSize(desc.size),
.addTotalItems(1),
])
}
}
await progress?([
.addTotalSize(Int64(localIndexData.count)),
.addTotalItems(1),
])
}
private func createIndex(from index: Descriptor, matching: (Platform) -> Bool) async throws -> Data {
guard let content = try await self.contentStore.get(digest: index.digest) else {
throw ContainerizationError(.notFound, message: "content with digest \(index.digest)")
}
var idx: Index = try content.decode()
let manifests = idx.manifests
var matchedManifests: [Descriptor] = []
var skippedPlatforms = false
for manifest in manifests {
guard let p = manifest.platform else {
continue
}
if matching(p) {
matchedManifests.append(manifest)
} else {
skippedPlatforms = true
}
}
if !skippedPlatforms {
return try content.data()
}
idx.manifests = matchedManifests
return try JSONEncoder().encode(idx)
}
private func pushContent(descriptor: Descriptor, stream: ReadStream) async throws {
do {
let generator = {
try stream.reset()
return stream.stream
}
try await client.push(name: name, ref: tag, descriptor: descriptor, streamGenerator: generator, progress: progress)
await progress?([
.addSize(descriptor.size),
.addItems(1),
])
} catch let err as ContainerizationError {
guard err.code != .exists else {
// We reported the total items and size and have to account for them in existing content.
await progress?([
.addSize(descriptor.size),
.addItems(1),
])
return
}
throw err
}
}
private func getChildren(descs: [Descriptor]) async throws -> [Descriptor] {
var out: [Descriptor] = []
for desc in descs {
let mediaType = desc.mediaType
guard let content = try await self.contentStore.get(digest: desc.digest) else {
throw ContainerizationError(.notFound, message: "content with digest \(desc.digest)")
}
switch mediaType {
case MediaTypes.index, MediaTypes.dockerManifestList:
let index: Index = try content.decode()
out.append(contentsOf: index.manifests)
case MediaTypes.imageManifest, MediaTypes.dockerManifest:
let manifest: Manifest = try content.decode()
out.append(manifest.config)
out.append(contentsOf: manifest.layers)
default:
continue
}
}
return out
}
}
}
@@ -0,0 +1,257 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
//
import ContainerizationError
import ContainerizationExtras
import ContainerizationOCI
import Foundation
extension ImageStore {
public struct ImportOperation: Sendable {
static let decoder = JSONDecoder()
let client: ContentClient
let ingestDir: URL
let contentStore: ContentStore
let progress: ProgressHandler?
let name: String
let maxConcurrentDownloads: Int
public init(name: String, contentStore: ContentStore, client: ContentClient, ingestDir: URL, progress: ProgressHandler? = nil, maxConcurrentDownloads: Int = 3) {
self.client = client
self.ingestDir = ingestDir
self.contentStore = contentStore
self.progress = progress
self.name = name
self.maxConcurrentDownloads = maxConcurrentDownloads
}
/// Pull the required image layers for the provided descriptor and platform(s) into the given directory using the provided client. Returns a descriptor to the Index manifest.
public func `import`(root: Descriptor, matcher: (ContainerizationOCI.Platform) -> Bool) async throws -> Descriptor {
var toProcess = [root]
while !toProcess.isEmpty {
// Count the total number of blobs and their size
if let progress {
var size: Int64 = 0
for desc in toProcess {
size += desc.size
}
await progress([
.addTotalSize(size),
.addTotalItems(toProcess.count),
])
}
try await self.fetchAll(toProcess)
let children = try await self.walk(toProcess)
let filtered = try filterPlatforms(matcher: matcher, children)
toProcess = filtered.uniqued { $0.digest }
}
guard root.mediaType != MediaTypes.dockerManifestList && root.mediaType != MediaTypes.index else {
return root
}
// Create an index for the root descriptor and write it to the content store
let index = try await self.createIndex(for: root)
// In cases where the root descriptor pointed to `MediaTypes.imageManifest`
// Or `MediaTypes.dockerManifest`, it is required that we check the supported platform
// matches the platforms we were asked to pull. This can be done only after we created
// the Index.
let supportedPlatforms = index.manifests.compactMap { $0.platform }
guard supportedPlatforms.allSatisfy(matcher) else {
throw ContainerizationError(.unsupported, message: "image \(root.digest) does not support required platforms")
}
let writer = try ContentWriter(for: self.ingestDir)
let result = try writer.create(from: index)
return Descriptor(
mediaType: MediaTypes.index,
digest: result.digest.digestString,
size: Int64(result.size))
}
private func getManifestContent<T: Sendable & Codable>(descriptor: Descriptor) async throws -> T {
do {
if let content = try await self.contentStore.get(digest: descriptor.digest.trimmingDigestPrefix) {
return try content.decode()
}
if let content = try? LocalContent(path: ingestDir.appending(path: descriptor.digest.trimmingDigestPrefix)) {
return try content.decode()
}
return try await self.client.fetch(name: name, descriptor: descriptor)
} catch {
throw ContainerizationError(.internalError, message: "cannot fetch content with digest \(descriptor.digest)", cause: error)
}
}
private func walk(_ descriptors: [Descriptor]) async throws -> [Descriptor] {
var out: [Descriptor] = []
for desc in descriptors {
let mediaType = desc.mediaType
switch mediaType {
case MediaTypes.index, MediaTypes.dockerManifestList:
let index: Index = try await self.getManifestContent(descriptor: desc)
out.append(contentsOf: index.manifests)
case MediaTypes.imageManifest, MediaTypes.dockerManifest:
let manifest: Manifest = try await self.getManifestContent(descriptor: desc)
out.append(manifest.config)
out.append(contentsOf: manifest.layers)
default:
// TODO: Explicitly handle other content types
continue
}
}
return out
}
private func fetchAll(_ descriptors: [Descriptor]) async throws {
try await withThrowingTaskGroup(of: Void.self) { group in
var iterator = descriptors.makeIterator()
// Start initial batch of concurrent downloads based on maxConcurrentDownloads
for _ in 0..<self.maxConcurrentDownloads {
if let desc = iterator.next() {
group.addTask {
try await self.fetch(desc)
}
}
}
// As tasks complete, add new ones to maintain concurrency
for try await _ in group {
if let desc = iterator.next() {
group.addTask {
try await self.fetch(desc)
}
}
}
}
}
private func fetch(_ descriptor: Descriptor) async throws {
if let found = try await self.contentStore.get(digest: descriptor.digest) {
try FileManager.default.copyItem(at: found.path, to: ingestDir.appendingPathComponent(descriptor.digest.trimmingDigestPrefix))
await progress?([
// Count the size of the blob
.addSize(descriptor.size),
// Count the number of blobs
.addItems(1),
])
return
}
if descriptor.size > 1.mib() {
try await self.fetchBlob(descriptor)
} else {
try await self.fetchData(descriptor)
}
// Count the number of blobs
await progress?([
.addItems(1)
])
}
private func fetchBlob(_ descriptor: Descriptor) async throws {
let id = UUID().uuidString
let fm = FileManager.default
let tempFile = ingestDir.appendingPathComponent(id)
let (_, digest) = try await client.fetchBlob(name: name, descriptor: descriptor, into: tempFile, progress: progress)
guard digest.digestString == descriptor.digest else {
throw ContainerizationError(.internalError, message: "digest mismatch expected \(descriptor.digest), got \(digest.digestString)")
}
do {
try fm.moveItem(at: tempFile, to: ingestDir.appendingPathComponent(digest.encoded))
} catch let err as NSError {
guard err.code == NSFileWriteFileExistsError else {
throw err
}
try fm.removeItem(at: tempFile)
}
}
@discardableResult
private func fetchData(_ descriptor: Descriptor) async throws -> Data {
let data = try await client.fetchData(name: name, descriptor: descriptor)
let writer = try ContentWriter(for: ingestDir)
let result = try writer.write(data)
if let progress {
let size = Int64(result.size)
await progress([
.addSize(size)
])
}
guard result.digest.digestString == descriptor.digest else {
throw ContainerizationError(.internalError, message: "digest mismatch expected \(descriptor.digest), got \(result.digest.digestString)")
}
return data
}
private func createIndex(for root: Descriptor) async throws -> Index {
switch root.mediaType {
case MediaTypes.index, MediaTypes.dockerManifestList:
return try await self.getManifestContent(descriptor: root)
case MediaTypes.imageManifest, MediaTypes.dockerManifest:
let supportedPlatforms = try await getSupportedPlatforms(for: root)
guard supportedPlatforms.count == 1 else {
throw ContainerizationError(
.internalError,
message:
"descriptor \(root.mediaType) with digest \(root.digest) does not list any supported platform or supports more than one platform, supported platforms: \(supportedPlatforms)"
)
}
let platform = supportedPlatforms.first!
var root = root
root.platform = platform
let index = ContainerizationOCI.Index(
schemaVersion: 2, manifests: [root],
annotations: [
// indicate that this is a synthesized index which is not directly user facing
AnnotationKeys.containerizationIndexIndirect: "true"
])
return index
default:
throw ContainerizationError(.internalError, message: "failed to create index for descriptor \(root.digest), media type \(root.mediaType)")
}
}
private func getSupportedPlatforms(for root: Descriptor) async throws -> [ContainerizationOCI.Platform] {
var supportedPlatforms: [ContainerizationOCI.Platform] = []
var toProcess = [root]
while !toProcess.isEmpty {
let children = try await self.walk(toProcess)
for child in children {
if let p = child.platform {
supportedPlatforms.append(p)
continue
}
switch child.mediaType {
case MediaTypes.imageConfig, MediaTypes.dockerImageConfig:
let config: ContainerizationOCI.Image = try await self.getManifestContent(descriptor: child)
let p = ContainerizationOCI.Platform(
arch: config.architecture, os: config.os, osFeatures: config.osFeatures, variant: config.variant
)
supportedPlatforms.append(p)
default:
continue
}
}
toProcess = children
}
return supportedPlatforms
}
}
}
@@ -0,0 +1,110 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationError
import ContainerizationExtras
import ContainerizationOCI
import Foundation
extension ImageStore {
/// Exports the specified images and their associated layers to an OCI Image Layout directory.
/// This function saves the images identified by the `references` array, including their
/// manifests and layer blobs, into a directory structure compliant with the OCI Image Layout specification at the given `out` URL.
///
/// - Parameters:
/// - references: A list image references that exists in the `ImageStore` that are to be saved in the OCI Image Layout format.
/// - out: A URL to a directory on disk at which the OCI Image Layout structure will be created.
/// - platform: An optional parameter to indicate the platform to be saved for the images.
/// Defaults to `nil` signifying that layers for all supported platforms by the images will be saved.
///
public func save(references: [String], out: URL, platform: Platform? = nil) async throws {
let matcher = createPlatformMatcher(for: platform)
let fileManager = FileManager.default
let tempDir = fileManager.uniqueTemporaryDirectory()
defer {
try? fileManager.removeItem(at: tempDir)
}
var toSave: [Image] = []
for reference in references {
let image = try await self.get(reference: reference)
let allowedMediaTypes = [MediaTypes.dockerManifestList, MediaTypes.index]
guard allowedMediaTypes.contains(image.mediaType) else {
throw ContainerizationError(.internalError, message: "cannot save image \(image.reference) with Index media type \(image.mediaType)")
}
toSave.append(image)
}
let client = try LocalOCILayoutClient(root: out)
var saved: [Descriptor] = []
for image in toSave {
let ref = try Reference.parse(image.reference)
let name = ref.path
guard let tag = ref.tag ?? ref.digest else {
throw ContainerizationError(.invalidArgument, message: "invalid tag/digest for image reference \(image.reference)")
}
let operation = ExportOperation(name: name, tag: tag, contentStore: self.contentStore, client: client, progress: nil)
var descriptor = try await operation.export(index: image.descriptor, platforms: matcher)
client.setImageReferenceAnnotation(descriptor: &descriptor, reference: image.reference)
saved.append(descriptor)
}
try client.createOCILayoutStructure(directory: out, manifests: saved)
}
/// Imports one or more images and their associated layers from an OCI Image Layout directory.
///
/// - Parameters:
/// - directory: A URL to a directory on disk at that follows the OCI Image Layout structure.
/// - progress: An optional handler over which progress update events about the load operation can be received.
/// - Returns: The list of images that were loaded into the `ImageStore`.
///
public func load(from directory: URL, progress: ProgressHandler? = nil) async throws -> [Image] {
let client = try LocalOCILayoutClient(root: directory)
let index = try client.loadIndexFromOCILayout(directory: directory)
let matcher = createPlatformMatcher(for: nil)
var loaded: [Image.Description] = []
let (id, tempDir) = try await self.contentStore.newIngestSession()
do {
for descriptor in index.manifests {
let reference = client.getImageReferencefromDescriptor(descriptor: descriptor)
let ref = try Reference.parse(reference)
let name = ref.path
let operation = ImportOperation(name: name, contentStore: self.contentStore, client: client, ingestDir: tempDir, progress: progress)
let indexDesc = try await operation.import(root: descriptor, matcher: matcher)
loaded.append(Image.Description(reference: reference, descriptor: indexDesc))
}
let loadedImages = loaded
let importedImages = try await self.lock.withLock { lock in
var images: [Image] = []
try await self.contentStore.completeIngestSession(id)
for description in loadedImages {
let img = try await self._create(description: description, lock: lock)
images.append(img)
}
return images
}
guard importedImages.count > 0 else {
throw ContainerizationError(.internalError, message: "failed to import image")
}
return importedImages
} catch {
try? await self.contentStore.cancelIngestSession(id)
throw error
}
}
}
@@ -0,0 +1,90 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationError
import ContainerizationOCI
import Foundation
extension ImageStore {
/// A ReferenceManager handles the mappings between an image's
/// reference and the underlying descriptor inside of a content store.
internal actor ReferenceManager {
private let path: URL
private typealias State = [String: Descriptor]
private var images: State
public init(path: URL) throws {
try FileManager.default.createDirectory(at: path, withIntermediateDirectories: true)
self.path = path
self.images = [:]
}
private func load() throws -> State {
let statePath = self.path.appendingPathComponent("state.json")
guard FileManager.default.fileExists(atPath: statePath.absolutePath()) else {
return [:]
}
do {
let data = try Data(contentsOf: statePath)
return try JSONDecoder().decode(State.self, from: data)
} catch {
throw ContainerizationError(.internalError, message: "failed to load image state \(error.localizedDescription)")
}
}
private func save(_ state: State) throws {
let statePath = self.path.appendingPathComponent("state.json")
try JSONEncoder().encode(state).write(to: statePath)
}
public func delete(reference: String) throws {
var state = try self.load()
state.removeValue(forKey: reference)
try self.save(state)
}
public func delete(image: Image.Description) throws {
try self.delete(reference: image.reference)
}
public func create(description: Image.Description) throws {
var state = try self.load()
state[description.reference] = description.descriptor
try self.save(state)
}
public func list() throws -> [Image.Description] {
let state = try self.load()
return state.map { key, val in
let description = Image.Description(reference: key, descriptor: val)
return description
}
}
public func get(reference: String) throws -> Image.Description {
let images = try self.list()
let hit = images.first(where: { image in
image.reference == reference
})
guard let hit else {
throw ContainerizationError(.notFound, message: "image \(reference) not found")
}
return hit
}
}
}
@@ -0,0 +1,395 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationError
import ContainerizationExtras
import ContainerizationOCI
import Foundation
/// An ImageStore handles the mappings between an image's
/// reference and the underlying descriptor inside of a content store.
public actor ImageStore: Sendable {
/// The ImageStore path it was created with.
public nonisolated let path: URL
private let referenceManager: ReferenceManager
internal let contentStore: ContentStore
internal let lock: AsyncLock = AsyncLock()
public init(path: URL, contentStore: ContentStore? = nil) throws {
try FileManager.default.createDirectory(at: path, withIntermediateDirectories: true)
if let contentStore {
self.contentStore = contentStore
} else {
self.contentStore = try LocalContentStore(path: path.appendingPathComponent("content"))
}
self.path = path
self.referenceManager = try ReferenceManager(path: path)
}
/// Return the default image store for the current user.
public static let `default`: ImageStore = {
do {
let root = try defaultRoot()
return try ImageStore(path: root)
} catch {
fatalError("unable to initialize default ImageStore \(error)")
}
}()
private static func defaultRoot() throws -> URL {
let root = FileManager.default.urls(
for: .applicationSupportDirectory,
in: .userDomainMask
).first
guard let root else {
throw ContainerizationError(.notFound, message: "unable to get Application Support directory for current user")
}
return root.appendingPathComponent("com.apple.containerization")
}
}
extension ImageStore {
/// Get an image from the `ImageStore`.
///
/// - Parameters:
/// - reference: Name of the image.
/// - pull: Pull the image if it is not found.
///
/// - Returns: A `Containerization.Image` object whose `reference` matches the given string.
/// This method throws a `ContainerizationError(code: .notFound)` if the provided reference does not exist in the `ImageStore`.
public func get(reference: String, pull: Bool = false) async throws -> Image {
do {
let desc = try await self.referenceManager.get(reference: reference)
return Image(description: desc, contentStore: self.contentStore)
} catch let error as ContainerizationError {
if error.code == .notFound && pull {
return try await self.pull(reference: reference)
}
throw error
}
}
/// Get a list of all images in the `ImageStore`.
///
/// - Returns: A `[Containerization.Image]` for all the images in the `ImageStore`.
public func list() async throws -> [Image] {
try await self.referenceManager.list().map { desc in
Image(description: desc, contentStore: self.contentStore)
}
}
/// Create a new image in the `ImageStore`.
///
/// - Parameters:
/// - description: The underlying `Image.Description` that contains information about the reference and index descriptor for the image to be created.
///
/// - Note: It is assumed that the underlying manifests and blob layers for the image already exists in the `ContentStore` that the `ImageStore` was initialized with. This method is invoked when the `pull(...)` , `load(...)` and `tag(...)` methods are used.
/// - Returns: A `Containerization.Image`
@discardableResult
public func create(description: Image.Description) async throws -> Image {
try await self.lock.withLock { ctx in
try await self._create(description: description, lock: ctx)
}
}
@discardableResult
internal func _create(description: Image.Description, lock: AsyncLock.Context) async throws -> Image {
try await self.referenceManager.create(description: description)
return Image(description: description, contentStore: self.contentStore)
}
/// Delete an image from the `ImageStore`.
///
/// - Parameters:
/// - reference: Name of the image that is to be deleted.
/// - performCleanup: Perform a garbage collection on the `ContentStore`, removing all unreferenced image layers and manifests,
public func delete(reference: String, performCleanup: Bool = false) async throws {
try await self.lock.withLock { lockCtx in
try await self.referenceManager.delete(reference: reference)
if performCleanup {
try await self._cleanUpOrphanedBlobs(lockCtx)
}
}
}
/// Clean up orphaned blobs that are no longer referenced by any image.
///
/// - Returns: Returns a tuple of `(deleted, freed)`.
/// `deleted` : A list of the names of the content items that were deleted from the `ContentStore`,
/// `freed` : The total size of the items that were deleted.
@discardableResult
public func cleanUpOrphanedBlobs() async throws -> (deleted: [String], freed: UInt64) {
try await self.lock.withLock { lockCtx in
try await self._cleanUpOrphanedBlobs(lockCtx)
}
}
/// Calculate the size of orphaned blobs without deleting them.
///
/// - Returns: The total size in bytes of blobs that are not referenced by any image.
public func calculateOrphanedBlobsSize() async throws -> UInt64 {
try await self.lock.withLock { lockCtx in
try await self._calculateOrphanedBlobsSize(lockCtx)
}
}
@discardableResult
private func _cleanUpOrphanedBlobs(_ lock: AsyncLock.Context) async throws -> (deleted: [String], freed: UInt64) {
let images = try await self.list()
var referenced: [String] = []
for image in images {
try await referenced.append(contentsOf: image.referencedDigests().uniqued())
}
let (deleted, size) = try await self.contentStore.delete(keeping: referenced)
return (deleted, size)
}
private func _calculateOrphanedBlobsSize(_ lock: AsyncLock.Context) async throws -> UInt64 {
let images = try await self.list()
var referenced: [String] = []
for image in images {
try await referenced.append(contentsOf: image.referencedDigests().uniqued())
}
// Calculate size of blobs not in the referenced list
let referencedSet = Set(referenced.map { $0.trimmingDigestPrefix })
let blobsPath = self.path.appendingPathComponent("content/blobs/sha256")
let fileManager = FileManager.default
let allBlobs = try fileManager.contentsOfDirectory(
at: blobsPath,
includingPropertiesForKeys: [.fileSizeKey],
options: [.skipsHiddenFiles]
)
var orphanedSize: UInt64 = 0
for blobURL in allBlobs {
let digest = blobURL.lastPathComponent
if !referencedSet.contains(digest) {
if let resourceValues = try? blobURL.resourceValues(forKeys: [.fileSizeKey]),
let size = resourceValues.fileSize
{
orphanedSize += UInt64(size)
}
}
}
return orphanedSize
}
/// Tag an existing image such that it can be referenced by another name.
///
/// - Parameters:
/// - existing: The reference to an image that already exists in the `ImageStore`.
/// - new: The new reference by which the image should also be referenced as.
/// - Note: The new image created in the `ImageStore` will have the same `Image.Description`
/// as that of the image with reference `existing.`
/// - Returns: A `Containerization.Image` object to the newly created image.
public func tag(existing: String, new: String) async throws -> Image {
let old = try await self.get(reference: existing)
let descriptor = old.descriptor
do {
_ = try Reference.parse(new)
} catch {
throw ContainerizationError(.invalidArgument, message: "invalid reference \(new), error: \(error)")
}
let newDescription = Image.Description(reference: new, descriptor: descriptor)
return try await self.create(description: newDescription)
}
}
extension ImageStore {
/// Pull an image and its associated manifest and blob layers from a remote registry.
///
/// - Parameters:
/// - reference: A string that references an image in a remote registry of the form `<host>[:<port>]/repository:<tag>`
/// For example: "docker.io/library/alpine:latest".
/// - platform: An optional parameter to indicate the platform to be pulled for the image.
/// Defaults to `nil` signifying that layers for all supported platforms by the image will be pulled.
/// - insecure: A boolean indicating if the connection to the remote registry should be made via plain-text http or not.
/// Defaults to false, meaning the connection to the registry will be over https.
/// - auth: An object that implements the `Authentication` protocol,
/// used to add any credentials to the HTTP requests that are made to the registry.
/// Defaults to `nil` meaning no additional credentials are added to any HTTP requests made to the registry.
/// - progress: An optional handler over which progress update events about the pull operation can be received.
///
/// - Returns: A `Containerization.Image` object to the newly pulled image.
public func pull(
reference: String, platform: Platform? = nil, insecure: Bool = false,
auth: Authentication? = nil, progress: ProgressHandler? = nil, maxConcurrentDownloads: Int = 3
) async throws -> Image {
let matcher = createPlatformMatcher(for: platform)
let client = try RegistryClient(reference: reference, insecure: insecure, auth: auth, tlsConfiguration: TLSUtils.makeEnvironmentAwareTLSConfiguration())
let ref = try Reference.parse(reference)
let name = ref.path
guard let tag = ref.tag ?? ref.digest else {
throw ContainerizationError(.invalidArgument, message: "invalid tag/digest for image reference \(reference)")
}
let rootDescriptor = try await client.resolve(name: name, tag: tag)
let (id, tempDir) = try await self.contentStore.newIngestSession()
let operation = ImportOperation(
name: name, contentStore: self.contentStore, client: client, ingestDir: tempDir, progress: progress, maxConcurrentDownloads: maxConcurrentDownloads)
do {
let index = try await operation.import(root: rootDescriptor, matcher: matcher)
return try await self.lock.withLock { lock in
try await self.contentStore.completeIngestSession(id)
let description = Image.Description(reference: reference, descriptor: index)
let image = try await self._create(description: description, lock: lock)
return image
}
} catch {
try? await self.contentStore.cancelIngestSession(id)
throw error
}
}
/// Push an image and its associated manifest and blob layers to a remote registry.
///
/// - Parameters:
/// - reference: A string that references an image in the `ImageStore`. It must be of the form `<host>[:<port>]/repository:<tag>`
/// For example: "ghcr.io/foo-bar-baz/image:v1".
/// - platform: An optional parameter to indicate the platform to be pushed for the image.
/// Defaults to `nil` signifying that layers for all supported platforms by the image will be pushed to the remote registry.
/// - insecure: A boolean indicating if the connection to the remote registry should be made via plain-text http or not.
/// Defaults to false, meaning the connection to the registry will be over https.
/// - auth: An object that implements the `Authentication` protocol,
/// used to add any credentials to the HTTP requests that are made to the registry.
/// Defaults to `nil` meaning no additional credentials are added to any HTTP requests made to the registry.
/// - progress: An optional handler over which progress update events about the push operation can be received.
///
public func push(reference: String, platform: Platform? = nil, insecure: Bool = false, auth: Authentication? = nil, progress: ProgressHandler? = nil) async throws {
let matcher = createPlatformMatcher(for: platform)
let client = try RegistryClient(reference: reference, insecure: insecure, auth: auth, tlsConfiguration: TLSUtils.makeEnvironmentAwareTLSConfiguration())
try await self.pushSingle(reference: reference, client: client, matcher: matcher, progress: progress)
}
/// Push multiple image references to a remote registry, sharing a single ``RegistryClient``.
///
/// All references must resolve to the same registry host. Passing references that target
/// different hosts throws a ``ContainerizationError`` with code ``invalidArgument``.
///
/// - Parameters:
/// - references: An array of fully qualified image reference strings to push.
/// Each must include a host (e.g., `"ghcr.io/myrepo/myimage:v1"`).
/// - platform: An optional parameter to indicate the platform to be pushed for each image.
/// Defaults to `nil` signifying that layers for all supported platforms will be pushed.
/// - insecure: A boolean indicating if the connection to the remote registry should be made via plain-text http or not.
/// Defaults to false, meaning the connection to the registry will be over https.
/// - auth: An object that implements the `Authentication` protocol,
/// used to add any credentials to the HTTP requests that are made to the registry.
/// Defaults to `nil` meaning no additional credentials are added to any HTTP requests made to the registry.
/// - maxConcurrentUploads: Maximum number of concurrent tag pushes. Defaults to 3.
/// - progress: An optional handler over which progress update events about the push operations can be received.
///
public func push(
references: [String], platform: Platform? = nil, insecure: Bool = false,
auth: Authentication? = nil, maxConcurrentUploads: Int = 3, progress: ProgressHandler? = nil
) async throws {
guard let firstReference = references.first else {
return
}
// Parse all references upfront: validate hosts and avoid re-parsing inside tasks.
let parsed = try references.map { ref in try Reference.parse(ref) }
let hosts = parsed.compactMap { $0.resolvedDomain }
guard hosts.count == references.count else {
throw ContainerizationError(.invalidArgument, message: "all references must include a host")
}
let uniqueHosts = Set(hosts)
guard uniqueHosts.count == 1 else {
throw ContainerizationError(
.invalidArgument,
message: "all references must target the same registry host, got: \(uniqueHosts.sorted().joined(separator: ", "))")
}
let matcher = createPlatformMatcher(for: platform)
let client = try RegistryClient(
reference: firstReference, insecure: insecure, auth: auth,
tlsConfiguration: TLSUtils.makeEnvironmentAwareTLSConfiguration())
let pushOne: @Sendable (String) async -> (String, String?) = { reference in
do {
try await self.pushSingle(reference: reference, client: client, matcher: matcher, progress: progress)
return (reference, nil)
} catch {
return (reference, String(describing: error))
}
}
var iterator = references.makeIterator()
var failures: [(reference: String, message: String)] = []
await withTaskGroup(of: (String, String?).self) { group in
for _ in 0..<maxConcurrentUploads {
guard let reference = iterator.next() else { break }
group.addTask { await pushOne(reference) }
}
for await (ref, error) in group {
if let error {
failures.append((ref, error))
}
if let reference = iterator.next() {
group.addTask { await pushOne(reference) }
}
}
}
if !failures.isEmpty {
let details = failures.map { "\($0.reference): \($0.message)" }.joined(separator: "\n")
throw ContainerizationError(.internalError, message: "failed to push one or more images:\n\(details)")
}
}
private func pushSingle(
reference: String, client: ContentClient, matcher: @Sendable (Platform) -> Bool, progress: ProgressHandler?
) async throws {
let allowedMediaTypes = [MediaTypes.dockerManifestList, MediaTypes.index]
let img = try await self.get(reference: reference)
guard allowedMediaTypes.contains(img.mediaType) else {
throw ContainerizationError(.internalError, message: "cannot push image \(reference): unsupported media type \(img.mediaType), expected an index or manifest list")
}
let ref = try Reference.parse(reference)
guard let tag = ref.tag ?? ref.digest else {
throw ContainerizationError(.invalidArgument, message: "invalid tag/digest for image reference \(reference)")
}
let operation = ExportOperation(name: ref.path, tag: tag, contentStore: self.contentStore, client: client, progress: progress)
try await operation.export(index: img.descriptor, platforms: matcher)
}
}
extension ImageStore {
/// Get the image for the init block from the image store.
/// If the image does not exist locally, pull the image.
public func getInitImage(reference: String, auth: Authentication? = nil, progress: ProgressHandler? = nil) async throws -> InitImage {
do {
let image = try await self.get(reference: reference)
return InitImage(image: image)
} catch let error as ContainerizationError {
if error.code == .notFound {
let image = try await self.pull(reference: reference, auth: auth, progress: progress)
return InitImage(image: image)
}
throw error
}
}
}
@@ -0,0 +1,85 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationError
import ContainerizationOCI
import Foundation
/// Data representing the image to use as the root filesystem for a virtual machine.
/// Typically this image would contain the guest agent used to facilitate container
/// workloads, as well as any extras that may be useful to have in the guest.
public struct InitImage: Sendable {
public var name: String { image.reference }
let image: Image
public init(image: Image) {
self.image = image
}
}
extension InitImage {
/// Unpack the initial filesystem for the desired platform at a given path.
public func initBlock(at: URL, for platform: SystemPlatform) async throws -> Mount {
let unpacker = EXT4Unpacker(blockSizeInBytes: 512.mib())
var fs = try await unpacker.unpack(self.image, for: platform.ociPlatform(), at: at)
fs.options = ["ro"]
return fs
}
/// Create a new InitImage with the reference as the name.
/// The `rootfs` parameter must be a tar.gz file whose contents make up the filesystem for the image.
public static func create(
reference: String, rootfs: URL, platform: Platform,
labels: [String: String] = [:], imageStore: ImageStore, contentStore: ContentStore
) async throws -> InitImage {
let indexDescriptorStore = AsyncStore<Descriptor>()
try await contentStore.ingest { dir in
let writer = try ContentWriter(for: dir)
var result = try writer.create(from: rootfs)
let layerDescriptor = Descriptor(mediaType: ContainerizationOCI.MediaTypes.imageLayerGzip, digest: result.digest.digestString, size: result.size)
// TODO: compute and fill in the correct diffID for the above layer
// We currently put in the sha of the fully compressed layer, this needs to be replaced with
// the sha of the uncompressed layer.
let rootfsConfig = ContainerizationOCI.Rootfs(type: "layers", diffIDs: [result.digest.digestString])
let runtimeConfig = ContainerizationOCI.ImageConfig(labels: labels)
let imageConfig = ContainerizationOCI.Image(architecture: platform.architecture, os: platform.os, config: runtimeConfig, rootfs: rootfsConfig)
result = try writer.create(from: imageConfig)
let configDescriptor = Descriptor(mediaType: ContainerizationOCI.MediaTypes.imageConfig, digest: result.digest.digestString, size: result.size)
let manifest = Manifest(config: configDescriptor, layers: [layerDescriptor])
result = try writer.create(from: manifest)
let manifestDescriptor = Descriptor(mediaType: ContainerizationOCI.MediaTypes.imageManifest, digest: result.digest.digestString, size: result.size, platform: platform)
let index = ContainerizationOCI.Index(manifests: [manifestDescriptor])
result = try writer.create(from: index)
let indexDescriptor = Descriptor(mediaType: ContainerizationOCI.MediaTypes.index, digest: result.digest.digestString, size: result.size)
await indexDescriptorStore.set(indexDescriptor)
}
guard let indexDescriptor = await indexDescriptorStore.get() else {
throw ContainerizationError(.notFound, message: "image for \(reference) not found")
}
let description = Image.Description(reference: reference, descriptor: indexDescriptor)
let image = try await imageStore.create(description: description)
return InitImage(image: image)
}
}
@@ -0,0 +1,94 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationError
import ContainerizationOCI
import Foundation
/// A multi-arch kernel image represented by an OCI image.
public struct KernelImage: Sendable {
/// The media type for a kernel image.
public static let mediaType = "application/vnd.apple.containerization.kernel"
/// The name or reference of the image.
public var name: String { image.reference }
let image: Image
public init(image: Image) {
self.image = image
}
}
extension KernelImage {
/// Return the kernel from a multi arch image for a specific system platform.
public func kernel(for platform: SystemPlatform) async throws -> Kernel {
let manifest = try await image.manifest(for: platform.ociPlatform())
guard let descriptor = manifest.layers.first, descriptor.mediaType == Self.mediaType else {
throw ContainerizationError(.notFound, message: "kernel descriptor for \(platform) not found")
}
let content = try await image.getContent(digest: descriptor.digest)
return Kernel(
path: content.path,
platform: platform
)
}
/// Create a new kernel image with the reference as the name.
/// This will create a multi arch image containing kernel's for each provided architecture.
public static func create(reference: String, binaries: [Kernel], labels: [String: String] = [:], imageStore: ImageStore, contentStore: ContentStore) async throws -> KernelImage
{
let indexDescriptorStore = AsyncStore<Descriptor>()
try await contentStore.ingest { ingestPath in
var descriptors = [Descriptor]()
let writer = try ContentWriter(for: ingestPath)
for kernel in binaries {
var result = try writer.create(from: kernel.path)
let platform = kernel.platform.ociPlatform()
let layerDescriptor = Descriptor(
mediaType: mediaType,
digest: result.digest.digestString,
size: result.size,
platform: platform)
let rootfsConfig = ContainerizationOCI.Rootfs(type: "layers", diffIDs: [result.digest.digestString])
let runtimeConfig = ContainerizationOCI.ImageConfig(labels: labels)
let imageConfig = ContainerizationOCI.Image(architecture: platform.architecture, os: platform.os, config: runtimeConfig, rootfs: rootfsConfig)
result = try writer.create(from: imageConfig)
let configDescriptor = Descriptor(mediaType: ContainerizationOCI.MediaTypes.imageConfig, digest: result.digest.digestString, size: result.size)
let manifest = Manifest(config: configDescriptor, layers: [layerDescriptor])
result = try writer.create(from: manifest)
let manifestDescriptor = Descriptor(
mediaType: ContainerizationOCI.MediaTypes.imageManifest, digest: result.digest.digestString, size: result.size, platform: platform)
descriptors.append(manifestDescriptor)
}
let index = ContainerizationOCI.Index(manifests: descriptors)
let result = try writer.create(from: index)
let indexDescriptor = Descriptor(mediaType: ContainerizationOCI.MediaTypes.index, digest: result.digest.digestString, size: result.size)
await indexDescriptorStore.set(indexDescriptor)
}
guard let indexDescriptor = await indexDescriptorStore.get() else {
throw ContainerizationError(.notFound, message: "image for \(reference) not found")
}
let description = Image.Description(reference: reference, descriptor: indexDescriptor)
let image = try await imageStore.create(description: description)
return KernelImage(image: image)
}
}
@@ -0,0 +1,161 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationArchive
import ContainerizationEXT4
import ContainerizationError
import ContainerizationExtras
import ContainerizationOCI
import Foundation
import SystemPackage
public struct EXT4Unpacker: Unpacker {
let blockSizeInBytes: UInt64
public init(blockSizeInBytes: UInt64) {
self.blockSizeInBytes = blockSizeInBytes
}
/// Performs the unpacking of a tar archive into a filesystem.
/// - Parameters:
/// - archive: The archive to unpack.
/// - compression: The compression to use when unpacking the image.
/// - path: The path to the filesystem that will be created.
public func unpack(
archive: URL,
compression: ContainerizationArchive.Filter,
at path: URL
) async throws {
let cleanedPath = try prepareUnpackPath(path: path)
let filesystem = try EXT4.Formatter(
FilePath(cleanedPath),
minDiskSize: blockSizeInBytes
)
defer { try? filesystem.close() }
try await filesystem.unpack(
source: archive,
format: .paxRestricted,
compression: compression
)
}
/// Returns a `Mount` point after unpacking the image into a filesystem.
/// - Parameters:
/// - image: The image to unpack.
/// - platform: The platform content to unpack.
/// - path: The path to the directory where the filesystem will be created.
/// - progress: The progress handler to invoke as the unpacking progresses.
public func unpack(
_ image: Image,
for platform: Platform,
at path: URL,
progress: ProgressHandler? = nil
) async throws -> Mount {
let cleanedPath = try prepareUnpackPath(path: path)
let manifest = try await image.manifest(for: platform)
let filesystem = try EXT4.Formatter(
FilePath(
cleanedPath
),
minDiskSize: blockSizeInBytes
)
defer { try? filesystem.close() }
// Resolve layer paths upfront. When progress reporting is enabled and a layer
// uses zstd, decompress once so both the size-scanning pass and the unpack
// pass share the same decompressed file.
var resolvedLayers: [(file: URL, filter: ContainerizationArchive.Filter)] = []
var decompressedFiles: [URL] = []
for layer in manifest.layers {
try Task.checkCancellation()
let content = try await image.getContent(digest: layer.digest)
let compression = try compressionFilter(for: layer.mediaType)
if progress != nil && compression == .zstd {
let decompressed = try ArchiveReader.decompressZstd(content.path)
decompressedFiles.append(decompressed)
resolvedLayers.append((file: decompressed, filter: .none))
} else {
resolvedLayers.append((file: content.path, filter: compression))
}
}
defer {
for file in decompressedFiles {
ArchiveReader.cleanUpDecompressedZstd(file)
}
}
if let progress {
var totalSize: Int64 = 0
var totalItems: Int = 0
for layer in resolvedLayers {
try Task.checkCancellation()
let totals = try EXT4.Formatter.scanArchiveHeaders(
format: .paxRestricted, filter: layer.filter, file: layer.file)
totalSize += totals.size
totalItems += totals.items
}
var totalEvents: [ProgressEvent] = []
if totalSize > 0 {
totalEvents.append(.addTotalSize(totalSize))
}
if totalItems > 0 {
totalEvents.append(.addTotalItems(totalItems))
}
if !totalEvents.isEmpty {
await progress(totalEvents)
}
}
for resolved in resolvedLayers {
try Task.checkCancellation()
let reader = try ArchiveReader(
format: .paxRestricted,
filter: resolved.filter,
file: resolved.file
)
try await filesystem.unpack(reader: reader, progress: progress)
}
return .block(
format: "ext4",
source: cleanedPath,
destination: "/",
options: []
)
}
private func prepareUnpackPath(path: URL) throws -> String {
let blockPath = path.absolutePath()
guard !FileManager.default.fileExists(atPath: blockPath) else {
throw ContainerizationError(.exists, message: "block device already exists at \(blockPath)")
}
return blockPath
}
private func compressionFilter(for mediaType: String) throws -> ContainerizationArchive.Filter {
switch mediaType {
case MediaTypes.imageLayer, MediaTypes.dockerImageLayer:
return .none
case MediaTypes.imageLayerGzip, MediaTypes.dockerImageLayerGzip:
return .gzip
case MediaTypes.imageLayerZstd, MediaTypes.dockerImageLayerZstd:
return .zstd
default:
throw ContainerizationError(.unsupported, message: "media type \(mediaType) not supported.")
}
}
}
@@ -0,0 +1,40 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationExtras
import ContainerizationOCI
import Foundation
/// The `Unpacker` protocol defines a standardized interface that involves
/// decompressing, extracting image layers and preparing it for use.
///
/// The `Unpacker` is responsible for managing the lifecycle of the
/// unpacking process, including any temporary files or resources, until the
/// `Mount` object is produced.
public protocol Unpacker {
/// Unpacks the provided image to a specified path for a given platform.
///
/// This asynchronous method should handle the entire unpacking process, from reading
/// the `Image` layers for the given `Platform` via its `Manifest`,
/// to making the extracted contents available as a `Mount`.
/// Implementations of this method may apply platform-specific optimizations
/// or transformations during the unpacking.
///
/// Progress updates can be observed via the optional `progress` handler.
func unpack(_ image: Image, for platform: Platform, at path: URL, progress: ProgressHandler?) async throws -> Mount
}
+46
View File
@@ -0,0 +1,46 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationExtras
/// A network interface.
public protocol Interface: Sendable {
/// The interface IPv4 address and subnet prefix length, as a CIDR address.
/// Example: `192.168.64.3/24`
var ipv4Address: CIDRv4 { get }
/// The IPv4 gateway address for the default route, or nil for no IPv4 default route.
var ipv4Gateway: IPv4Address? { get }
/// The interface IPv6 address and subnet prefix length, as a CIDRv6 address, or nil for no IPv6 address.
/// Example: `fd00::1/64`
var ipv6Address: CIDRv6? { get }
/// The IPv6 gateway address for the default route, or nil for no IPv6 default route.
var ipv6Gateway: IPv6Address? { get }
/// The interface MAC address, or nil to auto-configure the address.
var macAddress: MACAddress? { get }
/// The interface MTU (Maximum Transmission Unit).
var mtu: UInt32 { get }
}
extension Interface {
public var mtu: UInt32 { 1500 }
public var ipv6Address: CIDRv6? { nil }
public var ipv6Gateway: IPv6Address? { nil }
}
+101
View File
@@ -0,0 +1,101 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import Foundation
import Logging
/// An object representing a Linux kernel used to boot a virtual machine.
/// In addition to a path to the kernel itself, this type stores relevant
/// data such as the commandline to pass to the kernel, and init arguments.
public struct Kernel: Sendable, Codable {
/// The command line arguments passed to the kernel on boot.
public struct CommandLine: Sendable, Codable {
public static let kernelDefaults = [
"console=hvc0",
"tsc=reliable",
]
/// Adds the debug argument to the kernel commandline.
mutating public func addDebug() {
self.kernelArgs.append("debug")
}
/// Adds a panic level to the kernel commandline.
mutating public func addPanic(level: Int) {
self.kernelArgs.append("panic=\(level)")
}
// Sets the log level for the Agent
mutating public func setAgentLogLevel(level: Logger.Level) {
self.initArgs.append(contentsOf: ["--log-level", level.description])
}
/// Additional kernel arguments.
public var kernelArgs: [String]
/// Additional arguments passed to the Initial Process / Agent.
public var initArgs: [String]
/// Initializes the kernel commandline using the mix of kernel arguments
/// and init arguments.
public init(
kernelArgs: [String] = kernelDefaults,
initArgs: [String] = []
) {
self.kernelArgs = kernelArgs
self.initArgs = initArgs
}
/// Initializes the kernel commandline to the defaults of Self.kernelDefaults,
/// adds a debug and panic flag as instructed, and optionally a set of init
/// process flags to supply to vminitd.
public init(debug: Bool, panic: Int, initArgs: [String] = []) {
var args = Self.kernelDefaults
if debug {
args.append("debug")
}
args.append("panic=\(panic)")
self.kernelArgs = args
self.initArgs = initArgs
}
}
/// Path on disk to the kernel binary.
public var path: URL
/// Platform for the kernel.
public var platform: SystemPlatform
/// Kernel and init process command line.
public var commandLine: Self.CommandLine
/// Kernel command line arguments.
public var kernelArgs: [String] {
self.commandLine.kernelArgs
}
/// Init process arguments.
public var initArgs: [String] {
self.commandLine.initArgs
}
public init(
path: URL,
platform: SystemPlatform,
commandline: Self.CommandLine = CommandLine(debug: false, panic: 0)
) {
self.path = path
self.platform = platform
self.commandLine = commandline
}
}
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+477
View File
@@ -0,0 +1,477 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationError
import ContainerizationExtras
import ContainerizationOCI
import ContainerizationOS
import Foundation
import Logging
import Synchronization
/// `LinuxProcess` represents a Linux process and is used to
/// setup and control the full lifecycle for the process.
public final class LinuxProcess: Sendable {
/// The ID of the process. This is purely metadata for the caller.
public let id: String
/// What container owns this process (if any).
public let owningContainer: String?
package struct StdioSetup: Sendable {
let port: UInt32
let writer: Writer
}
package struct StdioReaderSetup {
let port: UInt32
let reader: ReaderStream
}
package struct Stdio: Sendable {
let stdin: StdioReaderSetup?
let stdout: StdioSetup?
let stderr: StdioSetup?
}
private struct StdioHandles: Sendable {
var stdin: FileHandle?
var stdout: FileHandle?
var stderr: FileHandle?
mutating func close() throws {
if let stdin {
try stdin.close()
stdin.readabilityHandler = nil
self.stdin = nil
}
if let stdout {
try stdout.close()
stdout.readabilityHandler = nil
self.stdout = nil
}
if let stderr {
try stderr.close()
stderr.readabilityHandler = nil
self.stderr = nil
}
}
}
private struct State {
var spec: ContainerizationOCI.Spec
var pid: Int32
var stdio: StdioHandles
var stdinRelay: Task<(), Never>?
var ioTracker: IoTracker?
var deletionTask: Task<Void, Error>?
struct IoTracker {
let stream: AsyncStream<Void>
let cont: AsyncStream<Void>.Continuation
let configuredStreams: Int
}
}
/// The process ID for the container process. This will be -1
/// if the process has not been started.
public var pid: Int32 {
state.withLock { $0.pid }
}
private let state: Mutex<State>
private let ioSetup: Stdio
private let agent: any VirtualMachineAgent
private let vm: any VirtualMachineInstance
private let ociRuntimePath: String?
private let logger: Logger?
private let onDelete: (@Sendable () async -> Void)?
init(
_ id: String,
containerID: String? = nil,
spec: Spec,
io: Stdio,
ociRuntimePath: String?,
agent: any VirtualMachineAgent,
vm: any VirtualMachineInstance,
logger: Logger?,
onDelete: (@Sendable () async -> Void)? = nil
) {
self.id = id
self.owningContainer = containerID
self.state = Mutex<State>(.init(spec: spec, pid: -1, stdio: StdioHandles()))
self.ioSetup = io
self.agent = agent
self.ociRuntimePath = ociRuntimePath
self.vm = vm
self.logger = logger
self.onDelete = onDelete
}
}
extension LinuxProcess {
func setupIO(listeners: [VsockListener?]) async throws -> [FileHandle?] {
let handles = try await Timeout.run(seconds: 3) {
try await withThrowingTaskGroup(of: (Int, FileHandle?).self) { group in
var results = [FileHandle?](repeating: nil, count: 3)
for (index, listener) in listeners.enumerated() {
guard let listener else { continue }
group.addTask {
let first = await listener.first(where: { _ in true })
try listener.finish()
return (index, first)
}
}
for try await (index, fileHandle) in group {
results[index] = fileHandle
}
return results
}
}
// Note: stdin relay is started separately via startStdinRelay() after
// the process has started, to avoid a deadlock where closeStdin is
// called before the process is consuming from the pipe.
var configuredStreams = 0
let (stream, cc) = AsyncStream<Void>.makeStream()
if let stdout = self.ioSetup.stdout {
configuredStreams += 1
handles[1]?.readabilityHandler = { handle in
do {
let data = handle.availableData
if data.isEmpty {
// This block is called when the producer (the guest) closes
// the fd it is writing into.
handles[1]?.readabilityHandler = nil
cc.yield()
return
}
try stdout.writer.write(data)
} catch {
self.logger?.error("failed to write to stdout: \(error)")
}
}
}
if let stderr = self.ioSetup.stderr {
configuredStreams += 1
handles[2]?.readabilityHandler = { handle in
do {
let data = handle.availableData
if data.isEmpty {
handles[2]?.readabilityHandler = nil
cc.yield()
return
}
try stderr.writer.write(data)
} catch {
self.logger?.error("failed to write to stderr: \(error)")
}
}
}
if configuredStreams > 0 {
self.state.withLock {
$0.ioTracker = .init(stream: stream, cont: cc, configuredStreams: configuredStreams)
}
}
return handles
}
func startStdinRelay(handle: FileHandle) {
guard let stdin = self.ioSetup.stdin else { return }
self.state.withLock {
$0.stdinRelay = Task {
for await data in stdin.reader.stream() {
do {
try handle.write(contentsOf: data)
} catch {
self.logger?.error("failed to write to stdin: \(error)")
break
}
}
do {
self.logger?.debug("stdin relay finished, closing")
// There's two ways we can wind up here:
//
// 1. The stream finished on its own (e.g. we wrote all the
// data) and we will close the underlying stdin in the guest below.
//
// 2. The client explicitly called closeStdin() themselves
// which will cancel this relay task AFTER actually closing
// the fds. If the client did that, then this task will be
// cancelled, and the fds are already gone so there's nothing
// for us to do.
if Task.isCancelled {
return
}
try await self._closeStdin()
} catch {
self.logger?.error("failed to close stdin: \(error)")
}
}
}
}
/// Start the process.
public func start() async throws {
do {
let spec = self.state.withLock { $0.spec }
var listeners = [VsockListener?](repeating: nil, count: 3)
if let stdin = self.ioSetup.stdin {
listeners[0] = try self.vm.listen(stdin.port)
}
if let stdout = self.ioSetup.stdout {
listeners[1] = try self.vm.listen(stdout.port)
}
if let stderr = self.ioSetup.stderr {
if spec.process!.terminal {
throw ContainerizationError(
.invalidArgument,
message: "stderr should not be configured with terminal=true"
)
}
listeners[2] = try self.vm.listen(stderr.port)
}
let t = Task {
try await self.setupIO(listeners: listeners)
}
try await agent.createProcess(
id: self.id,
containerID: self.owningContainer,
stdinPort: self.ioSetup.stdin?.port,
stdoutPort: self.ioSetup.stdout?.port,
stderrPort: self.ioSetup.stderr?.port,
ociRuntimePath: self.ociRuntimePath,
configuration: spec,
options: nil
)
let result = try await t.value
let pid = try await self.agent.startProcess(
id: self.id,
containerID: self.owningContainer
)
// Start stdin relay after process launch to avoid filling the pipe
// buffer before the process is even running.
if let stdinHandle = result[0] {
self.startStdinRelay(handle: stdinHandle)
}
self.state.withLock {
$0.stdio = StdioHandles(
stdin: result[0],
stdout: result[1],
stderr: result[2]
)
$0.pid = pid
}
} catch {
if let err = error as? ContainerizationError {
throw err
}
throw ContainerizationError(
.internalError,
message: "failed to start process",
cause: error,
)
}
}
/// Kill the process with the specified signal.
public func kill(_ signal: Signal) async throws {
do {
try await agent.signalProcess(
id: self.id,
containerID: self.owningContainer,
signal: signal.rawValue
)
} catch {
throw ContainerizationError(
.internalError,
message: "failed to kill process",
cause: error
)
}
}
/// Resize the processes pty (if requested).
public func resize(to: Terminal.Size) async throws {
do {
try await agent.resizeProcess(
id: self.id,
containerID: self.owningContainer,
columns: UInt32(to.width),
rows: UInt32(to.height)
)
} catch {
throw ContainerizationError(
.internalError,
message: "failed to resize process",
cause: error
)
}
}
public func closeStdin() async throws {
do {
try await self._closeStdin()
self.state.withLock {
$0.stdinRelay?.cancel()
}
} catch {
throw ContainerizationError(
.internalError,
message: "failed to close stdin",
cause: error,
)
}
}
func _closeStdin() async throws {
try await self.agent.closeProcessStdin(
id: self.id,
containerID: self.owningContainer
)
}
/// Wait on the process to exit with an optional timeout. Returns the exit code of the process.
@discardableResult
public func wait(timeoutInSeconds: Int64? = nil) async throws -> ExitStatus {
do {
let exitStatus = try await self.agent.waitProcess(
id: self.id,
containerID: self.owningContainer,
timeoutInSeconds: timeoutInSeconds
)
await self.waitIoComplete()
return exitStatus
} catch {
if error is ContainerizationError {
throw error
}
throw ContainerizationError(
.internalError,
message: "failed to wait on process",
cause: error
)
}
}
/// Wait until the standard output and standard error streams for the process have concluded.
private func waitIoComplete() async {
let ioTracker = self.state.withLock { $0.ioTracker }
guard let ioTracker else {
return
}
do {
try await Timeout.run(seconds: 3) {
var counter = ioTracker.configuredStreams
for await _ in ioTracker.stream {
counter -= 1
if counter == 0 {
ioTracker.cont.finish()
break
}
}
}
} catch {
self.logger?.error("timeout waiting for IO to complete for process \(id): \(error)")
}
self.state.withLock {
$0.ioTracker = nil
}
}
/// Cleans up guest state and waits on and closes any host resources (stdio handles).
public func delete() async throws {
try await self._delete()
await self.onDelete?()
}
func _delete() async throws {
let task = self.state.withLock { state in
if let existingTask = state.deletionTask {
// Deletion already in progress or finished.
return existingTask
}
let task = Task<Void, Error> {
try await self.performDeletion()
}
state.deletionTask = task
return task
}
try await task.value
}
private func performDeletion() async throws {
do {
try await self.agent.deleteProcess(
id: self.id,
containerID: self.owningContainer
)
} catch {
self.state.withLock {
$0.stdinRelay?.cancel()
try? $0.stdio.close()
}
try? await self.agent.close()
throw ContainerizationError(
.internalError,
message: "failed to delete process",
cause: error,
)
}
do {
try self.state.withLock {
$0.stdinRelay?.cancel()
try $0.stdio.close()
}
} catch {
try? await self.agent.close()
throw ContainerizationError(
.internalError,
message: "failed to close stdio",
cause: error,
)
}
do {
try await self.agent.close()
} catch {
throw ContainerizationError(
.internalError,
message: "failed to close agent connection",
cause: error,
)
}
}
}
@@ -0,0 +1,453 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationError
import ContainerizationOCI
import ContainerizationOS
/// A resource limit (rlimit) configuration for a container process.
public struct LinuxRLimit: Sendable, Hashable {
/// The kind of resource limit.
public var kind: Kind
/// The hard limit value.
public var hard: UInt64
/// The soft limit value.
public var soft: UInt64
/// Creates a new resource limit.
///
/// - Parameters:
/// - kind: The kind of resource limit.
/// - hard: The hard limit value.
/// - soft: The soft limit value.
public init(kind: Kind, hard: UInt64, soft: UInt64) {
self.kind = kind
self.hard = hard
self.soft = soft
}
/// Creates a new resource limit with the same value for both hard and soft limits.
///
/// - Parameters:
/// - kind: The kind of resource limit.
/// - limit: The limit value for both hard and soft limits.
public init(kind: Kind, limit: UInt64) {
self.kind = kind
self.hard = limit
self.soft = limit
}
/// Convert to OCI POSIXRlimit format for transport.
public func toOCI() -> POSIXRlimit {
POSIXRlimit(type: self.kind.description, hard: self.hard, soft: self.soft)
}
}
extension LinuxRLimit {
/// The kind of resource limit.
public struct Kind: Sendable, Hashable {
private enum Value: Hashable, Sendable, CaseIterable {
case addressSpace
case coreFileSize
case cpuTime
case dataSize
case fileSize
case locks
case lockedMemory
case messageQueue
case nice
case openFiles
case numberOfProcesses
case residentSetSize
case realtimePriority
case realtimeTimeout
case signalsPending
case stackSize
}
private var value: Value
private init(_ value: Value) {
self.value = value
}
/// Maximum size of the process's virtual memory (address space) in bytes.
public static var addressSpace: Self {
Self(.addressSpace)
}
/// Maximum size of a core file in bytes.
public static var coreFileSize: Self {
Self(.coreFileSize)
}
/// Maximum amount of CPU time the process can consume in seconds.
public static var cpuTime: Self {
Self(.cpuTime)
}
/// Maximum size of the process's data segment in bytes.
public static var dataSize: Self {
Self(.dataSize)
}
/// Maximum size of files the process may create in bytes.
public static var fileSize: Self {
Self(.fileSize)
}
/// Maximum number of file locks.
public static var locks: Self {
Self(.locks)
}
/// Maximum number of bytes of memory that may be locked into RAM.
public static var lockedMemory: Self {
Self(.lockedMemory)
}
/// Maximum number of bytes that can be allocated for POSIX message queues.
public static var messageQueue: Self {
Self(.messageQueue)
}
/// Maximum nice value that can be set.
public static var nice: Self {
Self(.nice)
}
/// Maximum number of open file descriptors.
public static var openFiles: Self {
Self(.openFiles)
}
/// Maximum number of processes that can be created by the user.
public static var numberOfProcesses: Self {
Self(.numberOfProcesses)
}
/// Maximum size of the process's resident set (physical memory) in bytes.
public static var residentSetSize: Self {
Self(.residentSetSize)
}
/// Maximum real-time scheduling priority.
public static var realtimePriority: Self {
Self(.realtimePriority)
}
/// Maximum amount of CPU time for real-time scheduling in microseconds.
public static var realtimeTimeout: Self {
Self(.realtimeTimeout)
}
/// Maximum number of signals that may be queued.
public static var signalsPending: Self {
Self(.signalsPending)
}
/// Maximum size of the process stack in bytes.
public static var stackSize: Self {
Self(.stackSize)
}
/// Creates a Kind from its OCI string representation.
///
/// - Parameter string: The OCI string representation (e.g., "RLIMIT_NOFILE").
/// - Throws: `ContainerizationError` with code `.invalidArgument` if the string doesn't match a known rlimit kind.
public init(_ string: String) throws {
switch string {
case "RLIMIT_AS":
self = .addressSpace
case "RLIMIT_CORE":
self = .coreFileSize
case "RLIMIT_CPU":
self = .cpuTime
case "RLIMIT_DATA":
self = .dataSize
case "RLIMIT_FSIZE":
self = .fileSize
case "RLIMIT_LOCKS":
self = .locks
case "RLIMIT_MEMLOCK":
self = .lockedMemory
case "RLIMIT_MSGQUEUE":
self = .messageQueue
case "RLIMIT_NICE":
self = .nice
case "RLIMIT_NOFILE":
self = .openFiles
case "RLIMIT_NPROC":
self = .numberOfProcesses
case "RLIMIT_RSS":
self = .residentSetSize
case "RLIMIT_RTPRIO":
self = .realtimePriority
case "RLIMIT_RTTIME":
self = .realtimeTimeout
case "RLIMIT_SIGPENDING":
self = .signalsPending
case "RLIMIT_STACK":
self = .stackSize
default:
throw ContainerizationError(.invalidArgument, message: "invalid rlimit kind: '\(string)'")
}
}
}
}
extension LinuxRLimit.Kind: CustomStringConvertible {
/// The OCI string representation of the resource limit kind.
public var description: String {
switch self.value {
case .addressSpace:
"RLIMIT_AS"
case .coreFileSize:
"RLIMIT_CORE"
case .cpuTime:
"RLIMIT_CPU"
case .dataSize:
"RLIMIT_DATA"
case .fileSize:
"RLIMIT_FSIZE"
case .locks:
"RLIMIT_LOCKS"
case .lockedMemory:
"RLIMIT_MEMLOCK"
case .messageQueue:
"RLIMIT_MSGQUEUE"
case .nice:
"RLIMIT_NICE"
case .openFiles:
"RLIMIT_NOFILE"
case .numberOfProcesses:
"RLIMIT_NPROC"
case .residentSetSize:
"RLIMIT_RSS"
case .realtimePriority:
"RLIMIT_RTPRIO"
case .realtimeTimeout:
"RLIMIT_RTTIME"
case .signalsPending:
"RLIMIT_SIGPENDING"
case .stackSize:
"RLIMIT_STACK"
}
}
}
/// User-friendly Linux capabilities configuration
public struct LinuxCapabilities: Sendable {
/// Capabilities that define the maximum set of capabilities a process can have
public var bounding: [CapabilityName] = []
/// Capabilities that are actually in effect for the current process
public var effective: [CapabilityName] = []
/// Capabilities that can be inherited by child processes
public var inheritable: [CapabilityName] = []
/// Capabilities that are currently permitted for the process
public var permitted: [CapabilityName] = []
/// Capabilities that are preserved across execve() calls
public var ambient: [CapabilityName] = []
/// Grant all capabilities
public static let allCapabilities = LinuxCapabilities(
bounding: CapabilityName.allCases,
effective: CapabilityName.allCases,
inheritable: CapabilityName.allCases,
permitted: CapabilityName.allCases,
ambient: CapabilityName.allCases
)
/// Default configuration
public static let defaultOCICapabilities = LinuxCapabilities(
bounding: [
.chown,
.dacOverride,
.fsetid,
.fowner,
.mknod,
.netRaw,
.setgid,
.setuid,
.setfcap,
.setpcap,
.netBindService,
.sysChroot,
.kill,
.auditWrite,
],
effective: [
.chown,
.dacOverride,
.fsetid,
.fowner,
.mknod,
.netRaw,
.setgid,
.setuid,
.setfcap,
.setpcap,
.netBindService,
.sysChroot,
.kill,
.auditWrite,
],
permitted: [
.chown,
.dacOverride,
.fsetid,
.fowner,
.mknod,
.netRaw,
.setgid,
.setuid,
.setfcap,
.setpcap,
.netBindService,
.sysChroot,
.kill,
.auditWrite,
],
)
public init(
bounding: [CapabilityName] = [],
effective: [CapabilityName] = [],
inheritable: [CapabilityName] = [],
permitted: [CapabilityName] = [],
ambient: [CapabilityName] = []
) {
self.bounding = bounding
self.effective = effective
self.inheritable = inheritable
self.permitted = permitted
self.ambient = ambient
}
/// Convenience initializer that sets the same capabilities to effective, permitted, and bounding sets
/// This matches the typical pattern used by containerd/runc
public init(capabilities: [CapabilityName]) {
self.bounding = capabilities
self.effective = capabilities
self.inheritable = []
self.permitted = capabilities
self.ambient = []
}
/// Convert to OCI format for transport
public func toOCI() -> ContainerizationOCI.LinuxCapabilities {
ContainerizationOCI.LinuxCapabilities(
bounding: bounding.isEmpty ? nil : bounding.map { $0.description },
effective: effective.isEmpty ? nil : effective.map { $0.description },
inheritable: inheritable.isEmpty ? nil : inheritable.map { $0.description },
permitted: permitted.isEmpty ? nil : permitted.map { $0.description },
ambient: ambient.isEmpty ? nil : ambient.map { $0.description }
)
}
}
public struct LinuxProcessConfiguration: Sendable {
/// The default PATH value for a process.
public static let defaultPath = "/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin"
/// The arguments for the container process.
public var arguments: [String] = []
/// The environment variables for the container process.
public var environmentVariables: [String] = ["PATH=\(Self.defaultPath)"]
/// The working directory for the container process.
public var workingDirectory: String = "/"
/// The user the container process will run as.
public var user: ContainerizationOCI.User = .init()
/// The rlimits for the container process.
public var rlimits: [LinuxRLimit] = []
/// Whether to set the no_new_privileges bit on the container process. When true, the
/// process and its children cannot gain additional privileges via setuid/setgid binaries
/// or file capabilities.
public var noNewPrivileges: Bool = false
/// The Linux capabilities for the container process.
public var capabilities: LinuxCapabilities = .allCapabilities
/// Whether to allocate a pseudo terminal for the process. If you'd like interactive
/// behavior and are planning to use a terminal for stdin/out/err on the client side,
/// this should likely be set to true.
public var terminal: Bool = false
/// The stdin for the process.
public var stdin: ReaderStream?
/// The stdout for the process.
public var stdout: Writer?
/// The stderr for the process.
public var stderr: Writer?
public init() {}
public init(
arguments: [String],
environmentVariables: [String] = ["PATH=\(Self.defaultPath)"],
workingDirectory: String = "/",
user: ContainerizationOCI.User = .init(),
rlimits: [LinuxRLimit] = [],
noNewPrivileges: Bool = false,
capabilities: LinuxCapabilities = .allCapabilities,
terminal: Bool = false,
stdin: ReaderStream? = nil,
stdout: Writer? = nil,
stderr: Writer? = nil
) {
self.arguments = arguments
self.environmentVariables = environmentVariables
self.workingDirectory = workingDirectory
self.user = user
self.rlimits = rlimits
self.noNewPrivileges = noNewPrivileges
self.capabilities = capabilities
self.terminal = terminal
self.stdin = stdin
self.stdout = stdout
self.stderr = stderr
}
public init(from config: ImageConfig) {
self.workingDirectory = config.workingDir ?? "/"
self.environmentVariables = config.env ?? []
self.arguments = (config.entrypoint ?? []) + (config.cmd ?? [])
self.user = {
if let rawString = config.user {
return User(username: rawString)
}
return User()
}()
}
/// Sets up IO to be handled by the passed in Terminal, and edits the
/// process configuration to set the necessary state for using a pty.
mutating public func setTerminalIO(terminal: Terminal) {
self.environmentVariables.append("TERM=xterm")
self.terminal = true
self.stdin = terminal
self.stdout = terminal
}
func toOCI() -> ContainerizationOCI.Process {
ContainerizationOCI.Process(
args: self.arguments,
cwd: self.workingDirectory,
env: self.environmentVariables,
noNewPrivileges: self.noNewPrivileges,
capabilities: self.capabilities.toOCI(),
user: self.user,
rlimits: self.rlimits.map { $0.toOCI() },
terminal: self.terminal
)
}
}
+320
View File
@@ -0,0 +1,320 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationError
import Foundation
#if os(macOS)
import Virtualization
#endif
/// A filesystem mount exposed to a container.
public struct Mount: Sendable {
/// The filesystem or mount type. This is the string
/// that will be used for the mount syscall itself.
public var type: String
/// The source path of the mount.
public var source: String
/// The destination path of the mount.
public var destination: String
/// Filesystem or mount specific options.
public var options: [String]
/// Runtime specific options. This can be used
/// as a way to discern what kind of device a vmm
/// should create for this specific mount (virtioblock
/// virtiofs etc.).
public let runtimeOptions: RuntimeOptions
/// A type representing a "hint" of what type
/// of mount this really is (block, directory, purely
/// guest mount) and a set of type specific options, if any.
public enum RuntimeOptions: Sendable {
case virtioblk([String])
case virtiofs([String])
case shared
case any([String])
}
public init(
type: String,
source: String,
destination: String,
options: [String],
runtimeOptions: RuntimeOptions
) {
self.type = type
self.source = source
self.destination = destination
self.options = options
self.runtimeOptions = runtimeOptions
}
/// Mount representing a virtio block device.
public static func block(
format: String,
source: String,
destination: String,
options: [String] = [],
runtimeOptions: [String] = []
) -> Self {
.init(
type: format,
source: source,
destination: destination,
options: options,
runtimeOptions: .virtioblk(runtimeOptions)
)
}
/// Mount representing a virtiofs share.
public static func share(
source: String,
destination: String,
options: [String] = [],
runtimeOptions: [String] = []
) -> Self {
.init(
type: "virtiofs",
source: source,
destination: destination,
options: options,
runtimeOptions: .virtiofs(runtimeOptions)
)
}
/// A generic mount.
public static func any(
type: String,
source: String,
destination: String,
options: [String] = [],
runtimeOptions: [String] = []
) -> Self {
.init(
type: type,
source: source,
destination: destination,
options: options,
runtimeOptions: .any(runtimeOptions)
)
}
/// A mount referencing a shared pod volume by name.
public static func sharedMount(
name: String,
destination: String,
options: [String] = []
) -> Self {
.init(
type: "none",
source: name,
destination: destination,
options: options,
runtimeOptions: .shared
)
}
#if os(macOS)
/// Clone the Mount to the provided path.
///
/// This uses `clonefile` to provide a copy-on-write copy of the Mount.
public func clone(to: String) throws -> Self {
let fm = FileManager.default
let src = self.source
try fm.copyItem(atPath: src, toPath: to)
return .init(
type: self.type,
source: to,
destination: self.destination,
options: self.options,
runtimeOptions: self.runtimeOptions
)
}
#endif
}
#if os(macOS)
extension Mount {
private enum StorageAttachmentType {
case diskImage
case networkBlockDevice
}
private var storageAttachmentType: StorageAttachmentType {
let nbdSchemes = ["nbd://", "nbds://", "nbd+unix://", "nbds+unix://"]
if nbdSchemes.contains(where: { self.source.hasPrefix($0) }) {
return .networkBlockDevice
}
return .diskImage
}
func configure(config: inout VZVirtualMachineConfiguration) throws {
switch self.runtimeOptions {
case .virtioblk(let options):
let device: VZStorageDeviceAttachment
switch self.storageAttachmentType {
case .networkBlockDevice:
device = try VZNetworkBlockDeviceStorageDeviceAttachment.mountToVZAttachment(mount: self, options: options)
case .diskImage:
device = try VZDiskImageStorageDeviceAttachment.mountToVZAttachment(mount: self, options: options)
}
let attachment = VZVirtioBlockDeviceConfiguration(attachment: device)
config.storageDevices.append(attachment)
case .virtiofs(_):
// VirtioFS mounts are handled centrally via VZMultipleDirectoryShare in VZVirtualMachineInstance
// No per-mount device configuration needed
break
case .shared, .any:
break
}
}
}
extension VZDiskImageStorageDeviceAttachment {
static func mountToVZAttachment(mount: Mount, options: [String]) throws -> VZDiskImageStorageDeviceAttachment {
var synchronizationMode: VZDiskImageSynchronizationMode = .fsync
var cachingMode: VZDiskImageCachingMode = .cached
for option in options {
let split = option.split(separator: "=")
if split.count != 2 {
continue
}
let key = String(split[0])
let value = String(split[1])
switch key {
case "vzDiskImageCachingMode":
switch value {
case "automatic":
cachingMode = .automatic
case "cached":
cachingMode = .cached
case "uncached":
cachingMode = .uncached
default:
throw ContainerizationError(
.invalidArgument,
message: "unknown vzDiskImageCachingMode value for virtio block device: \(value)"
)
}
case "vzDiskImageSynchronizationMode":
switch value {
case "full":
synchronizationMode = .full
case "fsync":
synchronizationMode = .fsync
case "none":
synchronizationMode = .none
default:
throw ContainerizationError(
.invalidArgument,
message: "unknown vzDiskImageSynchronizationMode value for virtio block device: \(value)"
)
}
default:
throw ContainerizationError(
.invalidArgument,
message: "unknown vmm option encountered: \(key)"
)
}
}
return try VZDiskImageStorageDeviceAttachment(
url: URL(filePath: mount.source),
readOnly: mount.readonly,
cachingMode: cachingMode,
synchronizationMode: synchronizationMode
)
}
}
extension VZNetworkBlockDeviceStorageDeviceAttachment {
static func mountToVZAttachment(mount: Mount, options: [String]) throws -> VZNetworkBlockDeviceStorageDeviceAttachment {
guard let url = URL(string: mount.source) else {
throw ContainerizationError(
.invalidArgument,
message: "invalid NBD URL: \(mount.source)"
)
}
var timeout: TimeInterval = 5
var synchronizationMode: VZDiskSynchronizationMode = .full
for option in options {
let split = option.split(separator: "=")
if split.count != 2 {
continue
}
let key = String(split[0])
let value = String(split[1])
switch key {
case "vzTimeout":
guard let t = TimeInterval(value) else {
throw ContainerizationError(
.invalidArgument,
message: "invalid vzTimeout value for NBD device: \(value)"
)
}
timeout = t
case "vzSynchronizationMode":
switch value {
case "full":
synchronizationMode = .full
case "none":
synchronizationMode = .none
default:
throw ContainerizationError(
.invalidArgument,
message: "unknown vzSynchronizationMode value for NBD device: \(value)"
)
}
default:
throw ContainerizationError(
.invalidArgument,
message: "unknown vmm option encountered: \(key)"
)
}
}
return try VZNetworkBlockDeviceStorageDeviceAttachment(
url: url,
timeout: timeout,
isForcedReadOnly: mount.readonly,
synchronizationMode: synchronizationMode
)
}
}
#endif
extension Mount {
fileprivate var readonly: Bool {
self.options.contains("ro")
}
/// Returns true if this mount is a virtio block device.
public var isBlock: Bool {
if case .virtioblk = self.runtimeOptions {
return true
}
return false
}
}
@@ -0,0 +1,42 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationExtras
public struct NATInterface: Interface {
public var ipv4Address: CIDRv4
public var ipv4Gateway: IPv4Address?
public var ipv6Address: CIDRv6?
public var ipv6Gateway: IPv6Address?
public var macAddress: MACAddress?
public var mtu: UInt32
public init(
ipv4Address: CIDRv4,
ipv4Gateway: IPv4Address?,
ipv6Address: CIDRv6? = nil,
ipv6Gateway: IPv6Address? = nil,
macAddress: MACAddress? = nil,
mtu: UInt32 = 1500
) {
self.ipv4Address = ipv4Address
self.ipv4Gateway = ipv4Gateway
self.ipv6Address = ipv6Address
self.ipv6Gateway = ipv6Gateway
self.macAddress = macAddress
self.mtu = mtu
}
}
@@ -0,0 +1,84 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
#if os(macOS)
import vmnet
import Virtualization
import ContainerizationError
import ContainerizationExtras
import Synchronization
/// An interface that uses NAT to provide an IP address for a given
/// container/virtual machine.
@available(macOS 26, *)
public final class NATNetworkInterface: Interface, Sendable {
public let ipv4Address: CIDRv4
public let ipv4Gateway: IPv4Address?
public let macAddress: MACAddress?
public let mtu: UInt32
@available(macOS 26, *)
// `reference` isn't used concurrently.
public nonisolated(unsafe) let reference: vmnet_network_ref!
@available(macOS 26, *)
public init(
ipv4Address: CIDRv4,
ipv4Gateway: IPv4Address?,
reference: sending vmnet_network_ref,
macAddress: MACAddress? = nil,
mtu: UInt32 = 1500
) {
self.ipv4Address = ipv4Address
self.ipv4Gateway = ipv4Gateway
self.macAddress = macAddress
self.mtu = mtu
self.reference = reference
}
@available(macOS, obsoleted: 26, message: "Use init(ipv4Address:ipv4Gateway:reference:macAddress:) instead")
public init(
ipv4Address: CIDRv4,
ipv4Gateway: IPv4Address?,
macAddress: MACAddress? = nil,
mtu: UInt32 = 1500
) {
self.ipv4Address = ipv4Address
self.ipv4Gateway = ipv4Gateway
self.macAddress = macAddress
self.mtu = mtu
self.reference = nil
}
}
@available(macOS 26, *)
extension NATNetworkInterface: VZInterface {
public func device() throws -> VZVirtioNetworkDeviceConfiguration {
let config = VZVirtioNetworkDeviceConfiguration()
if let macAddress = self.macAddress {
guard let mac = VZMACAddress(string: macAddress.description) else {
throw ContainerizationError(.invalidArgument, message: "invalid mac address \(macAddress)")
}
config.macAddress = mac
}
config.attachment = VZVmnetNetworkDeviceAttachment(network: self.reference)
return config
}
}
#endif
+21
View File
@@ -0,0 +1,21 @@
//===----------------------------------------------------------------------===//
// Copyright © 2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
/// A network that can allocate and release interfaces for use with containers.
public protocol Network: Sendable {
mutating func createInterface(_ id: String) throws -> Interface?
mutating func releaseInterface(_ id: String) throws
}
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,493 @@
syntax = "proto3";
package com.apple.containerization.sandbox.v3;
import "google/protobuf/timestamp.proto";
// Context for interacting with a container's runtime environment.
service SandboxContext {
// Mount a filesystem.
rpc Mount(MountRequest) returns (MountResponse);
// Unmount a filesystem.
rpc Umount(UmountRequest) returns (UmountResponse);
// Set an environment variable on the init process.
rpc Setenv(SetenvRequest) returns (SetenvResponse);
// Get an environment variable from the init process.
rpc Getenv(GetenvRequest) returns (GetenvResponse);
// Create a new directory inside the sandbox.
rpc Mkdir(MkdirRequest) returns (MkdirResponse);
// Set sysctls in the context of the sandbox.
rpc Sysctl(SysctlRequest) returns (SysctlResponse);
// Set time in the guest.
rpc SetTime(SetTimeRequest) returns (SetTimeResponse);
// Set up an emulator in the guest for a specific binary format.
rpc SetupEmulator(SetupEmulatorRequest) returns (SetupEmulatorResponse);
// Write data to an existing or new file.
rpc WriteFile(WriteFileRequest) returns (WriteFileResponse);
// Copy a file or directory between the host and guest.
// Data transfer happens over a dedicated vsock connection;
// the gRPC stream is used only for control/metadata.
rpc Copy(CopyRequest) returns (stream CopyResponse);
// Stat a path in the guest filesystem.
rpc Stat(StatRequest) returns (StatResponse);
// Perform a filesystem operation on a mounted filesystem.
rpc FilesystemOperation(FilesystemOperationRequest) returns (FilesystemOperationResponse);
// Create a new process inside the container.
rpc CreateProcess(CreateProcessRequest) returns (CreateProcessResponse);
// Delete an existing process inside the container.
rpc DeleteProcess(DeleteProcessRequest) returns (DeleteProcessResponse);
// Start the provided process.
rpc StartProcess(StartProcessRequest) returns (StartProcessResponse);
// Send a signal to the provided process.
rpc KillProcess(KillProcessRequest) returns (KillProcessResponse);
// Wait for a process to exit and return the exit code.
rpc WaitProcess(WaitProcessRequest) returns (WaitProcessResponse);
// Resize the tty of a given process. This will error if the process does
// not have a pty allocated.
rpc ResizeProcess(ResizeProcessRequest) returns (ResizeProcessResponse);
// Close IO for a given process.
rpc CloseProcessStdin(CloseProcessStdinRequest) returns (CloseProcessStdinResponse);
// Get statistics for containers.
rpc ContainerStatistics(ContainerStatisticsRequest) returns (ContainerStatisticsResponse);
// Proxy a vsock port to a unix domain socket in the guest, or vice versa.
rpc ProxyVsock(ProxyVsockRequest) returns (ProxyVsockResponse);
// Stop a vsock proxy to a unix domain socket.
rpc StopVsockProxy(StopVsockProxyRequest) returns (StopVsockProxyResponse);
// Set the link state of a network interface.
rpc IpLinkSet(IpLinkSetRequest) returns (IpLinkSetResponse);
// Add an IPv4 address to a network interface.
rpc IpAddrAdd(IpAddrAddRequest) returns (IpAddrAddResponse);
// Add an IP route for a network interface.
rpc IpRouteAddLink(IpRouteAddLinkRequest) returns (IpRouteAddLinkResponse);
// Add an IP route for a network interface.
rpc IpRouteAddDefault(IpRouteAddDefaultRequest) returns (IpRouteAddDefaultResponse);
// Configure DNS resolver.
rpc ConfigureDns(ConfigureDnsRequest) returns (ConfigureDnsResponse);
// Configure /etc/hosts.
rpc ConfigureHosts(ConfigureHostsRequest) returns (ConfigureHostsResponse);
// Perform the sync syscall.
rpc Sync(SyncRequest) returns (SyncResponse);
// Send a signal to a process via the PID.
rpc Kill(KillRequest) returns (KillResponse);
}
message Stdio {
optional int32 stdinPort = 1;
optional int32 stdoutPort = 2;
optional int32 stderrPort = 3;
}
message SetupEmulatorRequest {
string binary_path = 1;
string name = 2;
string type = 3;
string offset = 4;
string magic = 5;
string mask = 6;
string flags = 7;
}
message SetupEmulatorResponse {}
message SetTimeRequest {
int64 sec = 1;
int32 usec = 2;
}
message SetTimeResponse {}
message SysctlRequest { map<string, string> settings = 1; }
message SysctlResponse {}
message ProxyVsockRequest {
enum Action {
INTO = 0;
OUT_OF = 1;
}
string id = 1;
uint32 vsock_port = 2;
string guestPath = 3;
optional uint32 guestSocketPermissions = 4;
Action action = 5;
}
message ProxyVsockResponse {}
message StopVsockProxyRequest { string id = 1; }
message StopVsockProxyResponse {}
message MountRequest {
string type = 1;
string source = 2;
string destination = 3;
repeated string options = 4;
}
message MountResponse {}
message UmountRequest {
string path = 1;
int32 flags = 2;
}
message UmountResponse {}
message SetenvRequest {
string key = 1;
optional string value = 2;
}
message SetenvResponse {}
message GetenvRequest { string key = 1; }
message GetenvResponse { optional string value = 1; }
message CreateProcessRequest {
string id = 1;
optional string containerID = 2;
optional uint32 stdin = 3;
optional uint32 stdout = 4;
optional uint32 stderr = 5;
optional string ociRuntimePath = 6;
bytes configuration = 7;
optional bytes options = 8;
}
message CreateProcessResponse {}
message WaitProcessRequest {
string id = 1;
optional string containerID = 2;
}
message WaitProcessResponse {
int32 exitCode = 1;
google.protobuf.Timestamp exited_at = 2;
}
message ResizeProcessRequest {
string id = 1;
optional string containerID = 2;
uint32 rows = 3;
uint32 columns = 4;
}
message ResizeProcessResponse {}
message DeleteProcessRequest {
string id = 1;
optional string containerID = 2;
}
message DeleteProcessResponse {}
message StartProcessRequest {
string id = 1;
optional string containerID = 2;
}
message StartProcessResponse { int32 pid = 1; }
message KillProcessRequest {
string id = 1;
optional string containerID = 2;
int32 signal = 3;
}
message KillProcessResponse { int32 result = 1; }
message CloseProcessStdinRequest {
string id = 1;
optional string containerID = 2;
}
message CloseProcessStdinResponse {}
message MkdirRequest {
string path = 1;
bool all = 2;
uint32 perms = 3;
}
message MkdirResponse {}
message WriteFileRequest {
message WriteFileFlags {
bool create_parent_dirs = 1;
bool append = 2;
bool create_if_missing = 3;
}
string path = 1;
bytes data = 2;
uint32 mode = 3;
WriteFileFlags flags = 4;
}
message WriteFileResponse {}
message CopyRequest {
enum Direction {
// Copy from host into guest.
COPY_IN = 0;
// Copy from guest to host.
COPY_OUT = 1;
}
// Direction of the copy operation.
Direction direction = 1;
// Path in the guest (destination for COPY_IN, source for COPY_OUT).
string path = 2;
// File mode for single-file COPY_IN (defaults to 0644 if not set).
uint32 mode = 3;
// Create parent directories if they don't exist.
bool create_parents = 4;
// Vsock port the host is listening on for data transfer.
uint32 vsock_port = 5;
// For COPY_IN: indicates the data arriving on vsock is a tar+gzip archive.
bool is_archive = 6;
}
message CopyResponse {
enum Status {
// Transfer metadata (first message for COPY_OUT: is_archive, total_size).
METADATA = 0;
// Data transfer completed successfully.
COMPLETE = 1;
}
// What this response represents.
Status status = 1;
// For COPY_OUT METADATA: indicates the data on vsock will be a tar+gzip archive.
bool is_archive = 2;
// For COPY_OUT METADATA: total size in bytes (0 if unknown, e.g. for archives).
uint64 total_size = 3;
// Non-empty if an error occurred.
string error = 4;
}
message StatRequest { string path = 1; }
message Stat {
uint64 dev = 1; // st_dev: ID of device containing file
uint64 ino = 2; // st_ino: inode number
uint32 mode = 3; // st_mode: file type and mode (permissions)
uint64 nlink = 4; // st_nlink: number of hard links
uint32 uid = 5; // st_uid: user ID of owner
uint32 gid = 6; // st_gid: group ID of owner
uint64 rdev = 7; // st_rdev: device ID (if special file)
int64 size = 8; // st_size: total size in bytes
int64 blksize = 9; // st_blksize: preferred block size for filesystem I/O
int64 blocks = 10; // st_blocks: number of 512-byte blocks allocated
google.protobuf.Timestamp atime = 11; // st_atim: time of last access
google.protobuf.Timestamp mtime = 12; // st_mtim: time of last modification
google.protobuf.Timestamp ctime = 13; // st_ctim: time of last status change
}
message StatResponse {
Stat stat = 1;
string error = 2; // Non-empty if stat failed.
}
message FiTrimParams {
oneof schedule {
OneShot one_shot = 1;
}
message OneShot {}
}
message FiFreezeParams {}
message FiThawParams {}
message FiTrimResult {
uint64 trimmed_bytes = 1;
}
message FilesystemOperationRequest {
string path = 1;
oneof operation {
FiTrimParams trim = 2;
FiFreezeParams freeze = 3;
FiThawParams thaw = 4;
}
}
message FilesystemOperationResponse {
oneof result {
FiTrimResult trim = 1;
}
}
message IpLinkSetRequest {
string interface = 1;
bool up = 2;
optional uint32 mtu = 3;
}
message IpLinkSetResponse {}
message IpAddrAddRequest {
string interface = 1;
string ipv4Address = 2;
optional string ipv6Address = 3;
}
message IpAddrAddResponse {}
message IpRouteAddLinkRequest {
string interface = 1;
string dstIpv4Addr = 2;
string srcIpv4Addr = 3;
optional string dstIpv6Addr = 4;
optional string srcIpv6Addr = 5;
}
message IpRouteAddLinkResponse {}
message IpRouteAddDefaultRequest {
string interface = 1;
string ipv4Gateway = 2;
optional string ipv6Gateway = 3;
}
message IpRouteAddDefaultResponse {}
message ConfigureDnsRequest {
string location = 1;
repeated string nameservers = 2;
optional string domain = 3;
repeated string searchDomains = 4;
repeated string options = 5;
}
message ConfigureDnsResponse {}
message ConfigureHostsRequest {
message HostsEntry {
string ipAddress = 1;
repeated string hostnames = 2;
optional string comment = 3;
}
string location = 1;
repeated HostsEntry entries = 2;
optional string comment = 3;
}
message ConfigureHostsResponse {}
message SyncRequest {}
message SyncResponse {}
message KillRequest {
int32 pid = 1;
int32 signal = 3;
}
message KillResponse { int32 result = 1; }
// Categories of statistics that can be requested.
enum StatCategory {
STAT_CATEGORY_UNSPECIFIED = 0;
STAT_CATEGORY_PROCESS = 1;
STAT_CATEGORY_MEMORY = 2;
STAT_CATEGORY_CPU = 3;
STAT_CATEGORY_BLOCK_IO = 4;
STAT_CATEGORY_NETWORK = 5;
STAT_CATEGORY_MEMORY_EVENTS = 6;
}
message ContainerStatisticsRequest {
repeated string container_ids = 1; // Empty = all containers
repeated StatCategory categories = 2; // Empty = all categories
}
message ContainerStatisticsResponse {
repeated ContainerStats containers = 1;
}
message ContainerStats {
string container_id = 1;
ProcessStats process = 2;
MemoryStats memory = 3;
CPUStats cpu = 4;
BlockIOStats block_io = 5;
repeated NetworkStats networks = 6;
MemoryEventStats memory_events = 7;
}
message ProcessStats {
uint64 current = 1;
uint64 limit = 2; // 0 or max value = unlimited
}
message MemoryStats {
uint64 usage_bytes = 1;
uint64 limit_bytes = 2;
uint64 swap_usage_bytes = 3;
uint64 swap_limit_bytes = 4;
uint64 cache_bytes = 5;
uint64 kernel_stack_bytes = 6;
uint64 slab_bytes = 7;
uint64 page_faults = 8;
uint64 major_page_faults = 9;
uint64 inactive_file = 10;
uint64 anon = 11;
uint64 workingset_refault_anon = 12;
uint64 workingset_refault_file = 13;
uint64 pgsteal_kswapd = 14;
uint64 pgsteal_direct = 15;
uint64 pgsteal_khugepaged = 16;
}
message CPUStats {
uint64 usage_usec = 1;
uint64 user_usec = 2;
uint64 system_usec = 3;
uint64 throttling_periods = 4;
uint64 throttled_periods = 5;
uint64 throttled_time_usec = 6;
}
message BlockIOStats {
repeated BlockIOEntry devices = 1;
}
message BlockIOEntry {
uint64 major = 1;
uint64 minor = 2;
uint64 read_bytes = 3;
uint64 write_bytes = 4;
uint64 read_operations = 5;
uint64 write_operations = 6;
}
message NetworkStats {
string interface = 1;
uint64 receivedPackets = 2;
uint64 transmittedPackets = 3;
uint64 receivedBytes = 4;
uint64 transmittedBytes = 5;
uint64 receivedErrors = 6;
uint64 transmittedErrors = 7;
}
// Memory event counters from cgroup2's memory.events file.
message MemoryEventStats {
// Number of times the cgroup was reclaimed due to low memory.
uint64 low = 1;
// Number of times the cgroup exceeded its high memory limit.
uint64 high = 2;
// Number of times the cgroup hit its max memory limit.
uint64 max = 3;
// Number of times the cgroup triggered OOM.
uint64 oom = 4;
// Number of processes killed by OOM killer.
uint64 oom_kill = 5;
// Number of times charge for memory failed because of limit.
uint64 oom_group_kill = 6;
}
+302
View File
@@ -0,0 +1,302 @@
//===----------------------------------------------------------------------===//
// Copyright © 2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
#if canImport(Darwin)
import Darwin
#elseif canImport(Glibc)
import Glibc
#elseif canImport(Musl)
import Musl
#else
#error("Signal not supported on this platform.")
#endif
/// A unix signal.
public struct Signal: RawRepresentable, Hashable, Sendable {
public let rawValue: Int32
public init(rawValue: Int32) {
self.rawValue = rawValue
}
/// Parse a signal from a string representation (e.g. "SIGKILL", "KILL", "9").
public init(_ name: String, from map: [String: Int32] = Signal.linux) throws {
var signalUpper = name.uppercased()
signalUpper.trimPrefix("SIG")
if let sig = Int32(signalUpper) {
if !map.values.contains(sig) {
throw SignalError.invalidSignal(name)
}
self.rawValue = sig
return
}
guard let sig = map[signalUpper] else {
throw SignalError.invalidSignal(name)
}
self.rawValue = sig
}
// Signals that are commonly sent to containers and share the same
// number across macOS/Linux.
public static let hup = Signal(rawValue: 1)
public static let int = Signal(rawValue: 2)
public static let quit = Signal(rawValue: 3)
public static let kill = Signal(rawValue: 9)
public static let term = Signal(rawValue: 15)
public static let winch = Signal(rawValue: 28)
/// Linux signals.
public enum Linux {
public static let hup = Signal(rawValue: 1)
public static let int = Signal(rawValue: 2)
public static let quit = Signal(rawValue: 3)
public static let ill = Signal(rawValue: 4)
public static let trap = Signal(rawValue: 5)
public static let abrt = Signal(rawValue: 6)
public static let bus = Signal(rawValue: 7)
public static let fpe = Signal(rawValue: 8)
public static let kill = Signal(rawValue: 9)
public static let usr1 = Signal(rawValue: 10)
public static let segv = Signal(rawValue: 11)
public static let usr2 = Signal(rawValue: 12)
public static let pipe = Signal(rawValue: 13)
public static let alrm = Signal(rawValue: 14)
public static let term = Signal(rawValue: 15)
public static let stkflt = Signal(rawValue: 16)
public static let chld = Signal(rawValue: 17)
public static let cont = Signal(rawValue: 18)
public static let stop = Signal(rawValue: 19)
public static let tstp = Signal(rawValue: 20)
public static let ttin = Signal(rawValue: 21)
public static let ttou = Signal(rawValue: 22)
public static let urg = Signal(rawValue: 23)
public static let xcpu = Signal(rawValue: 24)
public static let xfsz = Signal(rawValue: 25)
public static let vtalrm = Signal(rawValue: 26)
public static let prof = Signal(rawValue: 27)
public static let winch = Signal(rawValue: 28)
public static let io = Signal(rawValue: 29)
public static let poll = Signal(rawValue: 29)
public static let pwr = Signal(rawValue: 30)
public static let sys = Signal(rawValue: 31)
public static func rtmin(offset: Int32 = 0) -> Signal {
Signal(rawValue: 34 + offset)
}
public static let rtmax = Signal(rawValue: 64)
}
/// Darwin signals.
public enum Darwin {
public static let hup = Signal(rawValue: 1)
public static let int = Signal(rawValue: 2)
public static let quit = Signal(rawValue: 3)
public static let ill = Signal(rawValue: 4)
public static let trap = Signal(rawValue: 5)
public static let abrt = Signal(rawValue: 6)
public static let emt = Signal(rawValue: 7)
public static let fpe = Signal(rawValue: 8)
public static let kill = Signal(rawValue: 9)
public static let bus = Signal(rawValue: 10)
public static let segv = Signal(rawValue: 11)
public static let sys = Signal(rawValue: 12)
public static let pipe = Signal(rawValue: 13)
public static let alrm = Signal(rawValue: 14)
public static let term = Signal(rawValue: 15)
public static let urg = Signal(rawValue: 16)
public static let stop = Signal(rawValue: 17)
public static let tstp = Signal(rawValue: 18)
public static let cont = Signal(rawValue: 19)
public static let chld = Signal(rawValue: 20)
public static let ttin = Signal(rawValue: 21)
public static let ttou = Signal(rawValue: 22)
public static let io = Signal(rawValue: 23)
public static let xcpu = Signal(rawValue: 24)
public static let xfsz = Signal(rawValue: 25)
public static let vtalrm = Signal(rawValue: 26)
public static let prof = Signal(rawValue: 27)
public static let winch = Signal(rawValue: 28)
public static let info = Signal(rawValue: 29)
public static let usr1 = Signal(rawValue: 30)
public static let usr2 = Signal(rawValue: 31)
}
/// All Linux signals including real-time signals (RTMIN through RTMAX).
public static let linux: [String: Int32] = [
"ABRT": 6,
"ALRM": 14,
"BUS": 7,
"CHLD": 17,
"CLD": 17,
"CONT": 18,
"FPE": 8,
"HUP": 1,
"ILL": 4,
"INT": 2,
"IO": 29,
"IOT": 6,
"KILL": 9,
"PIPE": 13,
"POLL": 29,
"PROF": 27,
"PWR": 30,
"QUIT": 3,
"SEGV": 11,
"STKFLT": 16,
"STOP": 19,
"SYS": 31,
"TERM": 15,
"TRAP": 5,
"TSTP": 20,
"TTIN": 21,
"TTOU": 22,
"URG": 23,
"USR1": 10,
"USR2": 12,
"VTALRM": 26,
"WINCH": 28,
"XCPU": 24,
"XFSZ": 25,
"RTMIN": 34,
"RTMIN+1": 35,
"RTMIN+2": 36,
"RTMIN+3": 37,
"RTMIN+4": 38,
"RTMIN+5": 39,
"RTMIN+6": 40,
"RTMIN+7": 41,
"RTMIN+8": 42,
"RTMIN+9": 43,
"RTMIN+10": 44,
"RTMIN+11": 45,
"RTMIN+12": 46,
"RTMIN+13": 47,
"RTMIN+14": 48,
"RTMIN+15": 49,
"RTMIN+16": 50,
"RTMIN+17": 51,
"RTMIN+18": 52,
"RTMIN+19": 53,
"RTMIN+20": 54,
"RTMIN+21": 55,
"RTMIN+22": 56,
"RTMIN+23": 57,
"RTMIN+24": 58,
"RTMIN+25": 59,
"RTMIN+26": 60,
"RTMIN+27": 61,
"RTMIN+28": 62,
"RTMIN+29": 63,
"RTMAX": 64,
]
}
#if os(macOS)
extension Signal {
/// All signals for the macOS host.
public static let platform: [String: Int32] = [
"ABRT": SIGABRT,
"ALRM": SIGALRM,
"BUS": SIGBUS,
"CHLD": SIGCHLD,
"CONT": SIGCONT,
"EMT": SIGEMT,
"FPE": SIGFPE,
"HUP": SIGHUP,
"ILL": SIGILL,
"INFO": SIGINFO,
"INT": SIGINT,
"IO": SIGIO,
"IOT": SIGIOT,
"KILL": SIGKILL,
"PIPE": SIGPIPE,
"PROF": SIGPROF,
"QUIT": SIGQUIT,
"SEGV": SIGSEGV,
"STOP": SIGSTOP,
"SYS": SIGSYS,
"TERM": SIGTERM,
"TRAP": SIGTRAP,
"TSTP": SIGTSTP,
"TTIN": SIGTTIN,
"TTOU": SIGTTOU,
"URG": SIGURG,
"USR1": SIGUSR1,
"USR2": SIGUSR2,
"VTALRM": SIGVTALRM,
"WINCH": SIGWINCH,
"XCPU": SIGXCPU,
"XFSZ": SIGXFSZ,
]
}
#elseif os(Linux)
extension Signal {
/// All signals for the Linux host.
public static let platform = linux
}
#endif
extension Signal {
private static let platformToName: [Int32: String] =
Dictionary(Signal.platform.map { ($0.value, $0.key) }, uniquingKeysWith: { first, _ in first })
/// Returns the canonical name for this signal on the current platform.
public func platformName() -> String? {
Self.platformName(self.rawValue)
}
/// Returns the canonical name for a signal number on the current platform.
public static func platformName(_ signal: Int32) -> String? {
platformToName[signal]
}
}
#if os(macOS)
extension Signal {
/// Converts a macOS signal to the equivalent Linux signal.
public func linuxSignal() -> Signal? {
guard let name = Self.platformToName[self.rawValue],
let linuxNumber = Signal.linux[name]
else {
return nil
}
return Signal(rawValue: linuxNumber)
}
}
#endif
extension Signal: ExpressibleByIntegerLiteral {
public init(integerLiteral value: Int32) {
self.rawValue = value
}
}
/// Errors that can be encountered for converting signals.
public enum SignalError: Error, CustomStringConvertible {
case invalidSignal(String)
public var description: String {
switch self {
case .invalidSignal(let sig):
return "invalid signal: \(sig)"
}
}
}
@@ -0,0 +1,41 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationOCI
/// `SystemPlatform` describes an operating system and architecture pair.
/// This is primarily used to choose what kind of OCI image to pull from a
/// registry.
public struct SystemPlatform: Sendable, Codable {
public enum OS: String, CaseIterable, Sendable, Codable {
case linux
case darwin
}
public let os: OS
public enum Architecture: String, CaseIterable, Sendable, Codable {
case arm64
case amd64
}
public let architecture: Architecture
public func ociPlatform() -> ContainerizationOCI.Platform {
ContainerizationOCI.Platform(arch: architecture.rawValue, os: os.rawValue)
}
public static var linuxArm: SystemPlatform { .init(os: .linux, architecture: .arm64) }
public static var linuxAmd: SystemPlatform { .init(os: .linux, architecture: .amd64) }
}
+87
View File
@@ -0,0 +1,87 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import Foundation
import Logging
actor TimeSyncer {
private var task: Task<Void, Never>?
private var context: Vminitd?
private var paused: Bool
private let logger: Logger?
init(logger: Logger?) {
self.paused = false
self.logger = logger
}
func start(context: Vminitd, interval: Duration = .seconds(30)) {
guard self.task == nil else {
return
}
self.context = context
self.task = Task {
while true {
do {
do {
try await Task.sleep(for: interval)
} catch {
return
}
guard !paused else {
continue
}
var timeval = timeval()
guard gettimeofday(&timeval, nil) == 0 else {
throw POSIXError.fromErrno()
}
try await context.setTime(
sec: Int64(timeval.tv_sec),
usec: Int32(timeval.tv_usec)
)
} catch {
self.logger?.error("failed to sync time with guest agent: \(error)")
}
}
}
}
func pause() async {
self.paused = true
}
func resume() async {
self.paused = false
}
func close() async throws {
guard let task else {
// Already closed, nop.
return
}
task.cancel()
await task.value
try await self.context?.close()
self.task = nil
self.context = nil
}
}
@@ -0,0 +1,70 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import Foundation
import SystemPackage
/// Represents a UnixSocket that can be shared into or out of a container/guest.
public struct UnixSocketConfiguration: Sendable {
// TODO: Realistically, we can just hash this struct and use it as the "id".
/// The unique identifier for this socket configuration.
public var id: String {
_id
}
private let _id = UUID().uuidString
/// The path to the socket you'd like relayed. For .into
/// direction this should be the path on the host to a unix socket.
/// For direction .outOf this should be the path in the container/guest
/// to a unix socket.
public var source: URL
/// The path you'd like the socket to be relayed to. For .into
/// direction this should be the path in the container/guest. For
/// direction .outOf this should be the path on your host.
public var destination: URL
/// What to set the file permissions of the unix socket being created
/// to. For .into direction this will be the socket in the guest. For
/// .outOf direction this will be the socket on the host.
public var permissions: FilePermissions?
/// The direction of the relay. `.into` for sharing a unix socket on your
/// host into the container/guest. `outOf` shares a socket in the container/guest
/// onto your host.
public var direction: Direction
/// Type that denotes the direction of the unix socket relay.
public enum Direction: Sendable {
/// Share the socket into the container/guest.
case into
/// Share a socket in the container/guest onto the host.
case outOf
}
public init(
source: URL,
destination: URL,
permissions: FilePermissions? = nil,
direction: Direction = .into
) {
self.source = source
self.destination = destination
self.permissions = permissions
self.direction = direction
}
}
@@ -0,0 +1,243 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationError
import ContainerizationIO
import ContainerizationOS
import Foundation
import Logging
import Synchronization
package final class UnixSocketRelay: Sendable {
private let port: UInt32
private let configuration: UnixSocketConfiguration
private let vm: any VirtualMachineInstance
private let log: Logger?
private let state: Mutex<State>
private struct State {
var activeRelays: [String: BidirectionalRelay] = [:]
var t: Task<(), Never>? = nil
var listener: VsockListener? = nil
}
init(
port: UInt32,
socket: UnixSocketConfiguration,
vm: any VirtualMachineInstance,
log: Logger? = nil
) throws {
self.port = port
self.configuration = socket
self.vm = vm
self.log = log
self.state = Mutex<State>(.init())
}
deinit {
state.withLock { $0.t?.cancel() }
}
}
extension UnixSocketRelay {
func start() async throws {
switch configuration.direction {
case .outOf:
try await setupHostVsockDial()
case .into:
try setupHostVsockListener()
}
}
func stop() throws {
try state.withLock {
guard let t = $0.t else {
throw ContainerizationError(
.invalidState,
message: "failed to stop socket relay: relay has not been started"
)
}
t.cancel()
$0.t = nil
for (_, relay) in $0.activeRelays {
relay.stop()
}
$0.activeRelays.removeAll()
switch configuration.direction {
case .outOf:
// If we created the host conn, lets unlink it also. It's possible it was
// already unlinked if the relay failed earlier.
try? FileManager.default.removeItem(at: self.configuration.destination)
case .into:
try $0.listener?.finish()
}
}
}
private func setupHostVsockDial() async throws {
let hostConn = configuration.destination
let socketType = try UnixType(
path: hostConn.path,
unlinkExisting: true
)
let hostSocket = try Socket(type: socketType)
try hostSocket.listen()
log?.info(
"listening on host UDS",
metadata: [
"path": "\(hostConn.path)",
"vport": "\(port)",
])
let connectionStream = try hostSocket.acceptStream(closeOnDeinit: false)
state.withLock {
$0.t = Task {
do {
for try await connection in connectionStream {
try await self.handleHostUnixConn(
hostConn: connection,
port: self.port,
vm: self.vm,
log: self.log
)
}
} catch {
log?.error("failed in unix socket relay loop: \(error)")
}
try? FileManager.default.removeItem(at: hostConn)
}
}
}
private func setupHostVsockListener() throws {
let hostPath = configuration.source
let listener = try vm.listen(port)
log?.info(
"listening on guest vsock",
metadata: [
"path": "\(hostPath)",
"vport": "\(port)",
])
state.withLock {
$0.listener = listener
$0.t = Task {
do {
defer { try? listener.finish() }
for await connection in listener {
try await self.handleGuestVsockConn(
vsockConn: connection,
hostConnectionPath: hostPath,
port: self.port,
log: self.log
)
}
} catch {
self.log?.error("failed to setup relay between vsock \(self.port) and \(hostPath.path): \(error)")
}
}
}
}
private func handleHostUnixConn(
hostConn: ContainerizationOS.Socket,
port: UInt32,
vm: any VirtualMachineInstance,
log: Logger?
) async throws {
do {
let guestConn = try await vm.dial(port)
log?.debug(
"initiating connection from host to guest",
metadata: [
"vport": "\(port)",
"hostFd": "\(guestConn.fileDescriptor)",
"guestFd": "\(hostConn.fileDescriptor)",
])
try await self.relay(
hostConn: hostConn,
guestFd: guestConn.fileDescriptor
)
} catch {
log?.error("failed to relay between vsock \(port) and \(hostConn)")
throw error
}
}
private func handleGuestVsockConn(
vsockConn: FileHandle,
hostConnectionPath: URL,
port: UInt32,
log: Logger?
) async throws {
let hostPath = hostConnectionPath.path
let socketType = try UnixType(path: hostPath)
let hostSocket = try Socket(
type: socketType,
closeOnDeinit: false
)
log?.debug(
"initiating connection from guest to host",
metadata: [
"vport": "\(port)",
"hostFd": "\(hostSocket.fileDescriptor)",
"guestFd": "\(vsockConn.fileDescriptor)",
])
try hostSocket.connect()
do {
try await self.relay(
hostConn: hostSocket,
guestFd: vsockConn.fileDescriptor
)
} catch {
log?.error("failed to relay between vsock \(port) and \(hostPath)")
}
}
private func relay(
hostConn: Socket,
guestFd: Int32
) async throws {
let hostFd = hostConn.fileDescriptor
let relayID = UUID().uuidString
let relay = BidirectionalRelay(
fd1: hostFd,
fd2: guestFd,
log: log
)
state.withLock {
$0.activeRelays[relayID] = relay
}
do {
try relay.start()
} catch {
state.withLock { $0.activeRelays[relayID] = nil }
throw error
}
Task {
await relay.waitForCompletion()
state.withLock { $0.activeRelays[relayID] = nil }
}
}
}
@@ -0,0 +1,73 @@
//===----------------------------------------------------------------------===//
// Copyright © 2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationError
import Foundation
import Logging
package actor UnixSocketRelayManager {
private let vm: any VirtualMachineInstance
private var relays: [String: UnixSocketRelay]
private let log: Logger?
init(vm: any VirtualMachineInstance, log: Logger? = nil) {
self.vm = vm
self.relays = [:]
self.log = log
}
}
extension UnixSocketRelayManager {
func start(port: UInt32, socket: UnixSocketConfiguration) async throws {
guard relays[socket.id] == nil else {
throw ContainerizationError(
.invalidState,
message: "socket relay \(socket.id) already started"
)
}
let relay = try UnixSocketRelay(
port: port,
socket: socket,
vm: vm,
log: log
)
do {
relays[socket.id] = relay
try await relay.start()
} catch {
relays.removeValue(forKey: socket.id)
throw error
}
}
func stop(socket: UnixSocketConfiguration) async throws {
guard let storedRelay = relays.removeValue(forKey: socket.id) else {
throw ContainerizationError(
.notFound,
message: "failed to stop socket relay"
)
}
try storedRelay.stop()
}
func stopAll() async throws {
for (_, relay) in relays {
try relay.stop()
}
}
}
@@ -0,0 +1,103 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationOCI
import Foundation
/// Destination for boot log (serial console) output.
public struct BootLog: Sendable {
/// The underlying representation of the boot log destination.
internal enum Representation: Sendable {
case file(path: URL, append: Bool)
case fileHandle(FileHandle)
}
internal var base: Representation
/// Write boot logs to a file at the specified path.
///
/// - Parameters:
/// - path: The URL of the file to write boot logs to.
/// - append: Whether to append to an existing file or overwrite it. Defaults to true.
///
/// - Returns: A boot log destination that writes to a file.
public static func file(path: URL, append: Bool = true) -> BootLog {
self.init(base: .file(path: path, append: append))
}
/// Write boot logs to a file handle.
///
/// - Parameter fileHandle: The file handle to write boot logs to.
///
/// - Returns: A boot log destination that writes to a file handle.
public static func fileHandle(_ fileHandle: FileHandle) -> BootLog {
self.init(base: .fileHandle(fileHandle))
}
}
/// Protocol for VM creation configuration. Allows VMMs to extend with specific settings
/// while maintaining a common core configuration.
public protocol VMCreationConfig: Sendable {
/// The common VM configuration that all VMMs must support.
var configuration: VMConfiguration { get }
}
/// Standard VM creation configuration with only common settings.
public struct StandardVMConfig: VMCreationConfig {
public var configuration: VMConfiguration
public init(configuration: VMConfiguration) {
self.configuration = configuration
}
}
/// Configuration for creating a virtual machine instance.
public struct VMConfiguration: Sendable {
/// The amount of CPUs to allocate.
public var cpus: Int
/// The memory in bytes to allocate.
public var memoryInBytes: UInt64
/// The network interfaces to attach.
public var interfaces: [any Interface]
/// Mounts organized by metadata ID (e.g. container ID).
/// Each ID maps to an array of mounts for that workload.
public var mountsByID: [String: [Mount]]
/// Optional destination for serial boot logs.
public var bootLog: BootLog?
/// Enable nested virtualization support. If the VirtualMachineManager
/// does not support this feature, it MUST return an .unsupported ContainerizationError.
public var nestedVirtualization: Bool
/// Extension objects that participate in the VM instance lifecycle.
/// Extension packages append their types here; VZ-aware extensions
/// should conform to ``VZInstanceExtension``.
public var extensions: [any Sendable] = []
public init(
cpus: Int = 4,
memoryInBytes: UInt64 = 1024 * 1024 * 1024,
interfaces: [any Interface] = [],
mountsByID: [String: [Mount]] = [:],
bootLog: BootLog? = nil,
nestedVirtualization: Bool = false
) {
self.cpus = cpus
self.memoryInBytes = memoryInBytes
self.interfaces = interfaces
self.mountsByID = mountsByID
self.bootLog = bootLog
self.nestedVirtualization = nestedVirtualization
}
}
@@ -0,0 +1,152 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
#if os(macOS)
import Foundation
import Logging
import Virtualization
import ContainerizationError
extension VZVirtualMachine {
nonisolated func connect(queue: DispatchQueue, port: UInt32) async throws -> VZVirtioSocketConnection {
try await withCheckedThrowingContinuation { cont in
queue.sync {
guard let vsock = self.socketDevices[0] as? VZVirtioSocketDevice else {
let error = ContainerizationError(.invalidArgument, message: "no vsock device")
cont.resume(throwing: error)
return
}
vsock.connect(toPort: port) { result in
switch result {
case .success(let conn):
// `conn` isn't used concurrently.
nonisolated(unsafe) let conn = conn
cont.resume(returning: conn)
case .failure(let error):
cont.resume(throwing: error)
}
}
}
}
}
func listen(queue: DispatchQueue, port: UInt32, listener: VZVirtioSocketListener) throws {
try queue.sync {
guard let vsock = self.socketDevices[0] as? VZVirtioSocketDevice else {
throw ContainerizationError(.invalidArgument, message: "no vsock device")
}
vsock.setSocketListener(listener, forPort: port)
}
}
func removeListener(queue: DispatchQueue, port: UInt32) throws {
try queue.sync {
guard let vsock = self.socketDevices[0] as? VZVirtioSocketDevice else {
throw ContainerizationError(
.invalidArgument,
message: "no vsock device to remove"
)
}
vsock.removeSocketListener(forPort: port)
}
}
func start(queue: DispatchQueue) async throws {
try await withCheckedThrowingContinuation { (cont: CheckedContinuation<Void, Error>) in
queue.sync {
self.start { result in
if case .failure(let error) = result {
cont.resume(throwing: error)
return
}
cont.resume()
}
}
}
}
func stop(queue: DispatchQueue) async throws {
try await withCheckedThrowingContinuation { (cont: CheckedContinuation<Void, Error>) in
queue.sync {
self.stop { error in
if let error {
cont.resume(throwing: error)
return
}
cont.resume()
}
}
}
}
func pause(queue: DispatchQueue) async throws {
try await withCheckedThrowingContinuation { (cont: CheckedContinuation<Void, Error>) in
queue.sync {
self.pause { result in
if case .failure(let error) = result {
cont.resume(throwing: error)
return
}
cont.resume()
}
}
}
}
func resume(queue: DispatchQueue) async throws {
try await withCheckedThrowingContinuation { (cont: CheckedContinuation<Void, Error>) in
queue.sync {
self.resume { result in
if case .failure(let error) = result {
cont.resume(throwing: error)
return
}
cont.resume()
}
}
}
}
}
extension VZVirtualMachine {
func waitForAgent(queue: DispatchQueue) async throws -> FileHandle {
let agentConnectionRetryCount: Int = 200
let agentConnectionSleepDuration: Duration = .milliseconds(20)
for _ in 0...agentConnectionRetryCount {
do {
return try await self.connect(queue: queue, port: Vminitd.port).dupHandle()
} catch {
try await Task.sleep(for: agentConnectionSleepDuration)
continue
}
}
throw ContainerizationError(.timeout, message: "failed to get a connection to agent socket")
}
}
extension VZVirtioSocketConnection {
func dupHandle() throws -> FileHandle {
let fd = dup(self.fileDescriptor)
if fd == -1 {
throw POSIXError.fromErrno()
}
self.close()
return FileHandle(fileDescriptor: fd, closeOnDealloc: false)
}
}
#endif
@@ -0,0 +1,619 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
#if os(macOS)
import Foundation
import ContainerizationError
import ContainerizationExtras
import ContainerizationOCI
import Logging
import NIOCore
import NIOPosix
import Synchronization
@preconcurrency import Virtualization
public final class VZVirtualMachineInstance: Sendable {
public typealias Agent = Vminitd
/// Attached mounts on the virtual machine, organized by metadata ID.
private let _mounts: Mutex<[String: [AttachedFilesystem]]>
public var mounts: [String: [AttachedFilesystem]] {
_mounts.withLock { $0 }
}
/// The underlying Virtualization framework virtual machine.
public var vzVirtualMachine: VZVirtualMachine { vm }
/// The dispatch queue used for VZ operations.
public var vmQueue: DispatchQueue { queue }
/// Mutate the mount registry.
public func withMountRegistry<T: Sendable>(_ body: (inout sending [String: [AttachedFilesystem]]) throws -> sending T) rethrows -> T {
try _mounts.withLock(body)
}
/// Serialize VM operations with the instance lock.
public func withInstanceLock<T: Sendable>(_ body: @Sendable @escaping () async throws -> T) async throws -> T {
try await lock.withLock { _ in try await body() }
}
/// The hotplug provider, if hotplug is enabled for this instance.
public var hotplugProvider: (any HotplugProvider)? {
get { _hotplugProvider.withLock { $0 } }
set { _hotplugProvider.withLock { $0 = newValue } }
}
private let _hotplugProvider = Mutex<(any HotplugProvider)?>(nil)
/// Returns the runtime state of the vm.
public var state: VirtualMachineInstanceState {
vzStateToInstanceState()
}
/// The virtual machine instance configuration.
private let config: Configuration
public struct Configuration: Sendable {
/// Amount of cpus to allocated.
public var cpus: Int
/// Amount of memory in bytes allocated.
public var memoryInBytes: UInt64
/// Toggle rosetta's x86_64 emulation support.
public var rosetta: Bool
/// Toggle nested virtualization support.
public var nestedVirtualization: Bool
/// Mount attachments organized by metadata ID.
public var mountsByID: [String: [Mount]]
/// Network interface attachments.
public var interfaces: [any Interface]
/// Kernel image.
public var kernel: Kernel?
/// The root filesystem.
public var initialFilesystem: Mount?
/// Destination for the virtual machine's boot logs.
public var bootLog: BootLog?
/// Extension objects that participate in the VM instance lifecycle.
public var extensions: [any Sendable] = []
public init() {
self.cpus = 4
self.memoryInBytes = 1024.mib()
self.rosetta = false
self.nestedVirtualization = false
self.mountsByID = [:]
self.interfaces = []
}
}
// `vm` isn't used concurrently.
private nonisolated(unsafe) let vm: VZVirtualMachine
private let queue: DispatchQueue
private let lock: AsyncLock
private let group: EventLoopGroup
private let ownsGroup: Bool
private let timeSyncer: TimeSyncer
private let logger: Logger?
public convenience init(
group: EventLoopGroup? = nil,
logger: Logger? = nil,
with: (inout Configuration) throws -> Void
) throws {
var config = Configuration()
try with(&config)
try self.init(group: group, config: config, logger: logger)
}
init(group: EventLoopGroup?, config: Configuration, logger: Logger?) throws {
if let group {
self.ownsGroup = false
self.group = group
} else {
self.ownsGroup = true
self.group = MultiThreadedEventLoopGroup(numberOfThreads: System.coreCount)
}
self.config = config
self.lock = .init()
self.queue = DispatchQueue(label: "com.apple.containerization.vzvm.\(UUID().uuidString)")
self.logger = logger
self.timeSyncer = .init(logger: logger)
let allocator = Character.blockDeviceTagAllocator()
let (mountAttachments, _) = try config.mountAttachments(allocator: allocator)
self._mounts = Mutex(mountAttachments)
self.vm = VZVirtualMachine(
configuration: try config.toVZ(allocator: allocator),
queue: self.queue
)
for ext in config.extensions.compactMap({ $0 as? any VZInstanceExtension }) {
try ext.didCreate(self)
}
}
}
/// Protocol for extensions that participate in VZVirtualMachineInstance lifecycle.
/// Append conforming types to `Configuration.extensions` to hook into VM setup and teardown.
public protocol VZInstanceExtension: Sendable {
/// Modify the VZ configuration before the VM is created.
func configureVZ(
_ config: inout VZVirtualMachineConfiguration,
allocator: any AddressAllocator<Character>,
storageDeviceCount: Int,
mountsByID: [String: [Mount]]
) throws
/// Called after the VZVirtualMachine is created but before start.
func didCreate(_ instance: VZVirtualMachineInstance) throws
/// Called during stop before the VM is shut down.
func willStop(_ instance: VZVirtualMachineInstance) async throws
}
extension VZInstanceExtension {
public func configureVZ(
_ config: inout VZVirtualMachineConfiguration,
allocator: any AddressAllocator<Character>,
storageDeviceCount: Int,
mountsByID: [String: [Mount]]
) throws {}
public func didCreate(_ instance: VZVirtualMachineInstance) throws {}
public func willStop(_ instance: VZVirtualMachineInstance) async throws {}
}
extension VZVirtualMachineInstance: VirtualMachineInstance {
public func start() async throws {
try await lock.withLock { _ in
guard self.state == .stopped else {
throw ContainerizationError(
.invalidState,
message: "virtual machine is not stopped \(self.state)"
)
}
// Do any necessary setup needed prior to starting the guest.
try await self.prestart()
try await self.vm.start(queue: self.queue)
let agent = try Vminitd(
connection: try await self.vm.waitForAgent(queue: self.queue),
group: self.group
)
do {
if self.config.rosetta {
try await agent.enableRosetta()
}
} catch {
try await agent.close()
throw error
}
// Don't close our remote context as we are providing
// it to our time sync routine.
await self.timeSyncer.start(context: agent)
}
}
public func stop() async throws {
try await lock.withLock { connections in
// NOTE: We should record HOW the vm stopped eventually. If the vm exited
// unexpectedly virtualization framework offers you a way to store
// an error on how it exited. We should report that here instead of the
// generic vm is not running.
guard self.state == .running else {
throw ContainerizationError(.invalidState, message: "vm is not running")
}
try await self.timeSyncer.close()
if self.ownsGroup {
try await self.group.shutdownGracefully()
}
for ext in self.config.extensions.compactMap({ $0 as? any VZInstanceExtension }) {
try? await ext.willStop(self)
}
try await self.vm.stop(queue: self.queue)
}
}
// NOTE: Investigate what is the "right" way to handle already vended vsock
// connections for pause and resume.
public func pause() async throws {
try await lock.withLock { _ in
await self.timeSyncer.pause()
try await self.vm.pause(queue: self.queue)
}
}
public func resume() async throws {
try await lock.withLock { _ in
try await self.vm.resume(queue: self.queue)
await self.timeSyncer.resume()
}
}
public func dialAgent() async throws -> Vminitd {
try await lock.withLock { _ in
do {
let conn = try await self.vm.connect(
queue: self.queue,
port: Vminitd.port
)
let handle = try conn.dupHandle()
return try Vminitd(connection: handle, group: self.group)
} catch {
if let err = error as? ContainerizationError {
throw err
}
throw ContainerizationError(
.internalError,
message: "failed to dial agent",
cause: error
)
}
}
}
public func dial(_ port: UInt32) async throws -> FileHandle {
try await lock.withLock { _ in
do {
let conn = try await self.vm.connect(
queue: self.queue,
port: port
)
return try conn.dupHandle()
} catch {
if let err = error as? ContainerizationError {
throw err
}
throw ContainerizationError(
.internalError,
message: "failed to dial vsock port",
cause: error
)
}
}
}
public func listen(_ port: UInt32) throws -> VsockListener {
let stream = VsockListener(port: port, stopListen: self.stopListen)
let listener = VZVirtioSocketListener()
listener.delegate = stream
try self.vm.listen(
queue: queue,
port: port,
listener: listener
)
return stream
}
private func stopListen(_ port: UInt32) throws {
try self.vm.removeListener(
queue: queue,
port: port
)
}
// MARK: - Hotplug
public func hotplug(_ block: Mount, id: String) async throws -> AttachedFilesystem {
guard let hotplugProvider else {
throw ContainerizationError(.unsupported, message: "hotplug not supported")
}
return try await hotplugProvider.hotplug(block, id: id)
}
public func registerMounts(id: String, rootfs: AttachedFilesystem, additionalMounts: [Mount]) throws {
guard let hotplugProvider else { return }
try hotplugProvider.registerMounts(id: id, rootfs: rootfs, additionalMounts: additionalMounts)
}
public func releaseHotplug(id: String) async throws {
guard let hotplugProvider else { return }
try await hotplugProvider.releaseHotplug(id: id)
}
public func hotplugVirtioFS(_ mounts: [Mount], id: String) async throws {
guard let hotplugProvider else { return }
try await hotplugProvider.hotplugVirtioFS(mounts, id: id)
}
public func releaseVirtioFS(id: String) async throws {
guard let hotplugProvider else { return }
try await hotplugProvider.releaseVirtioFS(id: id)
}
}
extension VZVirtualMachineInstance {
func vzStateToInstanceState() -> VirtualMachineInstanceState {
self.queue.sync {
let state: VirtualMachineInstanceState
switch self.vm.state {
case .starting:
state = .starting
case .running:
state = .running
case .stopping:
state = .stopping
case .stopped:
state = .stopped
default:
state = .unknown
}
return state
}
}
func prestart() async throws {
if self.config.rosetta {
#if arch(arm64)
if VZLinuxRosettaDirectoryShare.availability == .notInstalled {
self.logger?.info("installing rosetta")
try await VZVirtualMachineInstance.Configuration.installRosetta()
}
#else
fatalError("rosetta is only supported on arm64")
#endif
}
}
}
extension VZVirtualMachineInstance.Configuration {
public static func installRosetta() async throws {
do {
#if arch(arm64)
try await VZLinuxRosettaDirectoryShare.installRosetta()
#else
fatalError("rosetta is only supported on arm64")
#endif
} catch {
throw ContainerizationError(
.internalError,
message: "failed to install rosetta",
cause: error
)
}
}
private func serialPort(destination: BootLog) throws -> [VZVirtioConsoleDeviceSerialPortConfiguration] {
let c = VZVirtioConsoleDeviceSerialPortConfiguration()
switch destination.base {
case .file(let path, let append):
c.attachment = try VZFileSerialPortAttachment(url: path, append: append)
case .fileHandle(let fileHandle):
c.attachment = VZFileHandleSerialPortAttachment(
fileHandleForReading: nil,
fileHandleForWriting: fileHandle
)
}
return [c]
}
func toVZ(allocator: any AddressAllocator<Character>) throws -> VZVirtualMachineConfiguration {
var config = VZVirtualMachineConfiguration()
config.cpuCount = self.cpus
let mib: UInt64 = 1 << 20
config.memorySize = (self.memoryInBytes + mib - 1) & ~(mib - 1)
config.entropyDevices = [VZVirtioEntropyDeviceConfiguration()]
config.socketDevices = [VZVirtioSocketDeviceConfiguration()]
if let bootLog = self.bootLog {
config.serialPorts = try serialPort(destination: bootLog)
} else {
// We always supply a serial console. If no explicit path was provided just send em to the void.
config.serialPorts = try serialPort(destination: .file(path: URL(filePath: "/dev/null")))
}
config.networkDevices = try self.interfaces.map {
guard let vzi = $0 as? VZInterface else {
throw ContainerizationError(.invalidArgument, message: "interface type not supported by VZ")
}
return try vzi.device()
}
if self.rosetta {
#if arch(arm64)
switch VZLinuxRosettaDirectoryShare.availability {
case .notSupported:
throw ContainerizationError(
.invalidArgument,
message: "rosetta was requested but is not supported on this machine"
)
case .notInstalled:
// NOTE: If rosetta isn't installed, we'll error with a nice error message
// during .start() of the virtual machine instance.
fallthrough
case .installed:
let share = try VZLinuxRosettaDirectoryShare()
let device = VZVirtioFileSystemDeviceConfiguration(tag: "rosetta")
device.share = share
config.directorySharingDevices.append(device)
@unknown default:
throw ContainerizationError(
.invalidArgument,
message: "unknown rosetta availability encountered: \(VZLinuxRosettaDirectoryShare.availability)"
)
}
#else
fatalError("rosetta is only supported on arm64")
#endif
}
guard let kernel = self.kernel else {
throw ContainerizationError(.invalidArgument, message: "kernel cannot be nil")
}
guard let initialFilesystem = self.initialFilesystem else {
throw ContainerizationError(.invalidArgument, message: "rootfs cannot be nil")
}
let loader = VZLinuxBootLoader(kernelURL: kernel.path)
loader.commandLine = kernel.linuxCommandline(initialFilesystem: initialFilesystem)
config.bootLoader = loader
try initialFilesystem.configure(config: &config)
// Track used virtiofs tags to avoid creating duplicate VZ devices.
// The same source directory mounted to multiple destinations shares one device.
var usedVirtioFSTags: Set<String> = []
for (_, mounts) in self.mountsByID {
for mount in mounts {
if case .virtiofs = mount.runtimeOptions {
let tag = try hashFilePath(path: mount.source)
if usedVirtioFSTags.contains(tag) {
continue
}
usedVirtioFSTags.insert(tag)
}
try mount.configure(config: &config)
}
}
// Create the unified virtiofs device with VZMultipleDirectoryShare
// This device hosts all virtiofs shares and supports runtime updates
var directories: [String: VZSharedDirectory] = [:]
for (_, mounts) in self.mountsByID {
for mount in mounts {
guard case .virtiofs(_) = mount.runtimeOptions else { continue }
guard FileManager.default.fileExists(atPath: mount.source) else {
throw ContainerizationError(.notFound, message: "directory \(mount.source) does not exist")
}
let name = try hashFilePath(path: mount.source)
directories[name] = VZSharedDirectory(
url: URL(fileURLWithPath: mount.source),
readOnly: mount.options.contains("ro")
)
}
}
let multiShare = VZMultipleDirectoryShare(directories: directories)
let virtiofsDevice = VZVirtioFileSystemDeviceConfiguration(tag: "virtiofs")
virtiofsDevice.share = multiShare
config.directorySharingDevices.append(virtiofsDevice)
let storageDeviceCount = config.storageDevices.count
let platform = VZGenericPlatformConfiguration()
// We shouldn't silently succeed if the user asked for virt and their hardware does
// not support it.
if !VZGenericPlatformConfiguration.isNestedVirtualizationSupported && self.nestedVirtualization {
throw ContainerizationError(
.unsupported,
message: "nested virtualization is not supported on the platform"
)
}
platform.isNestedVirtualizationEnabled = self.nestedVirtualization
config.platform = platform
for ext in self.extensions.compactMap({ $0 as? any VZInstanceExtension }) {
try ext.configureVZ(&config, allocator: allocator, storageDeviceCount: storageDeviceCount, mountsByID: self.mountsByID)
}
try config.validate()
return config
}
func mountAttachments(allocator: any AddressAllocator<Character>) throws -> (
attachments: [String: [AttachedFilesystem]], storageDeviceCount: Int
) {
var storageDeviceCount = 0
if let initialFilesystem {
// When the initial filesystem is a blk, allocate the first letter "vd(a)"
// as that is what this blk will be attached under.
if initialFilesystem.isBlock {
_ = try allocator.allocate()
storageDeviceCount += 1
}
}
var attachmentsByID: [String: [AttachedFilesystem]] = [:]
for (id, mounts) in self.mountsByID {
var attachments: [AttachedFilesystem] = []
for mount in mounts {
let attached = try AttachedFilesystem(mount: mount, allocator: allocator)
attachments.append(attached)
if mount.isBlock {
storageDeviceCount += 1
}
}
attachmentsByID[id] = attachments
}
return (attachmentsByID, storageDeviceCount)
}
}
extension Kernel {
func linuxCommandline(initialFilesystem: Mount) -> String {
var args = self.commandLine.kernelArgs
args.append("init=/sbin/vminitd")
// rootfs is always set as ro.
args.append("ro")
switch initialFilesystem.type {
case "virtiofs":
args.append(contentsOf: [
"rootfstype=virtiofs",
"root=rootfs",
])
case "ext4":
args.append(contentsOf: [
"rootfstype=ext4",
"root=/dev/vda",
])
default:
fatalError("unsupported initfs filesystem \(initialFilesystem.type)")
}
if self.commandLine.initArgs.count > 0 {
args.append("--")
args.append(contentsOf: self.commandLine.initArgs)
}
return args.joined(separator: " ")
}
}
public protocol VZInterface {
func device() throws -> VZVirtioNetworkDeviceConfiguration
}
extension NATInterface: VZInterface {
public func device() throws -> VZVirtioNetworkDeviceConfiguration {
let config = VZVirtioNetworkDeviceConfiguration()
if let macAddress = self.macAddress {
guard let mac = VZMACAddress(string: macAddress.description) else {
throw ContainerizationError(.invalidArgument, message: "invalid mac address \(macAddress)")
}
config.macAddress = mac
}
config.attachment = VZNATNetworkDeviceAttachment()
return config
}
}
#endif
@@ -0,0 +1,84 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
#if os(macOS)
import ContainerizationError
import ContainerizationOCI
import Foundation
import Logging
import NIOCore
/// A virtualization.framework backed `VirtualMachineManager` implementation.
public struct VZVirtualMachineManager: VirtualMachineManager {
private let kernel: Kernel
private let initialFilesystem: Mount
private let rosetta: Bool
private let nestedVirtualization: Bool
private let group: EventLoopGroup?
private let logger: Logger?
public init(
kernel: Kernel,
initialFilesystem: Mount,
rosetta: Bool = false,
nestedVirtualization: Bool = false,
group: EventLoopGroup? = nil,
logger: Logger? = nil
) {
self.kernel = kernel
self.initialFilesystem = initialFilesystem
self.rosetta = rosetta
self.nestedVirtualization = nestedVirtualization
self.group = group
self.logger = logger
}
public func create(config: some VMCreationConfig) throws -> any VirtualMachineInstance {
let vmConfig = config.configuration
// Use nested virtualization if requested in config or set as default in manager
let useNestedVirtualization = vmConfig.nestedVirtualization || self.nestedVirtualization
// Clamp to system RAM as Virtualization.framework bounds us to this.
let memoryInBytes = min(vmConfig.memoryInBytes, ProcessInfo.processInfo.physicalMemory)
// Clamp to system CPU count as Virtualization.framework bounds us to this.
let cpus = min(vmConfig.cpus, ProcessInfo.processInfo.activeProcessorCount)
return try VZVirtualMachineInstance(
group: self.group,
logger: self.logger,
with: { instanceConfig in
instanceConfig.cpus = cpus
instanceConfig.memoryInBytes = memoryInBytes
instanceConfig.kernel = self.kernel
instanceConfig.initialFilesystem = self.initialFilesystem
if let bootLog = vmConfig.bootLog {
instanceConfig.bootLog = bootLog
}
instanceConfig.interfaces = vmConfig.interfaces
instanceConfig.rosetta = self.rosetta
instanceConfig.nestedVirtualization = useNestedVirtualization
instanceConfig.mountsByID = vmConfig.mountsByID
instanceConfig.extensions = vmConfig.extensions
})
}
}
#endif
@@ -0,0 +1,22 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
/// Protocol to conform to if your agent is capable of relaying unix domain socket
/// connections.
public protocol SocketRelayAgent {
func relaySocket(port: UInt32, configuration: UnixSocketConfiguration) async throws
func stopSocketRelay(configuration: UnixSocketConfiguration) async throws
}
@@ -0,0 +1,84 @@
//===----------------------------------------------------------------------===//
// Copyright © 2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationExtras
import Logging
extension VirtualMachineAgent {
/// Configure a single network interface inside the sandbox: assign addresses,
/// bring the link up, and (when requested) install the link/default routes.
func setupInterface(
_ interface: any Interface,
name: String,
setDefaultRoute: Bool,
logger: Logger?
) async throws {
logger?.debug("setting up interface \(name) with v4 \(interface.ipv4Address) v6 \(interface.ipv6Address?.description ?? "<none>")")
try await addressAdd(
name: name,
address: .init(ipv4Address: interface.ipv4Address, ipv6Address: interface.ipv6Address)
)
try await up(name: name, mtu: interface.mtu)
guard setDefaultRoute else { return }
let ipv4Address = interface.ipv4Address
let ipv4Gateway = interface.ipv4Gateway
let ipv6Gateway = interface.ipv6Gateway
let ipv6Address = interface.ipv6Address
let needsIPv4LinkRoute: Bool
if let ipv4Gateway {
needsIPv4LinkRoute = !ipv4Address.contains(ipv4Gateway)
} else {
needsIPv4LinkRoute = false
}
let needsIPv6LinkRoute: Bool
if let ipv6Gateway, let ipv6Address {
needsIPv6LinkRoute = !ipv6Address.contains(ipv6Gateway)
} else {
needsIPv6LinkRoute = false
}
if needsIPv4LinkRoute, let ipv4Gateway {
logger?.debug("v4 gateway \(ipv4Gateway) is outside subnet \(ipv4Address), adding a route first")
}
if needsIPv6LinkRoute, let ipv6Gateway, let ipv6Address {
logger?.debug("v6 gateway \(ipv6Gateway) is outside subnet \(ipv6Address), adding a route first")
}
if needsIPv4LinkRoute || needsIPv6LinkRoute {
try await routeAddLink(
name: name,
route: .init(
ipv4Destination: needsIPv4LinkRoute ? ipv4Gateway : nil,
ipv4Source: needsIPv4LinkRoute ? ipv4Address.address : nil,
ipv6Destination: needsIPv6LinkRoute ? ipv6Gateway : nil,
ipv6Source: needsIPv6LinkRoute ? ipv6Address?.address : nil
)
)
}
if ipv4Gateway == nil && ipv6Gateway == nil {
logger?.debug("no gateway for \(name)")
}
try await routeAddDefault(
name: name,
route: .init(ipv4Gateway: ipv4Gateway, ipv6Gateway: ipv6Gateway)
)
}
}
@@ -0,0 +1,110 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationError
import ContainerizationExtras
import ContainerizationOCI
import Foundation
public struct WriteFileFlags {
public var createParentDirectories = false
public var append = false
public var create = false
}
public enum FilesystemOperation: Sendable {
case freeze
case thaw
case trim
}
/// A protocol for the agent running inside a virtual machine. If an operation isn't
/// supported the implementation MUST return a ContainerizationError with a code of
/// `.unsupported`.
public protocol VirtualMachineAgent: Sendable {
/// Perform a platform specific standard setup
/// of the runtime environment.
func standardSetup() async throws
/// Close any resources held by the agent.
func close() async throws
// Perform a filesystem operation on the given path.
func filesystemOperation(operation: FilesystemOperation, path: String) async throws
// POSIX-y
func getenv(key: String) async throws -> String
func setenv(key: String, value: String) async throws
func mount(_ mount: ContainerizationOCI.Mount) async throws
func umount(path: String, flags: Int32) async throws
func mkdir(path: String, all: Bool, perms: UInt32) async throws
@discardableResult
func kill(pid: Int32, signal: Int32) async throws -> Int32
func sync() async throws
func writeFile(path: String, data: Data, flags: WriteFileFlags, mode: UInt32) async throws
// Process lifecycle
func createProcess(
id: String,
containerID: String?,
stdinPort: UInt32?,
stdoutPort: UInt32?,
stderrPort: UInt32?,
ociRuntimePath: String?,
configuration: ContainerizationOCI.Spec,
options: Data?
) async throws
func startProcess(id: String, containerID: String?) async throws -> Int32
func signalProcess(id: String, containerID: String?, signal: Int32) async throws
func resizeProcess(id: String, containerID: String?, columns: UInt32, rows: UInt32) async throws
func waitProcess(id: String, containerID: String?, timeoutInSeconds: Int64?) async throws -> ExitStatus
func deleteProcess(id: String, containerID: String?) async throws
func closeProcessStdin(id: String, containerID: String?) async throws
// Networking
func up(name: String, mtu: UInt32?) async throws
func down(name: String) async throws
func addressAdd(name: String, address: InterfaceAddress) async throws
func routeAddLink(name: String, route: LinkRoute) async throws
func routeAddDefault(name: String, route: DefaultRoute) async throws
func configureDNS(config: DNS, location: String) async throws
func configureHosts(config: Hosts, location: String) async throws
// Container statistics
func containerStatistics(containerIDs: [String], categories: StatCategory) async throws -> [ContainerStatistics]
}
extension VirtualMachineAgent {
public func closeProcessStdin(id: String, containerID: String?) async throws {
throw ContainerizationError(.unsupported, message: "closeProcessStdin")
}
public func configureHosts(config: Hosts, location: String) async throws {
throw ContainerizationError(.unsupported, message: "configureHosts")
}
public func writeFile(path: String, data: Data, flags: WriteFileFlags, mode: UInt32) async throws {
throw ContainerizationError(.unsupported, message: "writeFile")
}
public func containerStatistics(containerIDs: [String], categories: StatCategory) async throws -> [ContainerStatistics] {
throw ContainerizationError(.unsupported, message: "containerStatistics")
}
public func sync() async throws {
throw ContainerizationError(.unsupported, message: "sync")
}
}
@@ -0,0 +1,105 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationError
import Foundation
/// The runtime state of the virtual machine instance.
public enum VirtualMachineInstanceState: Sendable {
case starting
case running
case stopped
case stopping
case unknown
}
/// A live instance of a virtual machine.
public protocol VirtualMachineInstance: Sendable {
associatedtype Agent: VirtualMachineAgent
// The state of the virtual machine.
var state: VirtualMachineInstanceState { get }
var mounts: [String: [AttachedFilesystem]] { get }
/// Dial the Agent. It's up the VirtualMachineInstance to determine
/// what port the agent is listening on.
func dialAgent() async throws -> Agent
/// Dial a vsock port in the guest.
func dial(_ port: UInt32) async throws -> FileHandle
/// Listen on a host vsock port.
func listen(_ port: UInt32) throws -> VsockListener
/// Start the virtual machine.
func start() async throws
/// Stop the virtual machine.
func stop() async throws
/// Pause the virtual machine.
func pause() async throws
/// Resume the virtual machine.
func resume() async throws
/// Hotplug a block device, returning the attached filesystem info.
/// Throws if the VMM does not support hotplug or not available
/// - Parameter block: The mount configuration for the block device to hotplug
/// - Parameter id: The metadata ID to associate with this mount (e.g. container ID)
/// - Returns: AttachedFilesystem with the device path in the guest
func hotplug(_ block: Mount, id: String) async throws -> AttachedFilesystem
/// Register mounts for a container after hotplug.
/// This is used to add the rootfs and additional mounts to the VM's mount registry
/// so they can be found when building the container's OCI spec.
/// - Parameter id: The container ID
/// - Parameter rootfs: The rootfs attachment from hotplug
/// - Parameter additionalMounts: Additional mounts (like /proc, /sys) to register
func registerMounts(id: String, rootfs: AttachedFilesystem, additionalMounts: [Mount]) throws
/// Release a hotplug device.
/// This should be called when a hotplugged container is stopped or fails to start.
/// - Parameter id: The container ID whose hotplug should be released
func releaseHotplug(id: String) async throws
/// Hotplug virtiofs directories into the running VM.
/// - Parameter mounts: The virtiofs mounts to add
/// - Parameter id: The container ID that owns these mounts
func hotplugVirtioFS(_ mounts: [Mount], id: String) async throws
/// Release virtiofs shares for a container.
/// - Parameter id: The container ID whose virtiofs shares should be released
func releaseVirtioFS(id: String) async throws
}
extension VirtualMachineInstance {
public func pause() async throws {
throw ContainerizationError(.unsupported, message: "pause")
}
public func resume() async throws {
throw ContainerizationError(.unsupported, message: "resume")
}
public func hotplug(_ block: Mount, id: String) async throws -> AttachedFilesystem {
throw ContainerizationError(.unsupported, message: "hotplug not supported")
}
public func registerMounts(id: String, rootfs: AttachedFilesystem, additionalMounts: [Mount]) throws {
// no-op default
}
public func releaseHotplug(id: String) async throws {
// no-op default
}
public func hotplugVirtioFS(_ mounts: [Mount], id: String) async throws {
// no-op default
}
public func releaseVirtioFS(id: String) async throws {
// no-op default
}
}
@@ -0,0 +1,20 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
/// A protocol to implement for virtual machine isolated containers.
public protocol VirtualMachineManager: Sendable {
func create(config: some VMCreationConfig) async throws -> any VirtualMachineInstance
}
@@ -0,0 +1,35 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationOS
extension Vminitd {
/// Enable Rosetta's x86_64 emulation.
public func enableRosetta() async throws {
let path = "/run/rosetta"
try await self.mount(
.init(
type: "virtiofs",
source: "rosetta",
destination: path
)
)
try await self.setupEmulator(
binaryPath: "\(path)/rosetta",
configuration: Binfmt.Entry.amd64()
)
}
}
@@ -0,0 +1,47 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
extension Vminitd: SocketRelayAgent {
/// Sets up a relay between a host socket to a newly created guest socket, or vice versa.
public func relaySocket(port: UInt32, configuration: UnixSocketConfiguration) async throws {
let request = Com_Apple_Containerization_Sandbox_V3_ProxyVsockRequest.with {
$0.id = configuration.id
$0.vsockPort = port
if let perms = configuration.permissions {
$0.guestSocketPermissions = UInt32(perms.rawValue)
}
switch configuration.direction {
case .into:
$0.guestPath = configuration.destination.path
$0.action = .into
case .outOf:
$0.guestPath = configuration.source.path
$0.action = .outOf
}
}
_ = try await client.proxyVsock(request)
}
/// Stops the specified socket relay.
public func stopSocketRelay(configuration: UnixSocketConfiguration) async throws {
let request = Com_Apple_Containerization_Sandbox_V3_StopVsockProxyRequest.with {
$0.id = configuration.id
}
_ = try await client.stopVsockProxy(request)
}
}
+641
View File
@@ -0,0 +1,641 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationError
import ContainerizationExtras
import ContainerizationOCI
import ContainerizationOS
import Foundation
import GRPCCore
import GRPCNIOTransportCore
import NIOCore
import NIOPosix
/// A remote connection into the vminitd Linux guest agent via a port (vsock).
/// Used to modify the runtime environment of the Linux sandbox.
public struct Vminitd: Sendable {
// Default vsock port that the agent and client use.
public static let port: UInt32 = 1024
let client: Com_Apple_Containerization_Sandbox_V3_SandboxContext.Client<HTTP2ClientTransport.WrappedChannel>
public let grpcClient: GRPCClient<HTTP2ClientTransport.WrappedChannel>
private let connectionTask: Task<Void, Error>
public init(connection: FileHandle, group: any EventLoopGroup) throws {
let channel = try ClientBootstrap(group: group)
.channelInitializer { channel in
channel.eventLoop.makeCompletedFuture(withResultOf: {
try channel.pipeline.syncOperations.addHandler(HTTP2ConnectBufferingHandler())
})
}
.withConnectedSocket(connection.fileDescriptor).wait()
let transport = HTTP2ClientTransport.WrappedChannel.wrapping(
channel: channel,
config: .defaults { $0.connection.maxIdleTime = nil }
)
let grpcClient = GRPCClient(transport: transport)
self.grpcClient = grpcClient
self.client = Com_Apple_Containerization_Sandbox_V3_SandboxContext.Client(wrapping: self.grpcClient)
// Not very structured concurrency friendly, but we'd need to expose a way on the protocol to "run" the
// agent otherwise, which some agents might not even need.
self.connectionTask = Task {
try await grpcClient.runConnections()
}
}
/// Close the connection to the guest agent.
public func close() async throws {
self.grpcClient.beginGracefulShutdown()
try await self.connectionTask.value
}
}
extension Vminitd: VirtualMachineAgent {
/// Perform the standard guest setup necessary for vminitd to be able to
/// run containers.
public func standardSetup() async throws {
try await up(name: "lo")
try await setenv(key: "PATH", value: LinuxProcessConfiguration.defaultPath)
// Vminitd mounts /proc, /sys, /sys/fs/cgroup and /run automatically.
let mounts: [ContainerizationOCI.Mount] = [
.init(type: "tmpfs", source: "tmpfs", destination: "/tmp"),
.init(type: "devpts", source: "devpts", destination: "/dev/pts", options: ["gid=5", "mode=620", "ptmxmode=666"]),
]
for mount in mounts {
try await self.mount(mount)
}
}
public func writeFile(path: String, data: Data, flags: WriteFileFlags, mode: UInt32) async throws {
_ = try await client.writeFile(
.with {
$0.path = path
$0.mode = mode
$0.data = data
$0.flags = .with {
$0.append = flags.append
$0.createIfMissing = flags.create
$0.createParentDirs = flags.createParentDirectories
}
})
}
/// Get statistics for containers. If `containerIDs` is empty returns stats for all containers
/// in the guest. If `categories` is empty, all categories are returned.
public func containerStatistics(containerIDs: [String], categories: StatCategory) async throws -> [ContainerStatistics] {
let response = try await client.containerStatistics(
.with {
$0.containerIds = containerIDs
$0.categories = categories.toProtoCategories()
})
return response.containers.map { protoStats in
ContainerStatistics(
id: protoStats.containerID,
process: categories.contains(.process) && protoStats.hasProcess
? .init(
current: protoStats.process.current,
limit: protoStats.process.limit
) : nil,
memory: categories.contains(.memory) && protoStats.hasMemory
? .init(
usageBytes: protoStats.memory.usageBytes,
limitBytes: protoStats.memory.limitBytes,
swapUsageBytes: protoStats.memory.swapUsageBytes,
swapLimitBytes: protoStats.memory.swapLimitBytes,
cacheBytes: protoStats.memory.cacheBytes,
kernelStackBytes: protoStats.memory.kernelStackBytes,
slabBytes: protoStats.memory.slabBytes,
pageFaults: protoStats.memory.pageFaults,
majorPageFaults: protoStats.memory.majorPageFaults,
inactiveFile: protoStats.memory.inactiveFile,
anon: protoStats.memory.anon,
workingsetRefaultAnon: protoStats.memory.workingsetRefaultAnon,
workingsetRefaultFile: protoStats.memory.workingsetRefaultFile,
pgstealKswapd: protoStats.memory.pgstealKswapd,
pgstealDirect: protoStats.memory.pgstealDirect,
pgstealKhugepaged: protoStats.memory.pgstealKhugepaged
) : nil,
cpu: categories.contains(.cpu) && protoStats.hasCpu
? .init(
usageUsec: protoStats.cpu.usageUsec,
userUsec: protoStats.cpu.userUsec,
systemUsec: protoStats.cpu.systemUsec,
throttlingPeriods: protoStats.cpu.throttlingPeriods,
throttledPeriods: protoStats.cpu.throttledPeriods,
throttledTimeUsec: protoStats.cpu.throttledTimeUsec
) : nil,
blockIO: categories.contains(.blockIO) && protoStats.hasBlockIo
? .init(
devices: protoStats.blockIo.devices.map { device in
.init(
major: device.major,
minor: device.minor,
readBytes: device.readBytes,
writeBytes: device.writeBytes,
readOperations: device.readOperations,
writeOperations: device.writeOperations
)
}
) : nil,
networks: categories.contains(.network)
? protoStats.networks.map { network in
ContainerStatistics.NetworkStatistics(
interface: network.interface,
receivedPackets: network.receivedPackets,
transmittedPackets: network.transmittedPackets,
receivedBytes: network.receivedBytes,
transmittedBytes: network.transmittedBytes,
receivedErrors: network.receivedErrors,
transmittedErrors: network.transmittedErrors
)
} : nil,
memoryEvents: categories.contains(.memoryEvents) && protoStats.hasMemoryEvents
? .init(
low: protoStats.memoryEvents.low,
high: protoStats.memoryEvents.high,
max: protoStats.memoryEvents.max,
oom: protoStats.memoryEvents.oom,
oomKill: protoStats.memoryEvents.oomKill
) : nil
)
}
}
/// Mount a filesystem in the sandbox's environment.
public func mount(_ mount: ContainerizationOCI.Mount) async throws {
_ = try await client.mount(
.with {
$0.type = mount.type
$0.source = mount.source
$0.destination = mount.destination
$0.options = mount.options
})
}
/// Unmount a filesystem in the sandbox's environment.
public func umount(path: String, flags: Int32) async throws {
_ = try await client.umount(
.with {
$0.path = path
$0.flags = flags
})
}
/// Create a directory inside the sandbox's environment.
public func mkdir(path: String, all: Bool, perms: UInt32) async throws {
_ = try await client.mkdir(
.with {
$0.path = path
$0.all = all
$0.perms = perms
})
}
/// Perform a filesystem operation on a path inside the sandbox's environment.
public func filesystemOperation(operation: FilesystemOperation, path: String) async throws {
_ = try await client.filesystemOperation(
.with {
$0.operation = operation.toProtoOperation()
$0.path = path
})
}
public func createProcess(
id: String,
containerID: String?,
stdinPort: UInt32?,
stdoutPort: UInt32?,
stderrPort: UInt32?,
ociRuntimePath: String?,
configuration: ContainerizationOCI.Spec,
options: Data?
) async throws {
let enc = JSONEncoder()
_ = try await client.createProcess(
.with {
$0.id = id
if let stdinPort {
$0.stdin = stdinPort
}
if let stdoutPort {
$0.stdout = stdoutPort
}
if let stderrPort {
$0.stderr = stderrPort
}
if let containerID {
$0.containerID = containerID
}
if let ociRuntimePath {
$0.ociRuntimePath = ociRuntimePath
}
$0.configuration = try enc.encode(configuration)
})
}
@discardableResult
public func startProcess(id: String, containerID: String?) async throws -> Int32 {
let request = Com_Apple_Containerization_Sandbox_V3_StartProcessRequest.with {
$0.id = id
if let containerID {
$0.containerID = containerID
}
}
let resp = try await client.startProcess(request)
return resp.pid
}
public func signalProcess(id: String, containerID: String?, signal: Int32) async throws {
let request = Com_Apple_Containerization_Sandbox_V3_KillProcessRequest.with {
$0.id = id
$0.signal = signal
if let containerID {
$0.containerID = containerID
}
}
_ = try await client.killProcess(request)
}
public func resizeProcess(id: String, containerID: String?, columns: UInt32, rows: UInt32) async throws {
let request = Com_Apple_Containerization_Sandbox_V3_ResizeProcessRequest.with {
if let containerID {
$0.containerID = containerID
}
$0.id = id
$0.columns = columns
$0.rows = rows
}
_ = try await client.resizeProcess(request)
}
public func waitProcess(
id: String,
containerID: String?,
timeoutInSeconds: Int64? = nil
) async throws -> ExitStatus {
let request = Com_Apple_Containerization_Sandbox_V3_WaitProcessRequest.with {
$0.id = id
if let containerID {
$0.containerID = containerID
}
}
var callOpts = GRPCCore.CallOptions.defaults
if let timeoutInSeconds {
callOpts.timeout = .seconds(timeoutInSeconds)
}
do {
let resp = try await client.waitProcess(request, options: callOpts)
return ExitStatus(exitCode: resp.exitCode, exitedAt: resp.exitedAt.date)
} catch {
if let err = error as? RPCError, err.code == .deadlineExceeded {
throw ContainerizationError(
.timeout,
message: "failed to wait for process exit within timeout of \(timeoutInSeconds!) seconds",
cause: err
)
}
throw error
}
}
public func deleteProcess(id: String, containerID: String?) async throws {
let request = Com_Apple_Containerization_Sandbox_V3_DeleteProcessRequest.with {
$0.id = id
if let containerID {
$0.containerID = containerID
}
}
_ = try await client.deleteProcess(request)
}
public func closeProcessStdin(id: String, containerID: String?) async throws {
let request = Com_Apple_Containerization_Sandbox_V3_CloseProcessStdinRequest.with {
$0.id = id
if let containerID {
$0.containerID = containerID
}
}
_ = try await client.closeProcessStdin(request)
}
public func up(name: String, mtu: UInt32? = nil) async throws {
let request = Com_Apple_Containerization_Sandbox_V3_IpLinkSetRequest.with {
$0.interface = name
$0.up = true
if let mtu { $0.mtu = mtu }
}
_ = try await client.ipLinkSet(request)
}
public func down(name: String) async throws {
let request = Com_Apple_Containerization_Sandbox_V3_IpLinkSetRequest.with {
$0.interface = name
$0.up = false
}
_ = try await client.ipLinkSet(request)
}
/// Get an environment variable from the sandbox's environment.
public func getenv(key: String) async throws -> String {
let response = try await client.getenv(
.with {
$0.key = key
})
return response.value
}
/// Set an environment variable in the sandbox's environment.
public func setenv(key: String, value: String) async throws {
_ = try await client.setenv(
.with {
$0.key = key
$0.value = value
})
}
}
/// Vminitd specific rpcs.
extension Vminitd {
/// Sets up an emulator in the guest.
public func setupEmulator(binaryPath: String, configuration: Binfmt.Entry) async throws {
let request = Com_Apple_Containerization_Sandbox_V3_SetupEmulatorRequest.with {
$0.binaryPath = binaryPath
$0.name = configuration.name
$0.type = configuration.type
$0.offset = configuration.offset
$0.magic = configuration.magic
$0.mask = configuration.mask
$0.flags = configuration.flags
}
_ = try await client.setupEmulator(request)
}
/// Sets the guest time.
public func setTime(sec: Int64, usec: Int32) async throws {
let request = Com_Apple_Containerization_Sandbox_V3_SetTimeRequest.with {
$0.sec = sec
$0.usec = usec
}
_ = try await client.setTime(request)
}
/// Set the provided sysctls inside the Sandbox's environment.
public func sysctl(settings: [String: String]) async throws {
let request = Com_Apple_Containerization_Sandbox_V3_SysctlRequest.with {
$0.settings = settings
}
_ = try await client.sysctl(request)
}
/// Add an IP address to the sandbox's network interfaces.
public func addressAdd(name: String, address: InterfaceAddress) async throws {
_ = try await client.ipAddrAdd(
.with {
$0.interface = name
$0.ipv4Address = address.ipv4Address.description
if let ipv6Address = address.ipv6Address {
$0.ipv6Address = ipv6Address.description
}
})
}
/// Add a link-scoped route in the sandbox's environment, used to install an
/// on-link host route (a /32 for v4, /128 for v6) to a gateway that lives
/// outside the interface's subnet so the kernel will accept the default route.
/// `route.ipv4Destination`/`route.ipv6Destination` carry the
/// gateway address; the wire format is a CIDR string with the per-family host prefix appended.
public func routeAddLink(name: String, route: LinkRoute) async throws {
_ = try await client.ipRouteAddLink(
.with {
$0.interface = name
if let ipv4Destination = route.ipv4Destination {
$0.dstIpv4Addr = "\(ipv4Destination.description)/32"
}
if let ipv4Source = route.ipv4Source {
$0.srcIpv4Addr = ipv4Source.description
}
if let ipv6Destination = route.ipv6Destination {
$0.dstIpv6Addr = "\(ipv6Destination.description)/128"
}
if let ipv6Source = route.ipv6Source {
$0.srcIpv6Addr = ipv6Source.description
}
})
}
/// Set the default route in the sandbox's environment.
public func routeAddDefault(name: String, route: DefaultRoute) async throws {
_ = try await client.ipRouteAddDefault(
.with {
$0.interface = name
$0.ipv4Gateway = route.ipv4Gateway?.description ?? ""
if let ipv6Gateway = route.ipv6Gateway {
$0.ipv6Gateway = ipv6Gateway.description
}
})
}
/// Configure DNS within the sandbox's environment.
public func configureDNS(config: DNS, location: String) async throws {
try config.validate()
_ = try await client.configureDns(
.with {
$0.location = location
$0.nameservers = config.nameservers
if let domain = config.domain {
$0.domain = domain
}
$0.searchDomains = config.searchDomains
$0.options = config.options
})
}
/// Configure /etc/hosts within the sandbox's environment.
public func configureHosts(config: Hosts, location: String) async throws {
_ = try await client.configureHosts(config.toAgentHostsRequest(location: location))
}
/// Perform a sync call.
public func sync() async throws {
_ = try await client.sync(.init())
}
public func kill(pid: Int32, signal: Int32) async throws -> Int32 {
let response = try await client.kill(
.with {
$0.pid = pid
$0.signal = signal
})
return response.result
}
/// Metadata received from the guest during a copy operation.
public struct CopyMetadata: Sendable {
/// Whether the data on the vsock channel is a tar+gzip archive.
public let isArchive: Bool
/// Total size in bytes (0 if unknown, e.g. for archives).
public let totalSize: UInt64
}
/// Stat a path in the guest filesystem and return its metadata.
public func stat(
path: URL
) async throws -> ContainerizationOS.Stat {
let request = Com_Apple_Containerization_Sandbox_V3_StatRequest.with {
$0.path = path.path
}
let response: Com_Apple_Containerization_Sandbox_V3_StatResponse
do {
response = try await client.stat(request)
} catch let error as RPCError where error.code == .notFound {
throw ContainerizationError(.notFound, message: "stat: path not found '\(path.path)'", cause: error)
}
guard response.error.isEmpty else {
throw ContainerizationError(.internalError, message: "stat: \(response.error)")
}
let s = response.stat
return ContainerizationOS.Stat(
dev: s.dev,
ino: s.ino,
mode: s.mode,
nlink: s.nlink,
uid: s.uid,
gid: s.gid,
rdev: s.rdev,
size: s.size,
blksize: s.blksize,
blocks: s.blocks,
atime: TimeSpec(seconds: s.atime.seconds, nanoseconds: s.atime.nanos),
mtime: TimeSpec(seconds: s.mtime.seconds, nanoseconds: s.mtime.nanos),
ctime: TimeSpec(seconds: s.ctime.seconds, nanoseconds: s.ctime.nanos)
)
}
/// Unified copy control plane. Sends a CopyRequest over gRPC and processes
/// the response stream. Data transfer happens over a separate vsock connection
/// managed by the caller.
///
/// For COPY_OUT, the `onMetadata` callback is invoked when the guest sends
/// metadata (is_archive, total_size) before data transfer begins.
/// For COPY_IN, `onMetadata` is not called.
public func copy(
direction: Com_Apple_Containerization_Sandbox_V3_CopyRequest.Direction,
guestPath: URL,
vsockPort: UInt32,
mode: UInt32 = 0,
createParents: Bool = false,
isArchive: Bool = false,
onMetadata: @Sendable @escaping (CopyMetadata) -> Void = { _ in }
) async throws {
let request = Com_Apple_Containerization_Sandbox_V3_CopyRequest.with {
$0.direction = direction
$0.path = guestPath.path
$0.mode = mode
$0.createParents = createParents
$0.vsockPort = vsockPort
$0.isArchive = isArchive
}
try await client.copy(
request,
onResponse: { stream in
for try await response in stream.messages {
if !response.error.isEmpty {
throw ContainerizationError(.internalError, message: "copy: \(response.error)")
}
switch response.status {
case .metadata:
onMetadata(CopyMetadata(isArchive: response.isArchive, totalSize: response.totalSize))
case .complete:
break
case .UNRECOGNIZED(let value):
throw ContainerizationError(.internalError, message: "copy: unrecognized response status \(value)")
}
}
})
}
}
extension Hosts {
func toAgentHostsRequest(location: String) -> Com_Apple_Containerization_Sandbox_V3_ConfigureHostsRequest {
Com_Apple_Containerization_Sandbox_V3_ConfigureHostsRequest.with {
$0.location = location
if let comment {
$0.comment = comment
}
$0.entries = entries.map {
let entry = $0
return Com_Apple_Containerization_Sandbox_V3_ConfigureHostsRequest.HostsEntry.with {
if let comment = entry.comment {
$0.comment = comment
}
$0.ipAddress = entry.ipAddress
$0.hostnames = entry.hostnames
}
}
}
}
}
extension StatCategory {
/// Convert StatCategory to proto enum values.
func toProtoCategories() -> [Com_Apple_Containerization_Sandbox_V3_StatCategory] {
var categories: [Com_Apple_Containerization_Sandbox_V3_StatCategory] = []
if contains(.process) {
categories.append(.process)
}
if contains(.memory) {
categories.append(.memory)
}
if contains(.cpu) {
categories.append(.cpu)
}
if contains(.blockIO) {
categories.append(.blockIo)
}
if contains(.network) {
categories.append(.network)
}
if contains(.memoryEvents) {
categories.append(.memoryEvents)
}
return categories
}
}
extension FilesystemOperation {
/// Convert FilesystemOperation to proto oneof value.
fileprivate func toProtoOperation() -> Com_Apple_Containerization_Sandbox_V3_FilesystemOperationRequest.OneOf_Operation {
switch self {
case .freeze:
return .freeze(.init())
case .thaw:
return .thaw(.init())
case .trim:
return .trim(
.with {
$0.oneShot = .init()
})
}
}
}
+302
View File
@@ -0,0 +1,302 @@
//===----------------------------------------------------------------------===//
// Copyright © 2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
#if os(macOS)
import ContainerizationError
import ContainerizationExtras
import Virtualization
import vmnet
/// A network backed by vmnet on macOS.
@available(macOS 26.0, *)
public struct VmnetNetwork: Network {
private var allocator: Allocator
// `reference` isn't used concurrently.
nonisolated(unsafe) private let reference: vmnet_network_ref
/// The IPv4 subnet of this network.
public let subnet: CIDRv4
/// The IPv6 prefix of this network.
public let prefixV6: CIDRv6?
/// The IPv4 gateway address of this network.
public var ipv4Gateway: IPv4Address {
subnet.gateway
}
/// The IPv6 gateway address of this network, if a prefix exists.
public var ipv6Gateway: IPv6Address? {
prefixV6?.gateway
}
struct Allocator: Sendable {
private let indexAllocatorV4: any AddressAllocator<UInt32>
private let indexAllocatorV6: (any AddressAllocator<UInt32>)?
private let cidrV4: CIDRv4
private let cidrV6: CIDRv6?
private var allocations: [String: (v4: UInt32, v6: UInt32?)]
init(cidrV4: CIDRv4, cidrV6: CIDRv6?) throws {
self.cidrV4 = cidrV4
self.cidrV6 = cidrV6
self.allocations = .init()
let v4Size = Int(cidrV4.upper.value - cidrV4.lower.value - 3)
self.indexAllocatorV4 = try UInt32.rotatingAllocator(
lower: cidrV4.lower.value + 2,
size: UInt32(v4Size)
)
if cidrV6 != nil {
// Independent v6 allocator. The host portion is sourced from a
// UInt32 index regardless of prefix length, and we never need
// more v6 entries than v4 can serve.
self.indexAllocatorV6 = try UInt32.rotatingAllocator(
lower: 2,
size: UInt32(v4Size)
)
} else {
self.indexAllocatorV6 = nil
}
}
mutating func allocate(_ id: String) throws -> (CIDRv4, CIDRv6?) {
if allocations[id] != nil {
throw ContainerizationError(.exists, message: "allocation with id \(id) already exists")
}
let v4Index = try indexAllocatorV4.allocate()
let v4 = try CIDRv4(IPv4Address(v4Index), prefix: cidrV4.prefix)
var v6Index: UInt32? = nil
let v6: CIDRv6?
if let indexAllocatorV6, let cidrV6 {
do {
let idx = try indexAllocatorV6.allocate()
v6Index = idx
let v6Value = (cidrV6.address.value & cidrV6.prefix.prefixMask128) | UInt128(idx)
v6 = try CIDRv6(IPv6Address(v6Value), prefix: cidrV6.prefix)
} catch {
// Roll back v4 so the pair stays atomic.
try? indexAllocatorV4.release(v4Index)
throw error
}
} else {
v6 = nil
}
allocations[id] = (v4: v4Index, v6: v6Index)
return (v4, v6)
}
mutating func release(_ id: String) throws {
if let entry = self.allocations[id] {
try indexAllocatorV4.release(entry.v4)
if let v6Index = entry.v6 {
try indexAllocatorV6?.release(v6Index)
}
allocations.removeValue(forKey: id)
}
}
}
/// A network interface supporting the vmnet_network_ref.
public struct Interface: Containerization.Interface, VZInterface, Sendable {
public let ipv4Address: CIDRv4
public let ipv4Gateway: IPv4Address?
public let ipv6Address: CIDRv6?
public let ipv6Gateway: IPv6Address?
public let macAddress: MACAddress?
public let mtu: UInt32
// `reference` isn't used concurrently.
nonisolated(unsafe) private let reference: vmnet_network_ref
public init(
reference: vmnet_network_ref,
ipv4Address: CIDRv4,
ipv4Gateway: IPv4Address? = nil,
ipv6Address: CIDRv6? = nil,
ipv6Gateway: IPv6Address? = nil,
macAddress: MACAddress? = nil,
mtu: UInt32 = 1500
) {
self.ipv4Address = ipv4Address
self.ipv4Gateway = ipv4Gateway
self.ipv6Address = ipv6Address
self.ipv6Gateway = ipv6Gateway
self.macAddress = macAddress
self.mtu = mtu
self.reference = reference
}
/// Returns the underlying `VZVirtioNetworkDeviceConfiguration`.
public func device() throws -> VZVirtioNetworkDeviceConfiguration {
let config = VZVirtioNetworkDeviceConfiguration()
if let macAddress = self.macAddress {
guard let mac = VZMACAddress(string: macAddress.description) else {
throw ContainerizationError(.invalidArgument, message: "invalid mac address \(macAddress)")
}
config.macAddress = mac
}
config.attachment = VZVmnetNetworkDeviceAttachment(network: self.reference)
return config
}
}
/// Creates a new network.
/// - Parameters:
/// - mode: The vmnet operating mode. Defaults to `.VMNET_SHARED_MODE`.
/// - subnetV4: The IPv4 subnet to use for this network.
/// - prefixV6: The IPv6 prefix to use for this network.
public init(
mode: vmnet.operating_modes_t = .VMNET_SHARED_MODE,
subnet: CIDRv4? = nil,
prefixV6: CIDRv6? = nil
) throws {
var status: vmnet_return_t = .VMNET_FAILURE
guard let config = vmnet_network_configuration_create(mode, &status) else {
throw ContainerizationError(.unsupported, message: "failed to create vmnet config with status \(status)")
}
vmnet_network_configuration_disable_dhcp(config)
if let subnet {
try Self.configureSubnetV4(config, subnetV4: subnet)
}
if let prefixV6 {
try Self.configurePrefixV6(config, prefixV6: prefixV6)
}
guard let ref = vmnet_network_create(config, &status), status == .VMNET_SUCCESS else {
throw ContainerizationError(.unsupported, message: "failed to create vmnet network with status \(status)")
}
let cidrV4 = try Self.getSubnetV4(ref)
let cidrV6 = Self.getPrefixV6(ref)
self.allocator = try .init(cidrV4: cidrV4, cidrV6: cidrV6)
self.subnet = cidrV4
self.prefixV6 = cidrV6
self.reference = ref
}
/// Returns a new interface for use with a container. Allocates an IPv4
/// address from the network's subnet, and — when the network has an IPv6
/// prefix — an IPv6 address from that prefix. The two allocations are
/// independent.
/// - Parameter id: The container ID.
public mutating func createInterface(_ id: String) throws -> Containerization.Interface? {
let (v4, v6) = try allocator.allocate(id)
return Self.Interface(
reference: self.reference,
ipv4Address: v4,
ipv4Gateway: self.ipv4Gateway,
ipv6Address: v6,
ipv6Gateway: self.ipv6Gateway
)
}
/// Returns a new interface for use with a container with a custom MTU.
/// - Parameters:
/// - id: The container ID.
/// - mtu: The MTU for the interface.
public mutating func createInterface(_ id: String, mtu: UInt32) throws -> Containerization.Interface? {
let (v4, v6) = try allocator.allocate(id)
return Self.Interface(
reference: self.reference,
ipv4Address: v4,
ipv4Gateway: self.ipv4Gateway,
ipv6Address: v6,
ipv6Gateway: self.ipv6Gateway,
mtu: mtu
)
}
/// Returns a new interface without a default gateway route. Useful for
/// secondary interfaces where another interface already provides the
/// default route.
/// - Parameter id: The container ID.
public mutating func createInterfaceWithoutGateway(_ id: String) throws -> Containerization.Interface? {
let (v4, v6) = try allocator.allocate(id)
return Self.Interface(
reference: self.reference,
ipv4Address: v4,
ipv6Address: v6
)
}
/// Performs cleanup of an interface.
/// - Parameter id: The container ID.
public mutating func releaseInterface(_ id: String) throws {
try allocator.release(id)
}
private static func getSubnetV4(_ ref: vmnet_network_ref) throws -> CIDRv4 {
var subnet = in_addr()
var mask = in_addr()
vmnet_network_get_ipv4_subnet(ref, &subnet, &mask)
let sa = UInt32(bigEndian: subnet.s_addr)
let mv = UInt32(bigEndian: mask.s_addr)
let lower = IPv4Address(sa & mv)
let upper = IPv4Address(lower.value + ~mv)
return try CIDRv4(lower: lower, upper: upper)
}
private static func configureSubnetV4(_ config: vmnet_network_configuration_ref, subnetV4: CIDRv4) throws {
let gateway = subnetV4.gateway
var ga = in_addr()
inet_pton(AF_INET, gateway.description, &ga)
let mask = IPv4Address(subnetV4.prefix.prefixMask32)
var ma = in_addr()
inet_pton(AF_INET, mask.description, &ma)
guard vmnet_network_configuration_set_ipv4_subnet(config, &ga, &ma) == .VMNET_SUCCESS else {
throw ContainerizationError(.internalError, message: "failed to set IPv4 subnet \(subnetV4) for network")
}
}
private static func getPrefixV6(_ ref: vmnet_network_ref) -> CIDRv6? {
var p = in6_addr()
var len: UInt8 = 0
vmnet_network_get_ipv6_prefix(ref, &p, &len)
guard len > 0, let prefix = Prefix.ipv6(len) else {
return nil
}
let bytes: [UInt8] = withUnsafeBytes(of: p) { Array($0) }
guard let address = try? IPv6Address(bytes) else {
return nil
}
return try? CIDRv6(address, prefix: prefix)
}
private static func configurePrefixV6(_ config: vmnet_network_configuration_ref, prefixV6: CIDRv6) throws {
var p = in6_addr()
inet_pton(AF_INET6, prefixV6.lower.description, &p)
guard vmnet_network_configuration_set_ipv6_prefix(config, &p, prefixV6.prefix.length) == .VMNET_SUCCESS else {
throw ContainerizationError(.internalError, message: "failed to set IPv6 prefix \(prefixV6) for network")
}
}
}
#endif
@@ -0,0 +1,76 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import Foundation
#if os(macOS)
import Virtualization
#endif
/// A stream of vsock connections.
public final class VsockListener: NSObject, Sendable, AsyncSequence {
public typealias Element = FileHandle
/// The port the connections are for.
public let port: UInt32
private let connections: AsyncStream<FileHandle>
private let cont: AsyncStream<FileHandle>.Continuation
private let stopListening: @Sendable (_ port: UInt32) throws -> Void
package init(port: UInt32, stopListen: @Sendable @escaping (_ port: UInt32) throws -> Void) {
self.port = port
let (stream, continuation) = AsyncStream.makeStream(of: FileHandle.self)
self.connections = stream
self.cont = continuation
self.stopListening = stopListen
}
public func finish() throws {
self.cont.finish()
try self.stopListening(self.port)
}
public func makeAsyncIterator() -> AsyncStream<FileHandle>.AsyncIterator {
connections.makeAsyncIterator()
}
}
#if os(macOS)
extension VsockListener: VZVirtioSocketListenerDelegate {
public func listener(
_: VZVirtioSocketListener, shouldAcceptNewConnection conn: VZVirtioSocketConnection,
from _: VZVirtioSocketDevice
) -> Bool {
let fd = dup(conn.fileDescriptor)
guard fd != -1 else {
return false
}
conn.close()
let fh = FileHandle(fileDescriptor: fd, closeOnDealloc: false)
let result = cont.yield(fh)
if case .terminated = result {
try? fh.close()
return false
}
return true
}
}
#endif
@@ -0,0 +1,103 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import CArchive
import Foundation
/// An enumeration of the errors that can be thrown while interacting with an archive.
public enum ArchiveError: Error, CustomStringConvertible {
case unableToCreateArchive
case noUnderlyingArchive
case noArchiveInCallback
case noDelegateConfigured
case delegateFreedBeforeCallback
case unableToSetFormat(CInt, Format)
case unableToAddFilter(CInt, Filter)
case unableToWriteEntryHeader(CInt)
case unableToWriteData(CLong)
case unableToCloseArchive(CInt)
case unableToOpenArchive(CInt)
case unableToSetOption(CInt)
case failedToSetLocale(locales: [String])
case failedToGetProperty(String, URLResourceKey)
case failedToDetectFilter
case failedToDetectFormat
case failedToExtractArchive(String)
case failedToCreateArchive(String)
case invalidBaseAddressArchiveWrite
/// Description of the error
public var description: String {
switch self {
case .unableToCreateArchive:
return "unable to create an archive."
case .noUnderlyingArchive:
return "no underlying archive was provided."
case .noArchiveInCallback:
return "no archive was provided in the callback."
case .noDelegateConfigured:
return "no delegate was configured."
case .delegateFreedBeforeCallback:
return "the delegate was freed before the callback was invoked."
case .unableToSetFormat(let code, let name):
return "unable to set the archive format \(name), code \(code)"
case .unableToAddFilter(let code, let name):
return "unable to set the archive filter \(name), code \(code)"
case .unableToWriteEntryHeader(let code):
return "unable to write the entry header to the archive, code \(code)"
case .unableToWriteData(let code):
return "unable to write data to the archive, code \(code)"
case .unableToCloseArchive(let code):
return "unable to close the archive, code \(code)"
case .unableToOpenArchive(let code):
return "unable to open the archive, code \(code)"
case .unableToSetOption(_):
return "unable to set an option on the archive."
case .failedToSetLocale(let locales):
return "failed to set locale to \(locales)"
case .failedToGetProperty(let path, let propertyName):
return "failed to read property \(propertyName) from file at path \(path)"
case .failedToDetectFilter:
return "failed to detect filter from archive."
case .failedToDetectFormat:
return "failed to detect format from archive."
case .failedToExtractArchive(let reason):
return "failed to extract archive: \(reason)"
case .failedToCreateArchive(let reason):
return "failed to create archive: \(reason)"
case .invalidBaseAddressArchiveWrite:
return "got an invalid base address for pointer when writing data to archive"
}
}
}
public struct LibArchiveError: Error {
public let source: ArchiveError
public let description: String
}
func wrap(_ f: @autoclosure () -> CInt, _ e: (CInt) -> ArchiveError, underlying: OpaquePointer? = nil) throws {
let result = f()
guard result == ARCHIVE_OK else {
let error = e(result)
guard let underlying = underlying,
let description = archive_error_string(underlying).map(String.init(cString:))
else {
throw error
}
throw LibArchiveError(source: error, description: description)
}
}
@@ -0,0 +1,428 @@
//===----------------------------------------------------------------------===//
// Copyright © 2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import CArchive
import ContainerizationError
import ContainerizationOS
import Foundation
import SystemPackage
/// A protocol for reading data in chunks, compatible with both `InputStream` and zero-allocation archive readers.
public protocol ReadableStream {
/// Reads up to `maxLength` bytes into the provided buffer.
/// Returns the number of bytes actually read, 0 for EOF, or -1 for error.
func read(_ buffer: UnsafeMutablePointer<UInt8>, maxLength: Int) -> Int
}
extension InputStream: ReadableStream {}
/// Small wrapper type to read data from an archive entry.
public struct ArchiveEntryReader: ReadableStream {
private weak var reader: ArchiveReader?
init(reader: ArchiveReader) {
self.reader = reader
}
/// Reads up to `maxLength` bytes into the provided buffer.
/// Returns the number of bytes actually read, 0 for EOF, or -1 for error.
public func read(_ buffer: UnsafeMutablePointer<UInt8>, maxLength: Int) -> Int {
guard let archive = reader?.underlying else { return -1 }
let bytesRead = archive_read_data(archive, buffer, maxLength)
return bytesRead < 0 ? -1 : bytesRead
}
}
/// A class responsible for reading entries from an archive file.
public final class ArchiveReader {
private static let chunkSize = 4 * 1024 * 1024
/// A pointer to the underlying `archive` C structure.
var underlying: OpaquePointer?
/// The file handle associated with the archive file being read.
let fileHandle: FileHandle?
/// Temporary decompressed file URL if the input was zstd-compressed
private var tempDecompressedFile: URL?
/// Initializes an `ArchiveReader` to read from a specified file URL with an explicit `Format` and `Filter`.
/// Note: This method must be used when it is known that the archive at the specified URL follows the specified
/// `Format` and `Filter`.
public convenience init(format: Format, filter: Filter, file: URL) throws {
// If filter is zstd, decompress it and use filter .none
let fileToRead: URL
let tempFile: URL?
let actualFilter: Filter
if filter == .zstd {
let decompressed = try Self.decompressZstd(file)
tempFile = decompressed
fileToRead = decompressed
actualFilter = .none
} else {
tempFile = nil
fileToRead = file
actualFilter = filter
}
do {
let fileHandle = try FileHandle(forReadingFrom: fileToRead)
try self.init(format: format, filter: actualFilter, fileHandle: fileHandle)
} catch {
if let tempFile {
try? FileManager.default.removeItem(at: tempFile.deletingLastPathComponent())
}
throw error
}
self.tempDecompressedFile = tempFile
}
/// Initializes an `ArchiveReader` to read from the provided file descriptor with an explicit `Format` and `Filter`.
/// Note: This method must be used when it is known that the archive pointed to by the file descriptor follows the specified
/// `Format` and `Filter`.
public init(format: Format, filter: Filter, fileHandle: FileHandle) throws {
self.underlying = archive_read_new()
self.fileHandle = fileHandle
try archive_read_set_format(underlying, format.code)
.checkOk(elseThrow: .unableToSetFormat(format.code, format))
try archive_read_append_filter(underlying, filter.code)
.checkOk(elseThrow: .unableToAddFilter(filter.code, filter))
let fd = fileHandle.fileDescriptor
try archive_read_open_fd(underlying, fd, 4096)
.checkOk(elseThrow: { .unableToOpenArchive($0) })
}
/// Initialize the `ArchiveReader` to read from a specified file URL
/// by trying to auto determine the archives `Format` and `Filter`.
public init(file: URL) throws {
self.underlying = archive_read_new()
// Try to decompress as zstd first, fall back to original if it fails
let fileToRead: URL
if let decompressed = try? Self.decompressZstd(file) {
self.tempDecompressedFile = decompressed
fileToRead = decompressed
} else {
fileToRead = file
}
let fileHandle = try FileHandle(forReadingFrom: fileToRead)
self.fileHandle = fileHandle
try archive_read_support_filter_all(underlying)
.checkOk(elseThrow: .failedToDetectFilter)
try archive_read_support_format_all(underlying)
.checkOk(elseThrow: .failedToDetectFormat)
let fd = fileHandle.fileDescriptor
try archive_read_open_fd(underlying, fd, 4096)
.checkOk(elseThrow: { .unableToOpenArchive($0) })
}
/// Decompress a zstd file to a temporary location
public static func decompressZstd(_ source: URL) throws -> URL {
guard let tempDir = createTemporaryDirectory(baseName: "zstd-decompress") else {
throw ArchiveError.failedToDetectFormat
}
let tempFile = tempDir.appendingPathComponent(
source.deletingPathExtension().lastPathComponent
)
do {
let srcPath = source.path
let srcFd = open(srcPath, O_RDONLY)
guard srcFd >= 0 else { throw ArchiveError.failedToDetectFormat }
defer { close(srcFd) }
let dstFd = open(tempFile.path, O_WRONLY | O_CREAT | O_TRUNC, 0o644)
guard dstFd >= 0 else { throw ArchiveError.failedToDetectFormat }
defer { close(dstFd) }
guard zstd_decompress_fd(srcFd, dstFd) == 0 else {
throw ArchiveError.failedToDetectFormat
}
} catch {
try? FileManager.default.removeItem(at: tempDir)
throw error
}
return tempFile
}
/// Clean up the temporary directory created by `decompressZstd`.
/// The decompressed file is placed inside a unique temporary directory,
/// so removing that directory cleans up everything.
public static func cleanUpDecompressedZstd(_ file: URL) {
try? FileManager.default.removeItem(at: file.deletingLastPathComponent())
}
deinit {
archive_read_free(underlying)
try? fileHandle?.close()
if let tempFile = tempDecompressedFile {
Self.cleanUpDecompressedZstd(tempFile)
}
}
}
extension CInt {
fileprivate func checkOk(elseThrow error: @autoclosure () -> ArchiveError) throws {
guard self == ARCHIVE_OK else { throw error() }
}
fileprivate func checkOk(elseThrow error: (CInt) -> ArchiveError) throws {
guard self == ARCHIVE_OK else { throw error(self) }
}
}
extension ArchiveReader: Sequence {
public func makeIterator() -> Iterator {
Iterator(reader: self)
}
public struct Iterator: IteratorProtocol {
var reader: ArchiveReader
public mutating func next() -> (WriteEntry, Data)? {
let entry = WriteEntry()
let result = archive_read_next_header2(reader.underlying, entry.underlying)
if result == ARCHIVE_EOF {
return nil
}
let data = reader.readDataForEntry(entry)
return (entry, data)
}
}
/// Returns an iterator that yields archive entries.
public func makeStreamingIterator() -> StreamingIterator {
StreamingIterator(reader: self)
}
public struct StreamingIterator: Sequence, IteratorProtocol {
var reader: ArchiveReader
public func makeIterator() -> StreamingIterator {
self
}
public mutating func next() -> (WriteEntry, ArchiveEntryReader)? {
let entry = WriteEntry()
let result = archive_read_next_header2(reader.underlying, entry.underlying)
if result == ARCHIVE_EOF {
return nil
}
let streamReader = ArchiveEntryReader(reader: reader)
return (entry, streamReader)
}
}
internal func readDataForEntry(_ entry: WriteEntry) -> Data {
let bufferSize = Int(Swift.min(entry.size ?? 4096, 4096))
var entry = Data()
var part = Data(count: bufferSize)
while true {
let c = part.withUnsafeMutableBytes { buffer in
guard let baseAddress = buffer.baseAddress else {
return 0
}
return archive_read_data(self.underlying, baseAddress, buffer.count)
}
guard c > 0 else { break }
part.count = c
entry.append(part)
}
return entry
}
}
extension ArchiveReader {
public convenience init(name: String, bundle: Data, tempDirectoryBaseName: String? = nil) throws {
let baseName = tempDirectoryBaseName ?? "Unarchiver"
guard let tempDir = createTemporaryDirectory(baseName: baseName) else {
throw ArchiveError.failedToExtractArchive("failed to create temporary directory")
}
let url = tempDir.appendingPathComponent(name)
do {
try bundle.write(to: url, options: .atomic)
try self.init(format: .zip, filter: .none, file: url)
} catch {
try? FileManager.default.removeItem(at: tempDir)
throw error
}
// Register for cleanup in deinit (only needed when the zstd path didn't already set it)
if self.tempDecompressedFile == nil {
self.tempDecompressedFile = url
}
}
/// Extracts the contents of an archive to the provided directory.
/// Rejects member paths that escape the root directory or traverse
/// symbolic links, and uses a "last entry wins" replacement policy
/// for an existing file at a path to be extracted.
public func extractContents(to directory: URL) throws -> [String] {
// Create the root directory with standard permissions
// and create a FileDescriptor for secure path traversal.
let fm = FileManager.default
let rootFilePath = FilePath(directory.path)
try fm.createDirectory(atPath: directory.path, withIntermediateDirectories: true)
let rootFileDescriptor = try FileDescriptor.open(rootFilePath, .readOnly)
defer { try? rootFileDescriptor.close() }
// Iterate and extract archive entries, collecting rejected paths.
var foundEntry = false
var rejectedPaths = [String]()
for (entry, dataReader) in self.makeStreamingIterator() {
guard let memberPath = (entry.path.map { FilePath($0) }) else {
continue
}
foundEntry = true
// Try to extract the entry, catching path validation errors
let extracted = try extractEntry(
entry: entry,
dataReader: dataReader,
memberPath: memberPath,
rootFileDescriptor: rootFileDescriptor
)
if !extracted {
rejectedPaths.append(memberPath.string)
}
}
guard foundEntry else {
throw ArchiveError.failedToExtractArchive("no entries found in archive")
}
return rejectedPaths
}
/// This method extracts a given file from the archive.
/// This operation modifies the underlying file descriptor's position within the archive,
/// meaning subsequent reads will start from a new location.
/// To reset the underlying file descriptor to the beginning of the archive, close and
/// reopen the archive.
public func extractFile(path: String) throws -> (WriteEntry, Data) {
let entry = WriteEntry()
while archive_read_next_header2(self.underlying, entry.underlying) != ARCHIVE_EOF {
guard let entryPath = entry.path else { continue }
let trimCharSet = CharacterSet(charactersIn: "./")
let trimmedEntry = entryPath.trimmingCharacters(in: trimCharSet)
let trimmedRequired = path.trimmingCharacters(in: trimCharSet)
guard trimmedEntry == trimmedRequired else { continue }
let data = readDataForEntry(entry)
return (entry, data)
}
throw ArchiveError.failedToExtractArchive(" \(path) not found in archive")
}
/// Extracts a single archive entry.
/// Returns false if the entry was rejected due to path validation errors.
/// Throws on system errors.
private func extractEntry(
entry: WriteEntry,
dataReader: ArchiveEntryReader,
memberPath: FilePath,
rootFileDescriptor: FileDescriptor
) throws -> Bool {
guard let lastComponent = memberPath.lastComponent else {
return false
}
let relativePath = memberPath.removingLastComponent()
let type = entry.fileType
do {
switch type {
case .regular:
try FileDescriptorOps.mkdir(rootFileDescriptor, relativePath, makeIntermediates: true) { fd in
// Remove existing entry if present (mimics containerd's "last entry wins" behavior)
try? FileDescriptorOps.unlinkRecursive(fd, filename: lastComponent)
// Open file for writing using openat with O_NOFOLLOW to prevent TOC-TOU attacks
let fileMode = entry.permissions & 0o777 // Mask to permission bits only
let fileFd = openat(fd.rawValue, lastComponent.string, O_WRONLY | O_CREAT | O_EXCL | O_NOFOLLOW, fileMode)
guard fileFd >= 0 else {
throw ArchiveError.failedToExtractArchive("failed to create file: \(memberPath)")
}
defer { close(fileFd) }
try Self.copyDataReaderToFd(dataReader: dataReader, fileFd: fileFd, memberPath: memberPath)
setFileAttributes(fd: fileFd, entry: entry)
}
case .directory:
try FileDescriptorOps.mkdir(rootFileDescriptor, memberPath, makeIntermediates: true) { fd in
setFileAttributes(fd: fd.rawValue, entry: entry)
}
case .symbolicLink:
guard let targetPath = (entry.symlinkTarget.map { FilePath($0) }) else {
return false
}
var symlinkCreated = false
try FileDescriptorOps.mkdir(rootFileDescriptor, relativePath, makeIntermediates: true) { fd in
// Remove existing entry if present (mimics containerd's "last entry wins" behavior)
try? FileDescriptorOps.unlinkRecursive(fd, filename: lastComponent)
guard symlinkat(targetPath.string, fd.rawValue, lastComponent.string) == 0 else {
throw ArchiveError.failedToExtractArchive("failed to create symlink: \(targetPath) <- \(memberPath)")
}
symlinkCreated = true
}
return symlinkCreated
default:
return false
}
return true
} catch let error as FileDescriptorOps.Error {
// Just reject path validation errors, don't fail the extraction
switch error {
case .systemError:
// Fail for system errors
throw error
case .invalidRelativePath, .invalidPathComponent, .cannotFollowSymlink:
return false
}
}
}
private func setFileAttributes(fd: Int32, entry: WriteEntry) {
fchmod(fd, entry.permissions)
if let owner = entry.owner, let group = entry.group {
fchown(fd, owner, group)
}
}
private static func copyDataReaderToFd(dataReader: ArchiveEntryReader, fileFd: Int32, memberPath: FilePath) throws {
var buffer = [UInt8](repeating: 0, count: ArchiveReader.chunkSize)
while true {
let bytesRead = buffer.withUnsafeMutableBufferPointer { bufferPtr in
guard let baseAddress = bufferPtr.baseAddress else { return 0 }
return dataReader.read(baseAddress, maxLength: bufferPtr.count)
}
if bytesRead < 0 {
throw ArchiveError.failedToExtractArchive("failed to read data for: \(memberPath)")
}
if bytesRead == 0 {
break // EOF
}
let bytesWritten = write(fileFd, buffer, bytesRead)
guard bytesWritten == bytesRead else {
throw ArchiveError.failedToExtractArchive("failed to write data for: \(memberPath)")
}
}
}
}
@@ -0,0 +1,339 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import CArchive
import Foundation
import SystemPackage
/// A class responsible for writing archives in various formats.
public final class ArchiveWriter {
private static let chunkSize = 4 * 1024 * 1024
var underlying: OpaquePointer?
/// Initialize a new `ArchiveWriter` with the given configuration.
/// This method attempts to initialize an empty archive in memory, failing which it throws a `unableToCreateArchive` error.
public init(configuration: ArchiveWriterConfiguration) throws {
// because for some bizarre reason, UTF8 paths won't work unless this process explicitly sets a locale like en_US.UTF-8
try Self.attemptSetLocales(locales: configuration.locales)
guard let underlying = archive_write_new() else { throw ArchiveError.unableToCreateArchive }
self.underlying = underlying
try setFormat(configuration.format)
try addFilter(configuration.filter)
try setOptions(configuration.options)
}
/// Initialize a new `ArchiveWriter` for writing into the specified file with the given configuration options.
public convenience init(format: Format, filter: Filter, options: [Options] = [], locales: [String] = ArchiveWriterConfiguration.defaultLocales, file: URL) throws {
let config = ArchiveWriterConfiguration(
format: format,
filter: filter,
options: options,
locales: locales
)
try self.init(configuration: config)
try self.open(file: file)
}
/// Opens the given file for writing data into
public func open(file: URL) throws {
guard let underlying = underlying else { throw ArchiveError.noUnderlyingArchive }
let res = archive_write_open_filename(underlying, file.path)
try wrap(res, ArchiveError.unableToOpenArchive, underlying: underlying)
}
/// Opens the given fd for writing data into
public func open(fileDescriptor: Int32) throws {
guard let underlying = underlying else { throw ArchiveError.noUnderlyingArchive }
let res = archive_write_open_fd(underlying, fileDescriptor)
try wrap(res, ArchiveError.unableToOpenArchive, underlying: underlying)
}
/// Performs any necessary finalizations on the archive and releases resources.
public func finishEncoding() throws {
guard let u = underlying else { return }
underlying = nil
let r = archive_free(u)
guard r == ARCHIVE_OK else {
throw ArchiveError.unableToCloseArchive(r)
}
}
deinit {
if let u = underlying {
archive_free(u)
underlying = nil
}
}
private static func attemptSetLocales(locales: [String]) throws {
for locale in locales {
if setlocale(LC_ALL, locale) != nil {
return
}
}
throw ArchiveError.failedToSetLocale(locales: locales)
}
}
public class ArchiveWriterTransaction {
private let writer: ArchiveWriter
fileprivate init(writer: ArchiveWriter) {
self.writer = writer
}
public func writeHeader(entry: WriteEntry) throws {
try writer.writeHeader(entry: entry)
}
public func writeChunk(data: UnsafeRawBufferPointer) throws {
try writer.writeData(data: data)
}
public func finish() throws {
try writer.finishEntry()
}
}
extension ArchiveWriter {
public func makeTransactionWriter() -> ArchiveWriterTransaction {
ArchiveWriterTransaction(writer: self)
}
/// Create a new entry in the archive with the given properties.
/// - Parameters:
/// - entry: A `WriteEntry` object describing the metadata of the entry to be created
/// (e.g., name, modification date, permissions).
/// - data: The `Data` object containing the content for the new entry.
public func writeEntry(entry: WriteEntry, data: Data) throws {
try data.withUnsafeBytes { bytes in
try writeEntry(entry: entry, data: bytes)
}
}
/// Creates a new entry in the archive with the given properties.
///
/// This method performs the following:
/// 1. Writes the archive header using the provided `WriteEntry` metadata.
/// 2. Writes the content from the `UnsafeRawBufferPointer` into the archive.
/// 3. Finalizes the entry in the archive.
///
/// - Parameters:
/// - entry: A `WriteEntry` object describing the metadata of the entry to be created
/// (e.g., name, modification date, permissions, type).
/// - data: An optional `UnsafeRawBufferPointer` containing the raw bytes for the new entry's
/// content. Pass `nil` for entries that do not have content data (e.g., directories, symlinks).
public func writeEntry(entry: WriteEntry, data: UnsafeRawBufferPointer?) throws {
try writeHeader(entry: entry)
if let data = data {
try writeData(data: data)
}
try finishEntry()
}
fileprivate func writeHeader(entry: WriteEntry) throws {
guard let underlying = self.underlying else { throw ArchiveError.noUnderlyingArchive }
try wrap(
archive_write_header(underlying, entry.underlying), ArchiveError.unableToWriteEntryHeader,
underlying: underlying)
}
fileprivate func finishEntry() throws {
guard let underlying = self.underlying else { throw ArchiveError.noUnderlyingArchive }
archive_write_finish_entry(underlying)
}
fileprivate func writeData(data: UnsafeRawBufferPointer) throws {
guard let underlying = self.underlying else { throw ArchiveError.noUnderlyingArchive }
var offset = 0
while offset < data.count {
guard let baseAddress = data.baseAddress?.advanced(by: offset) else {
throw ArchiveError.invalidBaseAddressArchiveWrite
}
let result = archive_write_data(underlying, baseAddress, data.count - offset)
guard result > 0 else {
throw ArchiveError.unableToWriteData(result)
}
offset += Int(result)
}
}
}
extension ArchiveWriter {
private func archive(_ relativePath: FilePath, dirPath: FilePath) throws {
let fm = FileManager.default
let fullPath = dirPath.appending(relativePath.string)
var statInfo = stat()
guard lstat(fullPath.string, &statInfo) == 0 else {
let errNo = errno
let err = POSIXErrorCode(rawValue: errNo) ?? .EINVAL
throw ArchiveError.failedToCreateArchive("lstat failed for '\(fullPath)': \(POSIXError(err))")
}
let mode = statInfo.st_mode
let uid = statInfo.st_uid
let gid = statInfo.st_gid
var size: Int64 = 0
let type: URLFileResourceType
if (mode & S_IFMT) == S_IFREG {
type = .regular
size = Int64(statInfo.st_size)
} else if (mode & S_IFMT) == S_IFDIR {
type = .directory
} else if (mode & S_IFMT) == S_IFLNK {
type = .symbolicLink
} else {
return
}
#if os(macOS)
let created = Date(timeIntervalSince1970: Double(statInfo.st_ctimespec.tv_sec))
let access = Date(timeIntervalSince1970: Double(statInfo.st_atimespec.tv_sec))
let modified = Date(timeIntervalSince1970: Double(statInfo.st_mtimespec.tv_sec))
#else
let created = Date(timeIntervalSince1970: Double(statInfo.st_ctim.tv_sec))
let access = Date(timeIntervalSince1970: Double(statInfo.st_atim.tv_sec))
let modified = Date(timeIntervalSince1970: Double(statInfo.st_mtim.tv_sec))
#endif
let entry = WriteEntry()
if type == .symbolicLink {
let targetPath = try fm.destinationOfSymbolicLink(atPath: fullPath.string)
// Resolve the target relative to the symlink's parent, not the archive root.
let symlinkParent = fullPath.removingLastComponent()
let resolvedFull = symlinkParent.appending(targetPath).lexicallyNormalized()
guard resolvedFull.starts(with: dirPath) else {
return
}
entry.symlinkTarget = targetPath
}
entry.path = relativePath.string
entry.size = size
entry.creationDate = created
entry.modificationDate = modified
entry.contentAccessDate = access
entry.fileType = type
entry.group = gid
entry.owner = uid
entry.permissions = mode
if type == .regular {
let buf = UnsafeMutableRawBufferPointer.allocate(byteCount: Self.chunkSize, alignment: 1)
guard let baseAddress = buf.baseAddress else {
throw ArchiveError.failedToCreateArchive("cannot create temporary buffer of size \(Self.chunkSize)")
}
defer { buf.deallocate() }
let fd = Foundation.open(fullPath.string, O_RDONLY)
guard fd >= 0 else {
let err = POSIXErrorCode(rawValue: errno) ?? .EINVAL
throw ArchiveError.failedToCreateArchive("cannot open file \(fullPath.string) for reading: \(err)")
}
defer { close(fd) }
try self.writeHeader(entry: entry)
while true {
let n = read(fd, baseAddress, Self.chunkSize)
if n == 0 { break }
if n < 0 {
let err = POSIXErrorCode(rawValue: errno) ?? .EIO
throw ArchiveError.failedToCreateArchive("failed to read from file \(fullPath.string): \(err)")
}
try self.writeData(data: UnsafeRawBufferPointer(start: baseAddress, count: n))
}
try self.finishEntry()
} else {
try self.writeEntry(entry: entry, data: nil)
}
}
/// Recursively archives the content of a directory. Regular files, symlinks and directories are added into the archive.
/// Note: Symlinks are added to the archive if both the source and target for the symlink are both contained in the top level directory.
public func archiveDirectory(_ dir: URL) throws {
let fm = FileManager.default
let dirPath = FilePath(dir.path)
guard let enumerator = fm.enumerator(atPath: dirPath.string) else {
throw POSIXError(.ENOTDIR)
}
// Emit a leading "./" entry for the root directory, matching GNU/BSD tar behavior.
var rootStat = stat()
guard lstat(dirPath.string, &rootStat) == 0 else {
let err = POSIXErrorCode(rawValue: errno) ?? .EINVAL
throw ArchiveError.failedToCreateArchive("lstat failed for '\(dirPath)': \(POSIXError(err))")
}
let rootEntry = WriteEntry()
rootEntry.path = "./"
rootEntry.size = 0
rootEntry.fileType = .directory
rootEntry.owner = rootStat.st_uid
rootEntry.group = rootStat.st_gid
rootEntry.permissions = rootStat.st_mode
#if os(macOS)
rootEntry.creationDate = Date(timeIntervalSince1970: Double(rootStat.st_ctimespec.tv_sec))
rootEntry.contentAccessDate = Date(timeIntervalSince1970: Double(rootStat.st_atimespec.tv_sec))
rootEntry.modificationDate = Date(timeIntervalSince1970: Double(rootStat.st_mtimespec.tv_sec))
#else
rootEntry.creationDate = Date(timeIntervalSince1970: Double(rootStat.st_ctim.tv_sec))
rootEntry.contentAccessDate = Date(timeIntervalSince1970: Double(rootStat.st_atim.tv_sec))
rootEntry.modificationDate = Date(timeIntervalSince1970: Double(rootStat.st_mtim.tv_sec))
#endif
try self.writeHeader(entry: rootEntry)
for case let relativePath as String in enumerator {
try archive(FilePath(relativePath), dirPath: dirPath)
}
}
public func archive(_ paths: [FilePath], base: FilePath) throws {
let fm = FileManager.default
let base = base.lexicallyNormalized()
for path in paths {
guard path.starts(with: base) else {
throw ArchiveError.failedToCreateArchive("'\(path.string)' is not under '\(base.string)'")
}
let relativePath = path.components.dropFirst(base.components.count)
.reduce(into: FilePath("")) { $0.append($1) }
var isDir: ObjCBool = false
_ = fm.fileExists(atPath: path.string, isDirectory: &isDir)
if isDir.boolValue {
guard let enumerator = fm.enumerator(atPath: path.string) else {
throw POSIXError(.ENOTDIR)
}
try archive(relativePath, dirPath: base)
for case let child as String in enumerator {
let childPath = relativePath.appending(child)
try archive(childPath, dirPath: base)
}
} else {
try archive(relativePath, dirPath: base)
}
}
}
}
@@ -0,0 +1,200 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import CArchive
/// Represents the configuration settings for an `ArchiveWriter`.
///
/// This struct allows specifying the archive format, compression filter,
/// various format-specific options, and preferred locales for string encoding.
public struct ArchiveWriterConfiguration {
public static let defaultLocales = ["en_US.UTF-8", "C.UTF-8"]
/// The desired archive format
public var format: Format
/// The compression filter to apply to the archive
public var filter: Filter
/// An array of format-specific options to apply to the archive.
/// This includes options like compression level and extended attribute format.
public var options: [Options]
/// An array of preferred locale identifiers for string encoding
public var locales: [String]
/// Initializes a new `ArchiveWriterConfiguration`.
///
/// Sets up the configuration with the specified format, filter, options, and locales.
public init(
format: Format, filter: Filter, options: [Options] = [], locales: [String] = Self.defaultLocales
) {
self.format = format
self.filter = filter
self.options = options
self.locales = locales
}
}
extension ArchiveWriter {
internal func setFormat(_ format: Format) throws {
guard let underlying = self.underlying else { throw ArchiveError.noUnderlyingArchive }
let r = archive_write_set_format(underlying, format.code)
guard r == ARCHIVE_OK else { throw ArchiveError.unableToSetFormat(r, format) }
}
internal func addFilter(_ filter: Filter) throws {
guard let underlying = self.underlying else { throw ArchiveError.noUnderlyingArchive }
let r = archive_write_add_filter(underlying, filter.code)
guard r == ARCHIVE_OK else { throw ArchiveError.unableToAddFilter(r, filter) }
}
internal func setOptions(_ options: [Options]) throws {
guard let underlying = self.underlying else { throw ArchiveError.noUnderlyingArchive }
try options.forEach {
switch $0 {
case .compressionLevel(let level):
try wrap(
archive_write_set_option(underlying, nil, "compression-level", "\(level)"),
ArchiveError.unableToSetOption, underlying: self.underlying)
case .compression(.store):
try wrap(
archive_write_set_option(underlying, nil, "compression", "store"), ArchiveError.unableToSetOption,
underlying: self.underlying)
case .compression(.deflate):
try wrap(
archive_write_set_option(underlying, nil, "compression", "deflate"), ArchiveError.unableToSetOption,
underlying: self.underlying)
case .xattrformat(let value):
let v = value.description
try wrap(
archive_write_set_option(underlying, nil, "xattrheader", v), ArchiveError.unableToSetOption,
underlying: self.underlying)
}
}
}
}
public enum Options {
case compressionLevel(UInt32)
case compression(Compression)
case xattrformat(XattrFormat)
public enum Compression {
case store
case deflate
}
public enum XattrFormat: String, CustomStringConvertible {
case schily
case libarchive
case all
public var description: String {
switch self {
case .libarchive:
return "LIBARCHIVE"
case .schily:
return "SCHILY"
case .all:
return "ALL"
}
}
}
}
/// An enumeration of the supported archive formats.
public enum Format: String, Sendable {
/// POSIX-standard `ustar` archives
case ustar
case gnutar
/// POSIX `pax interchange format` archives
case pax
case paxRestricted
/// POSIX octet-oriented cpio archives
case cpio
case cpioNewc
/// Zip archive
case zip
/// two different variants of shar archives
case shar
case sharDump
/// ISO9660 CD images
case iso9660
/// 7-Zip archives
case sevenZip
/// ar archives
case arBSD
case arGNU
/// mtree file tree descriptions
case mtree
/// XAR archives
case xar
internal var code: CInt {
switch self {
case .ustar: return ARCHIVE_FORMAT_TAR_USTAR
case .pax: return ARCHIVE_FORMAT_TAR_PAX_INTERCHANGE
case .paxRestricted: return ARCHIVE_FORMAT_TAR_PAX_RESTRICTED
case .gnutar: return ARCHIVE_FORMAT_TAR_GNUTAR
case .cpio: return ARCHIVE_FORMAT_CPIO_POSIX
case .cpioNewc: return ARCHIVE_FORMAT_CPIO_AFIO_LARGE
case .zip: return ARCHIVE_FORMAT_ZIP
case .shar: return ARCHIVE_FORMAT_SHAR_BASE
case .sharDump: return ARCHIVE_FORMAT_SHAR_DUMP
case .iso9660: return ARCHIVE_FORMAT_ISO9660
case .sevenZip: return ARCHIVE_FORMAT_7ZIP
case .arBSD: return ARCHIVE_FORMAT_AR_BSD
case .arGNU: return ARCHIVE_FORMAT_AR_GNU
case .mtree: return ARCHIVE_FORMAT_MTREE
case .xar: return ARCHIVE_FORMAT_XAR
}
}
}
/// An enumeration of the supported filters (compression / encoding standards) for an archive.
public enum Filter: String, Sendable {
case none
case gzip
case bzip2
case compress
case lzma
case xz
case uu
case rpm
case lzip
case lrzip
case lzop
case grzip
case lz4
case zstd
internal var code: CInt {
switch self {
case .none: return ARCHIVE_FILTER_NONE
case .gzip: return ARCHIVE_FILTER_GZIP
case .bzip2: return ARCHIVE_FILTER_BZIP2
case .compress: return ARCHIVE_FILTER_COMPRESS
case .lzma: return ARCHIVE_FILTER_LZMA
case .xz: return ARCHIVE_FILTER_XZ
case .uu: return ARCHIVE_FILTER_UU
case .rpm: return ARCHIVE_FILTER_RPM
case .lzip: return ARCHIVE_FILTER_LZIP
case .lrzip: return ARCHIVE_FILTER_LRZIP
case .lzop: return ARCHIVE_FILTER_LZOP
case .grzip: return ARCHIVE_FILTER_GRZIP
case .lz4: return ARCHIVE_FILTER_LZ4
case .zstd: return ARCHIVE_FILTER_ZSTD
}
}
}
@@ -0,0 +1,65 @@
The libarchive distribution as a whole is Copyright by Tim Kientzle
and is subject to the copyright notice reproduced at the bottom of
this file.
Each individual file in this distribution should have a clear
copyright/licensing statement at the beginning of the file. If any do
not, please let me know and I will rectify it. The following is
intended to summarize the copyright status of the individual files;
the actual statements in the files are controlling.
* Except as listed below, all C sources (including .c and .h files)
and documentation files are subject to the copyright notice reproduced
at the bottom of this file.
* The following source files are also subject in whole or in part to
a 3-clause UC Regents copyright; please read the individual source
files for details:
libarchive/archive_read_support_filter_compress.c
libarchive/archive_write_add_filter_compress.c
libarchive/mtree.5
* The following source files are in the public domain:
libarchive/archive_getdate.c
* The following source files are triple-licensed with the ability to choose
from CC0 1.0 Universal, OpenSSL or Apache 2.0 licenses:
libarchive/archive_blake2.h
libarchive/archive_blake2_impl.h
libarchive/archive_blake2s_ref.c
libarchive/archive_blake2sp_ref.c
* The build files---including Makefiles, configure scripts,
and auxiliary scripts used as part of the compile process---have
widely varying licensing terms. Please check individual files before
distributing them to see if those restrictions apply to you.
I intend for all new source code to use the license below and hope over
time to replace code with other licenses with new implementations that
do use the license below. The varying licensing of the build scripts
seems to be an unavoidable mess.
Copyright (c) 2003-2018 <author(s)>
All rights reserved.
Redistribution and use in source and binary forms, with or without
modification, are permitted provided that the following conditions
are met:
1. Redistributions of source code must retain the above copyright
notice, this list of conditions and the following disclaimer
in this position and unchanged.
2. Redistributions in binary form must reproduce the above copyright
notice, this list of conditions and the following disclaimer in the
documentation and/or other materials provided with the distribution.
THIS SOFTWARE IS PROVIDED BY THE AUTHOR(S) ``AS IS'' AND ANY EXPRESS OR
IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES
OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED.
IN NO EVENT SHALL THE AUTHOR(S) BE LIABLE FOR ANY DIRECT, INDIRECT,
INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT
NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
(INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF
THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
@@ -0,0 +1,68 @@
/*
* Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* https://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#include "archive_bridge.h"
#include <zstd.h>
#include <stdlib.h>
#include <unistd.h>
void archive_set_error_wrapper(struct archive *a, int error_number, const char *error_string) {
archive_set_error(a, error_number, "%s", error_string);
}
int zstd_decompress_fd(int src_fd, int dst_fd) {
ZSTD_DStream *dstream = ZSTD_createDStream();
if (!dstream) return 1;
size_t init_result = ZSTD_initDStream(dstream);
if (ZSTD_isError(init_result)) {
ZSTD_freeDStream(dstream);
return 1;
}
size_t in_size = ZSTD_DStreamInSize();
size_t out_size = ZSTD_DStreamOutSize();
void *in_buf = malloc(in_size);
void *out_buf = malloc(out_size);
if (!in_buf || !out_buf) {
free(in_buf);
free(out_buf);
ZSTD_freeDStream(dstream);
return 1;
}
int rc = 0;
ssize_t bytes_read;
while ((bytes_read = read(src_fd, in_buf, in_size)) > 0) {
ZSTD_inBuffer input = { in_buf, (size_t)bytes_read, 0 };
while (input.pos < input.size) {
ZSTD_outBuffer output = { out_buf, out_size, 0 };
size_t result = ZSTD_decompressStream(dstream, &output, &input);
if (ZSTD_isError(result)) { rc = 1; goto done; }
if (output.pos > 0) {
ssize_t written = write(dst_fd, out_buf, output.pos);
if (written != (ssize_t)output.pos) { rc = 1; goto done; }
}
}
}
if (bytes_read < 0) rc = 1;
done:
free(in_buf);
free(out_buf);
ZSTD_freeDStream(dstream);
return rc;
}
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,12 @@
//
#pragma once
#include "archive.h"
#include <stdint.h>
void archive_set_error_wrapper(struct archive *a, int error_number, const char *error_string);
/// Decompress a zstd-compressed file at \p src_fd into \p dst_fd.
/// Returns 0 on success, or a non-zero error code on failure.
int zstd_decompress_fd(int src_fd, int dst_fd);
@@ -0,0 +1,731 @@
/*-
* Copyright (c) 2003-2008 Tim Kientzle
* Copyright (c) 2016 Martin Matuska
* All rights reserved.
*
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the following conditions
* are met:
* 1. Redistributions of source code must retain the above copyright
* notice, this list of conditions and the following disclaimer.
* 2. Redistributions in binary form must reproduce the above copyright
* notice, this list of conditions and the following disclaimer in the
* documentation and/or other materials provided with the distribution.
*
* THIS SOFTWARE IS PROVIDED BY THE AUTHOR(S) ``AS IS'' AND ANY EXPRESS OR
* IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES
* OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED.
* IN NO EVENT SHALL THE AUTHOR(S) BE LIABLE FOR ANY DIRECT, INDIRECT,
* INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT
* NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
* DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
* THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF
* THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
*/
#ifndef ARCHIVE_ENTRY_H_INCLUDED
#define ARCHIVE_ENTRY_H_INCLUDED
/* Note: Compiler will complain if this does not match archive.h! */
#define ARCHIVE_VERSION_NUMBER 3007007
/*
* Note: archive_entry.h is for use outside of libarchive; the
* configuration headers (config.h, archive_platform.h, etc.) are
* purely internal. Do NOT use HAVE_XXX configuration macros to
* control the behavior of this header! If you must conditionalize,
* use predefined compiler and/or platform macros.
*/
#include <sys/types.h>
#include <stddef.h> /* for wchar_t */
#include <stdint.h>
#include <time.h>
#if defined(_WIN32) && !defined(__CYGWIN__)
#include <windows.h>
#endif
/* Get a suitable 64-bit integer type. */
#if !defined(__LA_INT64_T_DEFINED)
# if ARCHIVE_VERSION_NUMBER < 4000000
#define __LA_INT64_T la_int64_t
# endif
#define __LA_INT64_T_DEFINED
# if defined(_WIN32) && !defined(__CYGWIN__) && !defined(__WATCOMC__)
typedef __int64 la_int64_t;
# else
#include <unistd.h>
# if defined(_SCO_DS) || defined(__osf__)
typedef long long la_int64_t;
# else
typedef int64_t la_int64_t;
# endif
# endif
#endif
/* The la_ssize_t should match the type used in 'struct stat' */
#if !defined(__LA_SSIZE_T_DEFINED)
/* Older code relied on the __LA_SSIZE_T macro; after 4.0 we'll switch to the typedef exclusively. */
# if ARCHIVE_VERSION_NUMBER < 4000000
#define __LA_SSIZE_T la_ssize_t
# endif
#define __LA_SSIZE_T_DEFINED
# if defined(_WIN32) && !defined(__CYGWIN__) && !defined(__WATCOMC__)
# if defined(_SSIZE_T_DEFINED) || defined(_SSIZE_T_)
typedef ssize_t la_ssize_t;
# elif defined(_WIN64)
typedef __int64 la_ssize_t;
# else
typedef long la_ssize_t;
# endif
# else
# include <unistd.h> /* ssize_t */
typedef ssize_t la_ssize_t;
# endif
#endif
/* Get a suitable definition for mode_t */
#if ARCHIVE_VERSION_NUMBER >= 3999000
/* Switch to plain 'int' for libarchive 4.0. It's less broken than 'mode_t' */
# define __LA_MODE_T int
#elif defined(_WIN32) && !defined(__CYGWIN__) && !defined(__BORLANDC__) && !defined(__WATCOMC__)
# define __LA_MODE_T unsigned short
#else
# define __LA_MODE_T mode_t
#endif
/* Large file support for Android */
#if defined(__LIBARCHIVE_BUILD) && defined(__ANDROID__)
#include "android_lf.h"
#endif
/*
* On Windows, define LIBARCHIVE_STATIC if you're building or using a
* .lib. The default here assumes you're building a DLL. Only
* libarchive source should ever define __LIBARCHIVE_BUILD.
*/
#if ((defined __WIN32__) || (defined _WIN32) || defined(__CYGWIN__)) && (!defined LIBARCHIVE_STATIC)
# ifdef __LIBARCHIVE_BUILD
# ifdef __GNUC__
# define __LA_DECL __attribute__((dllexport)) extern
# else
# define __LA_DECL __declspec(dllexport)
# endif
# else
# ifdef __GNUC__
# define __LA_DECL
# else
# define __LA_DECL __declspec(dllimport)
# endif
# endif
#elif defined __LIBARCHIVE_ENABLE_VISIBILITY
# define __LA_DECL __attribute__((visibility("default")))
#else
/* Static libraries on all platforms and shared libraries on non-Windows. */
# define __LA_DECL
#endif
#if defined(__GNUC__) && __GNUC__ >= 3 && __GNUC_MINOR__ >= 1
# define __LA_DEPRECATED __attribute__((deprecated))
#else
# define __LA_DEPRECATED
#endif
#ifdef __cplusplus
extern "C" {
#endif
/*
* Description of an archive entry.
*
* You can think of this as "struct stat" with some text fields added in.
*
* TODO: Add "comment", "charset", and possibly other entries that are
* supported by "pax interchange" format. However, GNU, ustar, cpio,
* and other variants don't support these features, so they're not an
* excruciatingly high priority right now.
*
* TODO: "pax interchange" format allows essentially arbitrary
* key/value attributes to be attached to any entry. Supporting
* such extensions may make this library useful for special
* applications (e.g., a package manager could attach special
* package-management attributes to each entry).
*/
struct archive;
struct archive_entry;
/*
* File-type constants. These are returned from archive_entry_filetype()
* and passed to archive_entry_set_filetype().
*
* These values match S_XXX defines on every platform I've checked,
* including Windows, AIX, Linux, Solaris, and BSD. They're
* (re)defined here because platforms generally don't define the ones
* they don't support. For example, Windows doesn't define S_IFLNK or
* S_IFBLK. Instead of having a mass of conditional logic and system
* checks to define any S_XXX values that aren't supported locally,
* I've just defined a new set of such constants so that
* libarchive-based applications can manipulate and identify archive
* entries properly even if the hosting platform can't store them on
* disk.
*
* These values are also used directly within some portable formats,
* such as cpio. If you find a platform that varies from these, the
* correct solution is to leave these alone and translate from these
* portable values to platform-native values when entries are read from
* or written to disk.
*/
/*
* In libarchive 4.0, we can drop the casts here.
* They're needed to work around Borland C's broken mode_t.
*/
#define AE_IFMT ((__LA_MODE_T)0170000)
#define AE_IFREG ((__LA_MODE_T)0100000)
#define AE_IFLNK ((__LA_MODE_T)0120000)
#define AE_IFSOCK ((__LA_MODE_T)0140000)
#define AE_IFCHR ((__LA_MODE_T)0020000)
#define AE_IFBLK ((__LA_MODE_T)0060000)
#define AE_IFDIR ((__LA_MODE_T)0040000)
#define AE_IFIFO ((__LA_MODE_T)0010000)
/*
* Symlink types
*/
#define AE_SYMLINK_TYPE_UNDEFINED 0
#define AE_SYMLINK_TYPE_FILE 1
#define AE_SYMLINK_TYPE_DIRECTORY 2
/*
* Basic object manipulation
*/
__LA_DECL struct archive_entry *archive_entry_clear(struct archive_entry *);
/* The 'clone' function does a deep copy; all of the strings are copied too. */
__LA_DECL struct archive_entry *archive_entry_clone(struct archive_entry *);
__LA_DECL void archive_entry_free(struct archive_entry *);
__LA_DECL struct archive_entry *archive_entry_new(void);
/*
* This form of archive_entry_new2() will pull character-set
* conversion information from the specified archive handle. The
* older archive_entry_new(void) form is equivalent to calling
* archive_entry_new2(NULL) and will result in the use of an internal
* default character-set conversion.
*/
__LA_DECL struct archive_entry *archive_entry_new2(struct archive *);
/*
* Retrieve fields from an archive_entry.
*
* There are a number of implicit conversions among these fields. For
* example, if a regular string field is set and you read the _w wide
* character field, the entry will implicitly convert narrow-to-wide
* using the current locale. Similarly, dev values are automatically
* updated when you write devmajor or devminor and vice versa.
*
* In addition, fields can be "set" or "unset." Unset string fields
* return NULL, non-string fields have _is_set() functions to test
* whether they've been set. You can "unset" a string field by
* assigning NULL; non-string fields have _unset() functions to
* unset them.
*
* Note: There is one ambiguity in the above; string fields will
* also return NULL when implicit character set conversions fail.
* This is usually what you want.
*/
__LA_DECL time_t archive_entry_atime(struct archive_entry *);
__LA_DECL long archive_entry_atime_nsec(struct archive_entry *);
__LA_DECL int archive_entry_atime_is_set(struct archive_entry *);
__LA_DECL time_t archive_entry_birthtime(struct archive_entry *);
__LA_DECL long archive_entry_birthtime_nsec(struct archive_entry *);
__LA_DECL int archive_entry_birthtime_is_set(struct archive_entry *);
__LA_DECL time_t archive_entry_ctime(struct archive_entry *);
__LA_DECL long archive_entry_ctime_nsec(struct archive_entry *);
__LA_DECL int archive_entry_ctime_is_set(struct archive_entry *);
__LA_DECL dev_t archive_entry_dev(struct archive_entry *);
__LA_DECL int archive_entry_dev_is_set(struct archive_entry *);
__LA_DECL dev_t archive_entry_devmajor(struct archive_entry *);
__LA_DECL dev_t archive_entry_devminor(struct archive_entry *);
__LA_DECL __LA_MODE_T archive_entry_filetype(struct archive_entry *);
__LA_DECL int archive_entry_filetype_is_set(struct archive_entry *);
__LA_DECL void archive_entry_fflags(struct archive_entry *,
unsigned long * /* set */,
unsigned long * /* clear */);
__LA_DECL const char *archive_entry_fflags_text(struct archive_entry *);
__LA_DECL la_int64_t archive_entry_gid(struct archive_entry *);
__LA_DECL int archive_entry_gid_is_set(struct archive_entry *);
__LA_DECL const char *archive_entry_gname(struct archive_entry *);
__LA_DECL const char *archive_entry_gname_utf8(struct archive_entry *);
__LA_DECL const wchar_t *archive_entry_gname_w(struct archive_entry *);
__LA_DECL void archive_entry_set_link_to_hardlink(struct archive_entry *);
__LA_DECL const char *archive_entry_hardlink(struct archive_entry *);
__LA_DECL const char *archive_entry_hardlink_utf8(struct archive_entry *);
__LA_DECL const wchar_t *archive_entry_hardlink_w(struct archive_entry *);
__LA_DECL int archive_entry_hardlink_is_set(struct archive_entry *);
__LA_DECL la_int64_t archive_entry_ino(struct archive_entry *);
__LA_DECL la_int64_t archive_entry_ino64(struct archive_entry *);
__LA_DECL int archive_entry_ino_is_set(struct archive_entry *);
__LA_DECL __LA_MODE_T archive_entry_mode(struct archive_entry *);
__LA_DECL time_t archive_entry_mtime(struct archive_entry *);
__LA_DECL long archive_entry_mtime_nsec(struct archive_entry *);
__LA_DECL int archive_entry_mtime_is_set(struct archive_entry *);
__LA_DECL unsigned int archive_entry_nlink(struct archive_entry *);
__LA_DECL const char *archive_entry_pathname(struct archive_entry *);
__LA_DECL const char *archive_entry_pathname_utf8(struct archive_entry *);
__LA_DECL const wchar_t *archive_entry_pathname_w(struct archive_entry *);
__LA_DECL __LA_MODE_T archive_entry_perm(struct archive_entry *);
__LA_DECL int archive_entry_perm_is_set(struct archive_entry *);
__LA_DECL int archive_entry_rdev_is_set(struct archive_entry *);
__LA_DECL dev_t archive_entry_rdev(struct archive_entry *);
__LA_DECL dev_t archive_entry_rdevmajor(struct archive_entry *);
__LA_DECL dev_t archive_entry_rdevminor(struct archive_entry *);
__LA_DECL const char *archive_entry_sourcepath(struct archive_entry *);
__LA_DECL const wchar_t *archive_entry_sourcepath_w(struct archive_entry *);
__LA_DECL la_int64_t archive_entry_size(struct archive_entry *);
__LA_DECL int archive_entry_size_is_set(struct archive_entry *);
__LA_DECL const char *archive_entry_strmode(struct archive_entry *);
__LA_DECL void archive_entry_set_link_to_symlink(struct archive_entry *);
__LA_DECL const char *archive_entry_symlink(struct archive_entry *);
__LA_DECL const char *archive_entry_symlink_utf8(struct archive_entry *);
__LA_DECL int archive_entry_symlink_type(struct archive_entry *);
__LA_DECL const wchar_t *archive_entry_symlink_w(struct archive_entry *);
__LA_DECL la_int64_t archive_entry_uid(struct archive_entry *);
__LA_DECL int archive_entry_uid_is_set(struct archive_entry *);
__LA_DECL const char *archive_entry_uname(struct archive_entry *);
__LA_DECL const char *archive_entry_uname_utf8(struct archive_entry *);
__LA_DECL const wchar_t *archive_entry_uname_w(struct archive_entry *);
__LA_DECL int archive_entry_is_data_encrypted(struct archive_entry *);
__LA_DECL int archive_entry_is_metadata_encrypted(struct archive_entry *);
__LA_DECL int archive_entry_is_encrypted(struct archive_entry *);
/*
* Set fields in an archive_entry.
*
* Note: Before libarchive 2.4, there were 'set' and 'copy' versions
* of the string setters. 'copy' copied the actual string, 'set' just
* stored the pointer. In libarchive 2.4 and later, strings are
* always copied.
*/
__LA_DECL void archive_entry_set_atime(struct archive_entry *, time_t, long);
__LA_DECL void archive_entry_unset_atime(struct archive_entry *);
#if defined(_WIN32) && !defined(__CYGWIN__)
__LA_DECL void archive_entry_copy_bhfi(struct archive_entry *, BY_HANDLE_FILE_INFORMATION *);
#endif
__LA_DECL void archive_entry_set_birthtime(struct archive_entry *, time_t, long);
__LA_DECL void archive_entry_unset_birthtime(struct archive_entry *);
__LA_DECL void archive_entry_set_ctime(struct archive_entry *, time_t, long);
__LA_DECL void archive_entry_unset_ctime(struct archive_entry *);
__LA_DECL void archive_entry_set_dev(struct archive_entry *, dev_t);
__LA_DECL void archive_entry_set_devmajor(struct archive_entry *, dev_t);
__LA_DECL void archive_entry_set_devminor(struct archive_entry *, dev_t);
__LA_DECL void archive_entry_set_filetype(struct archive_entry *, unsigned int);
__LA_DECL void archive_entry_set_fflags(struct archive_entry *,
unsigned long /* set */, unsigned long /* clear */);
/* Returns pointer to start of first invalid token, or NULL if none. */
/* Note that all recognized tokens are processed, regardless. */
__LA_DECL const char *archive_entry_copy_fflags_text(struct archive_entry *,
const char *);
__LA_DECL const char *archive_entry_copy_fflags_text_len(struct archive_entry *,
const char *, size_t);
__LA_DECL const wchar_t *archive_entry_copy_fflags_text_w(struct archive_entry *,
const wchar_t *);
__LA_DECL void archive_entry_set_gid(struct archive_entry *, la_int64_t);
__LA_DECL void archive_entry_set_gname(struct archive_entry *, const char *);
__LA_DECL void archive_entry_set_gname_utf8(struct archive_entry *, const char *);
__LA_DECL void archive_entry_copy_gname(struct archive_entry *, const char *);
__LA_DECL void archive_entry_copy_gname_w(struct archive_entry *, const wchar_t *);
__LA_DECL int archive_entry_update_gname_utf8(struct archive_entry *, const char *);
__LA_DECL void archive_entry_set_hardlink(struct archive_entry *, const char *);
__LA_DECL void archive_entry_set_hardlink_utf8(struct archive_entry *, const char *);
__LA_DECL void archive_entry_copy_hardlink(struct archive_entry *, const char *);
__LA_DECL void archive_entry_copy_hardlink_w(struct archive_entry *, const wchar_t *);
__LA_DECL int archive_entry_update_hardlink_utf8(struct archive_entry *, const char *);
__LA_DECL void archive_entry_set_ino(struct archive_entry *, la_int64_t);
__LA_DECL void archive_entry_set_ino64(struct archive_entry *, la_int64_t);
__LA_DECL void archive_entry_set_link(struct archive_entry *, const char *);
__LA_DECL void archive_entry_set_link_utf8(struct archive_entry *, const char *);
__LA_DECL void archive_entry_copy_link(struct archive_entry *, const char *);
__LA_DECL void archive_entry_copy_link_w(struct archive_entry *, const wchar_t *);
__LA_DECL int archive_entry_update_link_utf8(struct archive_entry *, const char *);
__LA_DECL void archive_entry_set_mode(struct archive_entry *, __LA_MODE_T);
__LA_DECL void archive_entry_set_mtime(struct archive_entry *, time_t, long);
__LA_DECL void archive_entry_unset_mtime(struct archive_entry *);
__LA_DECL void archive_entry_set_nlink(struct archive_entry *, unsigned int);
__LA_DECL void archive_entry_set_pathname(struct archive_entry *, const char *);
__LA_DECL void archive_entry_set_pathname_utf8(struct archive_entry *, const char *);
__LA_DECL void archive_entry_copy_pathname(struct archive_entry *, const char *);
__LA_DECL void archive_entry_copy_pathname_w(struct archive_entry *, const wchar_t *);
__LA_DECL int archive_entry_update_pathname_utf8(struct archive_entry *, const char *);
__LA_DECL void archive_entry_set_perm(struct archive_entry *, __LA_MODE_T);
__LA_DECL void archive_entry_set_rdev(struct archive_entry *, dev_t);
__LA_DECL void archive_entry_set_rdevmajor(struct archive_entry *, dev_t);
__LA_DECL void archive_entry_set_rdevminor(struct archive_entry *, dev_t);
__LA_DECL void archive_entry_set_size(struct archive_entry *, la_int64_t);
__LA_DECL void archive_entry_unset_size(struct archive_entry *);
__LA_DECL void archive_entry_copy_sourcepath(struct archive_entry *, const char *);
__LA_DECL void archive_entry_copy_sourcepath_w(struct archive_entry *, const wchar_t *);
__LA_DECL void archive_entry_set_symlink(struct archive_entry *, const char *);
__LA_DECL void archive_entry_set_symlink_type(struct archive_entry *, int);
__LA_DECL void archive_entry_set_symlink_utf8(struct archive_entry *, const char *);
__LA_DECL void archive_entry_copy_symlink(struct archive_entry *, const char *);
__LA_DECL void archive_entry_copy_symlink_w(struct archive_entry *, const wchar_t *);
__LA_DECL int archive_entry_update_symlink_utf8(struct archive_entry *, const char *);
__LA_DECL void archive_entry_set_uid(struct archive_entry *, la_int64_t);
__LA_DECL void archive_entry_set_uname(struct archive_entry *, const char *);
__LA_DECL void archive_entry_set_uname_utf8(struct archive_entry *, const char *);
__LA_DECL void archive_entry_copy_uname(struct archive_entry *, const char *);
__LA_DECL void archive_entry_copy_uname_w(struct archive_entry *, const wchar_t *);
__LA_DECL int archive_entry_update_uname_utf8(struct archive_entry *, const char *);
__LA_DECL void archive_entry_set_is_data_encrypted(struct archive_entry *, char is_encrypted);
__LA_DECL void archive_entry_set_is_metadata_encrypted(struct archive_entry *, char is_encrypted);
/*
* Routines to bulk copy fields to/from a platform-native "struct
* stat." Libarchive used to just store a struct stat inside of each
* archive_entry object, but this created issues when trying to
* manipulate archives on systems different than the ones they were
* created on.
*
* TODO: On Linux and other LFS systems, provide both stat32 and
* stat64 versions of these functions and all of the macro glue so
* that archive_entry_stat is magically defined to
* archive_entry_stat32 or archive_entry_stat64 as appropriate.
*/
__LA_DECL const struct stat *archive_entry_stat(struct archive_entry *);
__LA_DECL void archive_entry_copy_stat(struct archive_entry *, const struct stat *);
/*
* Storage for Mac OS-specific AppleDouble metadata information.
* Apple-format tar files store a separate binary blob containing
* encoded metadata with ACL, extended attributes, etc.
* This provides a place to store that blob.
*/
__LA_DECL const void * archive_entry_mac_metadata(struct archive_entry *, size_t *);
__LA_DECL void archive_entry_copy_mac_metadata(struct archive_entry *, const void *, size_t);
/*
* Digest routine. This is used to query the raw hex digest for the
* given entry. The type of digest is provided as an argument.
*/
#define ARCHIVE_ENTRY_DIGEST_MD5 0x00000001
#define ARCHIVE_ENTRY_DIGEST_RMD160 0x00000002
#define ARCHIVE_ENTRY_DIGEST_SHA1 0x00000003
#define ARCHIVE_ENTRY_DIGEST_SHA256 0x00000004
#define ARCHIVE_ENTRY_DIGEST_SHA384 0x00000005
#define ARCHIVE_ENTRY_DIGEST_SHA512 0x00000006
__LA_DECL const unsigned char * archive_entry_digest(struct archive_entry *, int /* type */);
/*
* ACL routines. This used to simply store and return text-format ACL
* strings, but that proved insufficient for a number of reasons:
* = clients need control over uname/uid and gname/gid mappings
* = there are many different ACL text formats
* = would like to be able to read/convert archives containing ACLs
* on platforms that lack ACL libraries
*
* This last point, in particular, forces me to implement a reasonably
* complete set of ACL support routines.
*/
/*
* Permission bits.
*/
#define ARCHIVE_ENTRY_ACL_EXECUTE 0x00000001
#define ARCHIVE_ENTRY_ACL_WRITE 0x00000002
#define ARCHIVE_ENTRY_ACL_READ 0x00000004
#define ARCHIVE_ENTRY_ACL_READ_DATA 0x00000008
#define ARCHIVE_ENTRY_ACL_LIST_DIRECTORY 0x00000008
#define ARCHIVE_ENTRY_ACL_WRITE_DATA 0x00000010
#define ARCHIVE_ENTRY_ACL_ADD_FILE 0x00000010
#define ARCHIVE_ENTRY_ACL_APPEND_DATA 0x00000020
#define ARCHIVE_ENTRY_ACL_ADD_SUBDIRECTORY 0x00000020
#define ARCHIVE_ENTRY_ACL_READ_NAMED_ATTRS 0x00000040
#define ARCHIVE_ENTRY_ACL_WRITE_NAMED_ATTRS 0x00000080
#define ARCHIVE_ENTRY_ACL_DELETE_CHILD 0x00000100
#define ARCHIVE_ENTRY_ACL_READ_ATTRIBUTES 0x00000200
#define ARCHIVE_ENTRY_ACL_WRITE_ATTRIBUTES 0x00000400
#define ARCHIVE_ENTRY_ACL_DELETE 0x00000800
#define ARCHIVE_ENTRY_ACL_READ_ACL 0x00001000
#define ARCHIVE_ENTRY_ACL_WRITE_ACL 0x00002000
#define ARCHIVE_ENTRY_ACL_WRITE_OWNER 0x00004000
#define ARCHIVE_ENTRY_ACL_SYNCHRONIZE 0x00008000
#define ARCHIVE_ENTRY_ACL_PERMS_POSIX1E \
(ARCHIVE_ENTRY_ACL_EXECUTE \
| ARCHIVE_ENTRY_ACL_WRITE \
| ARCHIVE_ENTRY_ACL_READ)
#define ARCHIVE_ENTRY_ACL_PERMS_NFS4 \
(ARCHIVE_ENTRY_ACL_EXECUTE \
| ARCHIVE_ENTRY_ACL_READ_DATA \
| ARCHIVE_ENTRY_ACL_LIST_DIRECTORY \
| ARCHIVE_ENTRY_ACL_WRITE_DATA \
| ARCHIVE_ENTRY_ACL_ADD_FILE \
| ARCHIVE_ENTRY_ACL_APPEND_DATA \
| ARCHIVE_ENTRY_ACL_ADD_SUBDIRECTORY \
| ARCHIVE_ENTRY_ACL_READ_NAMED_ATTRS \
| ARCHIVE_ENTRY_ACL_WRITE_NAMED_ATTRS \
| ARCHIVE_ENTRY_ACL_DELETE_CHILD \
| ARCHIVE_ENTRY_ACL_READ_ATTRIBUTES \
| ARCHIVE_ENTRY_ACL_WRITE_ATTRIBUTES \
| ARCHIVE_ENTRY_ACL_DELETE \
| ARCHIVE_ENTRY_ACL_READ_ACL \
| ARCHIVE_ENTRY_ACL_WRITE_ACL \
| ARCHIVE_ENTRY_ACL_WRITE_OWNER \
| ARCHIVE_ENTRY_ACL_SYNCHRONIZE)
/*
* Inheritance values (NFS4 ACLs only); included in permset.
*/
#define ARCHIVE_ENTRY_ACL_ENTRY_INHERITED 0x01000000
#define ARCHIVE_ENTRY_ACL_ENTRY_FILE_INHERIT 0x02000000
#define ARCHIVE_ENTRY_ACL_ENTRY_DIRECTORY_INHERIT 0x04000000
#define ARCHIVE_ENTRY_ACL_ENTRY_NO_PROPAGATE_INHERIT 0x08000000
#define ARCHIVE_ENTRY_ACL_ENTRY_INHERIT_ONLY 0x10000000
#define ARCHIVE_ENTRY_ACL_ENTRY_SUCCESSFUL_ACCESS 0x20000000
#define ARCHIVE_ENTRY_ACL_ENTRY_FAILED_ACCESS 0x40000000
#define ARCHIVE_ENTRY_ACL_INHERITANCE_NFS4 \
(ARCHIVE_ENTRY_ACL_ENTRY_FILE_INHERIT \
| ARCHIVE_ENTRY_ACL_ENTRY_DIRECTORY_INHERIT \
| ARCHIVE_ENTRY_ACL_ENTRY_NO_PROPAGATE_INHERIT \
| ARCHIVE_ENTRY_ACL_ENTRY_INHERIT_ONLY \
| ARCHIVE_ENTRY_ACL_ENTRY_SUCCESSFUL_ACCESS \
| ARCHIVE_ENTRY_ACL_ENTRY_FAILED_ACCESS \
| ARCHIVE_ENTRY_ACL_ENTRY_INHERITED)
/* We need to be able to specify combinations of these. */
#define ARCHIVE_ENTRY_ACL_TYPE_ACCESS 0x00000100 /* POSIX.1e only */
#define ARCHIVE_ENTRY_ACL_TYPE_DEFAULT 0x00000200 /* POSIX.1e only */
#define ARCHIVE_ENTRY_ACL_TYPE_ALLOW 0x00000400 /* NFS4 only */
#define ARCHIVE_ENTRY_ACL_TYPE_DENY 0x00000800 /* NFS4 only */
#define ARCHIVE_ENTRY_ACL_TYPE_AUDIT 0x00001000 /* NFS4 only */
#define ARCHIVE_ENTRY_ACL_TYPE_ALARM 0x00002000 /* NFS4 only */
#define ARCHIVE_ENTRY_ACL_TYPE_POSIX1E (ARCHIVE_ENTRY_ACL_TYPE_ACCESS \
| ARCHIVE_ENTRY_ACL_TYPE_DEFAULT)
#define ARCHIVE_ENTRY_ACL_TYPE_NFS4 (ARCHIVE_ENTRY_ACL_TYPE_ALLOW \
| ARCHIVE_ENTRY_ACL_TYPE_DENY \
| ARCHIVE_ENTRY_ACL_TYPE_AUDIT \
| ARCHIVE_ENTRY_ACL_TYPE_ALARM)
/* Tag values mimic POSIX.1e */
#define ARCHIVE_ENTRY_ACL_USER 10001 /* Specified user. */
#define ARCHIVE_ENTRY_ACL_USER_OBJ 10002 /* User who owns the file. */
#define ARCHIVE_ENTRY_ACL_GROUP 10003 /* Specified group. */
#define ARCHIVE_ENTRY_ACL_GROUP_OBJ 10004 /* Group who owns the file. */
#define ARCHIVE_ENTRY_ACL_MASK 10005 /* Modify group access (POSIX.1e only) */
#define ARCHIVE_ENTRY_ACL_OTHER 10006 /* Public (POSIX.1e only) */
#define ARCHIVE_ENTRY_ACL_EVERYONE 10107 /* Everyone (NFS4 only) */
/*
* Set the ACL by clearing it and adding entries one at a time.
* Unlike the POSIX.1e ACL routines, you must specify the type
* (access/default) for each entry. Internally, the ACL data is just
* a soup of entries. API calls here allow you to retrieve just the
* entries of interest. This design (which goes against the spirit of
* POSIX.1e) is useful for handling archive formats that combine
* default and access information in a single ACL list.
*/
__LA_DECL void archive_entry_acl_clear(struct archive_entry *);
__LA_DECL int archive_entry_acl_add_entry(struct archive_entry *,
int /* type */, int /* permset */, int /* tag */,
int /* qual */, const char * /* name */);
__LA_DECL int archive_entry_acl_add_entry_w(struct archive_entry *,
int /* type */, int /* permset */, int /* tag */,
int /* qual */, const wchar_t * /* name */);
/*
* To retrieve the ACL, first "reset", then repeatedly ask for the
* "next" entry. The want_type parameter allows you to request only
* certain types of entries.
*/
__LA_DECL int archive_entry_acl_reset(struct archive_entry *, int /* want_type */);
__LA_DECL int archive_entry_acl_next(struct archive_entry *, int /* want_type */,
int * /* type */, int * /* permset */, int * /* tag */,
int * /* qual */, const char ** /* name */);
/*
* Construct a text-format ACL. The flags argument is a bitmask that
* can include any of the following:
*
* Flags only for archive entries with POSIX.1e ACL:
* ARCHIVE_ENTRY_ACL_TYPE_ACCESS - Include POSIX.1e "access" entries.
* ARCHIVE_ENTRY_ACL_TYPE_DEFAULT - Include POSIX.1e "default" entries.
* ARCHIVE_ENTRY_ACL_STYLE_MARK_DEFAULT - Include "default:" before each
* default ACL entry.
* ARCHIVE_ENTRY_ACL_STYLE_SOLARIS - Output only one colon after "other" and
* "mask" entries.
*
* Flags only for archive entries with NFSv4 ACL:
* ARCHIVE_ENTRY_ACL_STYLE_COMPACT - Do not output the minus character for
* unset permissions and flags in NFSv4 ACL permission and flag fields
*
* Flags for for archive entries with POSIX.1e ACL or NFSv4 ACL:
* ARCHIVE_ENTRY_ACL_STYLE_EXTRA_ID - Include extra numeric ID field in
* each ACL entry.
* ARCHIVE_ENTRY_ACL_STYLE_SEPARATOR_COMMA - Separate entries with comma
* instead of newline.
*/
#define ARCHIVE_ENTRY_ACL_STYLE_EXTRA_ID 0x00000001
#define ARCHIVE_ENTRY_ACL_STYLE_MARK_DEFAULT 0x00000002
#define ARCHIVE_ENTRY_ACL_STYLE_SOLARIS 0x00000004
#define ARCHIVE_ENTRY_ACL_STYLE_SEPARATOR_COMMA 0x00000008
#define ARCHIVE_ENTRY_ACL_STYLE_COMPACT 0x00000010
__LA_DECL wchar_t *archive_entry_acl_to_text_w(struct archive_entry *,
la_ssize_t * /* len */, int /* flags */);
__LA_DECL char *archive_entry_acl_to_text(struct archive_entry *,
la_ssize_t * /* len */, int /* flags */);
__LA_DECL int archive_entry_acl_from_text_w(struct archive_entry *,
const wchar_t * /* wtext */, int /* type */);
__LA_DECL int archive_entry_acl_from_text(struct archive_entry *,
const char * /* text */, int /* type */);
/* Deprecated constants */
#define OLD_ARCHIVE_ENTRY_ACL_STYLE_EXTRA_ID 1024
#define OLD_ARCHIVE_ENTRY_ACL_STYLE_MARK_DEFAULT 2048
/* Deprecated functions */
__LA_DECL const wchar_t *archive_entry_acl_text_w(struct archive_entry *,
int /* flags */) __LA_DEPRECATED;
__LA_DECL const char *archive_entry_acl_text(struct archive_entry *,
int /* flags */) __LA_DEPRECATED;
/* Return bitmask of ACL types in an archive entry */
__LA_DECL int archive_entry_acl_types(struct archive_entry *);
/* Return a count of entries matching 'want_type' */
__LA_DECL int archive_entry_acl_count(struct archive_entry *, int /* want_type */);
/* Return an opaque ACL object. */
/* There's not yet anything clients can actually do with this... */
struct archive_acl;
__LA_DECL struct archive_acl *archive_entry_acl(struct archive_entry *);
/*
* extended attributes
*/
__LA_DECL void archive_entry_xattr_clear(struct archive_entry *);
__LA_DECL void archive_entry_xattr_add_entry(struct archive_entry *,
const char * /* name */, const void * /* value */,
size_t /* size */);
/*
* To retrieve the xattr list, first "reset", then repeatedly ask for the
* "next" entry.
*/
__LA_DECL int archive_entry_xattr_count(struct archive_entry *);
__LA_DECL int archive_entry_xattr_reset(struct archive_entry *);
__LA_DECL int archive_entry_xattr_next(struct archive_entry *,
const char ** /* name */, const void ** /* value */, size_t *);
/*
* sparse
*/
__LA_DECL void archive_entry_sparse_clear(struct archive_entry *);
__LA_DECL void archive_entry_sparse_add_entry(struct archive_entry *,
la_int64_t /* offset */, la_int64_t /* length */);
/*
* To retrieve the xattr list, first "reset", then repeatedly ask for the
* "next" entry.
*/
__LA_DECL int archive_entry_sparse_count(struct archive_entry *);
__LA_DECL int archive_entry_sparse_reset(struct archive_entry *);
__LA_DECL int archive_entry_sparse_next(struct archive_entry *,
la_int64_t * /* offset */, la_int64_t * /* length */);
/*
* Utility to match up hardlinks.
*
* The 'struct archive_entry_linkresolver' is a cache of archive entries
* for files with multiple links. Here's how to use it:
* 1. Create a lookup object with archive_entry_linkresolver_new()
* 2. Tell it the archive format you're using.
* 3. Hand each archive_entry to archive_entry_linkify().
* That function will return 0, 1, or 2 entries that should
* be written.
* 4. Call archive_entry_linkify(resolver, NULL) until
* no more entries are returned.
* 5. Call archive_entry_linkresolver_free(resolver) to free resources.
*
* The entries returned have their hardlink and size fields updated
* appropriately. If an entry is passed in that does not refer to
* a file with multiple links, it is returned unchanged. The intention
* is that you should be able to simply filter all entries through
* this machine.
*
* To make things more efficient, be sure that each entry has a valid
* nlinks value. The hardlink cache uses this to track when all links
* have been found. If the nlinks value is zero, it will keep every
* name in the cache indefinitely, which can use a lot of memory.
*
* Note that archive_entry_size() is reset to zero if the file
* body should not be written to the archive. Pay attention!
*/
struct archive_entry_linkresolver;
/*
* There are three different strategies for marking hardlinks.
* The descriptions below name them after the best-known
* formats that rely on each strategy:
*
* "Old cpio" is the simplest, it always returns any entry unmodified.
* As far as I know, only cpio formats use this. Old cpio archives
* store every link with the full body; the onus is on the dearchiver
* to detect and properly link the files as they are restored.
* "tar" is also pretty simple; it caches a copy the first time it sees
* any link. Subsequent appearances are modified to be hardlink
* references to the first one without any body. Used by all tar
* formats, although the newest tar formats permit the "old cpio" strategy
* as well. This strategy is very simple for the dearchiver,
* and reasonably straightforward for the archiver.
* "new cpio" is trickier. It stores the body only with the last
* occurrence. The complication is that we might not
* see every link to a particular file in a single session, so
* there's no easy way to know when we've seen the last occurrence.
* The solution here is to queue one link until we see the next.
* At the end of the session, you can enumerate any remaining
* entries by calling archive_entry_linkify(NULL) and store those
* bodies. If you have a file with three links l1, l2, and l3,
* you'll get the following behavior if you see all three links:
* linkify(l1) => NULL (the resolver stores l1 internally)
* linkify(l2) => l1 (resolver stores l2, you write l1)
* linkify(l3) => l2, l3 (all links seen, you can write both).
* If you only see l1 and l2, you'll get this behavior:
* linkify(l1) => NULL
* linkify(l2) => l1
* linkify(NULL) => l2 (at end, you retrieve remaining links)
* As the name suggests, this strategy is used by newer cpio variants.
* It's noticeably more complex for the archiver, slightly more complex
* for the dearchiver than the tar strategy, but makes it straightforward
* to restore a file using any link by simply continuing to scan until
* you see a link that is stored with a body. In contrast, the tar
* strategy requires you to rescan the archive from the beginning to
* correctly extract an arbitrary link.
*/
__LA_DECL struct archive_entry_linkresolver *archive_entry_linkresolver_new(void);
__LA_DECL void archive_entry_linkresolver_set_strategy(
struct archive_entry_linkresolver *, int /* format_code */);
__LA_DECL void archive_entry_linkresolver_free(struct archive_entry_linkresolver *);
__LA_DECL void archive_entry_linkify(struct archive_entry_linkresolver *,
struct archive_entry **, struct archive_entry **);
__LA_DECL struct archive_entry *archive_entry_partial_links(
struct archive_entry_linkresolver *res, unsigned int *links);
#ifdef __cplusplus
}
#endif
/* This is meaningless outside of this header. */
#undef __LA_DECL
#endif /* !ARCHIVE_ENTRY_H_INCLUDED */
@@ -0,0 +1,34 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationExtras
import Foundation
internal func createTemporaryDirectory(baseName: String) -> URL? {
let url = FileManager.default.uniqueTemporaryDirectory().appendingPathComponent(
"\(baseName).XXXXXX")
var path = url.absoluteURL.path
return path.withUTF8 { utf8Bytes in
var mutablePath = Array(utf8Bytes) + [0]
return mutablePath.withUnsafeMutableBufferPointer { buffer -> URL? in
guard let baseAddress = buffer.baseAddress else { return nil }
mkdtemp(baseAddress)
let resultPath = String(decoding: buffer[..<(buffer.count - 1)], as: UTF8.self)
return URL(fileURLWithPath: resultPath, isDirectory: true)
}
}
}
@@ -0,0 +1,318 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import CArchive
import Foundation
/// Represents a single entry (e.g., a file, directory, symbolic link)
/// that is to be read/written into an archive.
public final class WriteEntry {
let underlying: OpaquePointer
public init(_ archive: ArchiveWriter) {
underlying = archive_entry_new2(archive.underlying)
}
public init() {
underlying = archive_entry_new()
}
deinit {
archive_entry_free(underlying)
}
}
extension WriteEntry {
/// The size of the entry in bytes.
public var size: Int64? {
get {
guard archive_entry_size_is_set(underlying) != 0 else { return nil }
return archive_entry_size(underlying)
}
set {
if let s = newValue {
archive_entry_set_size(underlying, s)
} else {
archive_entry_unset_size(underlying)
}
}
}
/// The mode of the entry.
public var permissions: mode_t {
get {
archive_entry_perm(underlying)
}
set {
archive_entry_set_perm(underlying, newValue)
}
}
/// The owner id of the entry.
public var owner: uid_t? {
get {
uid_t(exactly: archive_entry_uid(underlying))
}
set {
archive_entry_set_uid(underlying, Int64(newValue ?? 0))
}
}
/// The group id of the entry
public var group: gid_t? {
get {
gid_t(exactly: archive_entry_gid(underlying))
}
set {
archive_entry_set_gid(underlying, Int64(newValue ?? 0))
}
}
/// The path of file this entry hardlinks to
public var hardlink: String? {
get {
guard let cstr = archive_entry_hardlink(underlying) else {
return nil
}
return String(cString: cstr)
}
set {
guard let newValue else {
archive_entry_set_hardlink(underlying, nil)
return
}
newValue.withCString {
archive_entry_set_hardlink(underlying, $0)
}
}
}
/// The UTF-8 encoded path of file this entry hardlinks to
public var hardlinkUtf8: String? {
get {
guard let cstr = archive_entry_hardlink_utf8(underlying) else {
return nil
}
return String(cString: cstr, encoding: .utf8)
}
set {
guard let newValue else {
archive_entry_set_hardlink_utf8(underlying, nil)
return
}
newValue.withCString {
archive_entry_set_hardlink_utf8(underlying, $0)
}
}
}
/// The string representation of the permissions of the entry
public var strmode: String? {
if let cstr = archive_entry_strmode(underlying) {
return String(cString: cstr)
}
return nil
}
/// The type of file this entry represents.
public var fileType: URLFileResourceType {
get {
switch archive_entry_filetype(underlying) {
case S_IFIFO: return .namedPipe
case S_IFCHR: return .characterSpecial
case S_IFDIR: return .directory
case S_IFBLK: return .blockSpecial
case S_IFREG: return .regular
case S_IFLNK: return .symbolicLink
case S_IFSOCK: return .socket
default: return .unknown
}
}
set {
switch newValue {
case .namedPipe: archive_entry_set_filetype(underlying, UInt32(S_IFIFO as mode_t))
case .characterSpecial: archive_entry_set_filetype(underlying, UInt32(S_IFCHR as mode_t))
case .directory: archive_entry_set_filetype(underlying, UInt32(S_IFDIR as mode_t))
case .blockSpecial: archive_entry_set_filetype(underlying, UInt32(S_IFBLK as mode_t))
case .regular: archive_entry_set_filetype(underlying, UInt32(S_IFREG as mode_t))
case .symbolicLink: archive_entry_set_filetype(underlying, UInt32(S_IFLNK as mode_t))
case .socket: archive_entry_set_filetype(underlying, UInt32(S_IFSOCK as mode_t))
default: archive_entry_set_filetype(underlying, 0)
}
}
}
/// The date that the entry was last accessed
public var contentAccessDate: Date? {
get {
Date(
underlying,
archive_entry_atime_is_set,
archive_entry_atime,
archive_entry_atime_nsec)
}
set {
setDate(
newValue,
underlying, archive_entry_set_atime,
archive_entry_unset_atime)
}
}
/// The date that the entry was created
public var creationDate: Date? {
get {
Date(
underlying,
archive_entry_ctime_is_set,
archive_entry_ctime,
archive_entry_ctime_nsec)
}
set {
setDate(
newValue,
underlying, archive_entry_set_ctime,
archive_entry_unset_ctime)
}
}
/// The date that the entry was modified
public var modificationDate: Date? {
get {
Date(
underlying,
archive_entry_mtime_is_set,
archive_entry_mtime,
archive_entry_mtime_nsec)
}
set {
setDate(
newValue,
underlying, archive_entry_set_mtime,
archive_entry_unset_mtime)
}
}
/// The file path of the entry
public var path: String? {
get {
guard let pathname = archive_entry_pathname(underlying) else {
return nil
}
return String(cString: pathname)
}
set {
guard let newValue else {
archive_entry_set_pathname(underlying, nil)
return
}
newValue.withCString {
archive_entry_set_pathname(underlying, $0)
}
}
}
/// The UTF-8 encoded file path of the entry
public var pathUtf8: String? {
get {
guard let pathname = archive_entry_pathname_utf8(underlying) else {
return nil
}
return String(cString: pathname)
}
set {
guard let newValue else {
archive_entry_set_pathname_utf8(underlying, nil)
return
}
newValue.withCString {
archive_entry_set_pathname_utf8(underlying, $0)
}
}
}
/// The symlink target that the entry points to
public var symlinkTarget: String? {
get {
guard let target = archive_entry_symlink(underlying) else {
return nil
}
return String(cString: target)
}
set {
guard let newValue else {
archive_entry_set_symlink(underlying, nil)
return
}
newValue.withCString {
archive_entry_set_symlink(underlying, $0)
}
}
}
/// The extended attributes of the entry
public var xattrs: [String: Data] {
get {
archive_entry_xattr_reset(self.underlying)
var attrs: [String: Data] = [:]
var namePtr: UnsafePointer<CChar>?
var valuePtr: UnsafeRawPointer?
var size: Int = 0
while archive_entry_xattr_next(self.underlying, &namePtr, &valuePtr, &size) == 0 {
let _name = namePtr.map { String(cString: $0) }
let _value = valuePtr.map { Data(bytes: $0, count: size) }
guard let name = _name, let value = _value else {
continue
}
attrs[name] = value
}
return attrs
}
set {
archive_entry_xattr_clear(self.underlying)
for (key, value) in newValue {
value.withUnsafeBytes { ptr in
archive_entry_xattr_add_entry(self.underlying, key, ptr.baseAddress, [UInt8](value).count)
}
}
}
}
fileprivate func setDate(
_ date: Date?, _ underlying: OpaquePointer, _ setter: (OpaquePointer, time_t, CLong) -> Void,
_ unset: (OpaquePointer) -> Void
) {
if let d = date {
let ti = d.timeIntervalSince1970
let seconds = floor(ti)
let nsec = max(0, min(1_000_000_000, ti - seconds * 1_000_000_000))
setter(underlying, time_t(seconds), CLong(nsec))
} else {
unset(underlying)
}
}
}
extension Date {
init?(
_ underlying: OpaquePointer, _ isSet: (OpaquePointer) -> CInt, _ seconds: (OpaquePointer) -> time_t,
_ nsec: (OpaquePointer) -> CLong
) {
guard isSet(underlying) != 0 else { return nil }
let ti = TimeInterval(seconds(underlying)) + TimeInterval(nsec(underlying)) * 0.000_000_001
self.init(timeIntervalSince1970: ti)
}
}
@@ -0,0 +1,4 @@
# ``ContainerizationEXT4``
`ContainerizationEXT4` provides functionality to read the superblock of an existing ext4 block device and format a new block device with
the ext4 file system.
@@ -0,0 +1,97 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
extension EXT4.InodeFlag {
public static func | (lhs: Self, rhs: Self) -> Self {
Self(rawValue: lhs.rawValue | rhs.rawValue)
}
public static func | (lhs: Self, rhs: Self) -> UInt32 {
lhs.rawValue | rhs.rawValue
}
public static func | (lhs: Self, rhs: UInt32) -> UInt32 {
lhs.rawValue | rhs
}
}
extension EXT4.CompatFeature {
public static func | (lhs: Self, rhs: Self) -> Self {
EXT4.CompatFeature(rawValue: lhs.rawValue | rhs.rawValue)
}
public static func | (lhs: Self, rhs: Self) -> UInt32 {
lhs.rawValue | rhs.rawValue
}
}
extension EXT4.IncompatFeature {
public static func | (lhs: Self, rhs: Self) -> Self {
EXT4.IncompatFeature(rawValue: lhs.rawValue | rhs.rawValue)
}
public static func | (lhs: Self, rhs: Self) -> UInt32 {
lhs.rawValue | rhs.rawValue
}
}
extension EXT4.RoCompatFeature {
public static func | (lhs: Self, rhs: Self) -> Self {
EXT4.RoCompatFeature(rawValue: lhs.rawValue | rhs.rawValue)
}
public static func | (lhs: Self, rhs: Self) -> UInt32 {
lhs.rawValue | rhs.rawValue
}
}
extension EXT4.FileModeFlag {
public static func | (lhs: Self, rhs: Self) -> Self {
Self(rawValue: lhs.rawValue | rhs.rawValue)
}
public static func | (lhs: Self, rhs: Self) -> UInt16 {
lhs.rawValue | rhs.rawValue
}
}
extension EXT4.XAttrEntry {
init(using bytes: [UInt8]) throws {
guard bytes.count == 16 else {
throw EXT4.Error.invalidXattrEntry
}
nameLength = bytes[0]
nameIndex = bytes[1]
let rawValue = Array(bytes[2...3])
valueOffset = rawValue.withUnsafeBytes { $0.loadLittleEndian(as: UInt16.self) }
let rawValueInum = Array(bytes[4...7])
valueInum = rawValueInum.withUnsafeBytes { $0.loadLittleEndian(as: UInt32.self) }
let rawSize = Array(bytes[8...11])
valueSize = rawSize.withUnsafeBytes { $0.loadLittleEndian(as: UInt32.self) }
let rawHash = Array(bytes[12...])
hash = rawHash.withUnsafeBytes { $0.loadLittleEndian(as: UInt32.self) }
}
}
extension EXT4 {
static func tupleToArray<T>(_ tuple: T) -> [UInt8] {
let reflection = Mirror(reflecting: tuple)
return reflection.children.compactMap { $0.value as? UInt8 }
}
}
@@ -0,0 +1,101 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import Foundation
import SystemPackage
extension EXT4 {
class FileTree {
class FileTreeNode {
let inode: InodeNumber
let name: String
var children: [Ptr<FileTreeNode>] = []
var blocks: (start: UInt32, end: UInt32)?
var additionalBlocks: [(start: UInt32, end: UInt32)]?
var link: InodeNumber?
private weak var parent: Ptr<FileTreeNode>?
init(
inode: InodeNumber,
name: String,
parent: Ptr<FileTreeNode>?,
children: [Ptr<FileTreeNode>] = [],
blocks: (start: UInt32, end: UInt32)? = nil,
additionalBlocks: [(start: UInt32, end: UInt32)]? = nil,
link: InodeNumber? = nil
) {
self.inode = inode
self.name = name
self.children = children
self.blocks = blocks
self.additionalBlocks = additionalBlocks
self.link = link
self.parent = parent
}
deinit {
self.children.removeAll()
self.children = []
self.blocks = nil
self.additionalBlocks = nil
self.link = nil
}
var path: FilePath? {
var components: [String] = [self.name]
var _ptr = self.parent
while let ptr = _ptr {
components.append(ptr.pointee.name)
_ptr = ptr.pointee.parent
}
let path = components.reversed().joined(separator: "/")
return FilePath(path).lexicallyNormalized()
}
}
var root: Ptr<FileTreeNode>
init(_ root: InodeNumber, _ name: String) {
self.root = Ptr(FileTreeNode(inode: root, name: name, parent: nil))
}
func lookup(path: FilePath) -> Ptr<FileTreeNode>? {
var components: [String] = path.items
var node = self.root
if components.first == "/" {
components = Array(components.dropFirst())
}
if components.count == 0 {
return node
}
for component in components {
var found = false
for childPtr in node.pointee.children {
let child = childPtr.pointee
if child.name == component {
node = childPtr
found = true
break
}
}
guard found else {
return nil
}
}
return node
}
}
}
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,197 @@
//===----------------------------------------------------------------------===//
// Copyright © 2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationOS
import Foundation
// JBD2 on-disk format reference:
// https://www.kernel.org/doc/html/latest/filesystems/ext4/journal.html
extension EXT4.Formatter {
/// Entry point called from close() when journaling is enabled.
func initializeJournal(
config: EXT4.JournalConfig,
filesystemUUID: (
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8
)
) throws -> UInt32 {
let journalBlocks = try calculateJournalSize(requestedSize: config.size, usableBlocks: blockCount)
// Align to block boundary before recording start.
if self.pos % self.blockSize != 0 {
try self.seek(block: self.currentBlock + 1)
}
let journalStartBlock = self.currentBlock
try writeJournalSuperblock(journalBlocks: journalBlocks, filesystemUUID: filesystemUUID)
try zeroJournalBlocks(count: journalBlocks - 1)
try setupJournalInode(startBlock: journalStartBlock, blockCount: journalBlocks)
return journalBlocks
}
// MARK: - Private helpers
private func calculateJournalSize(requestedSize: UInt64?, usableBlocks: UInt32) throws -> UInt32 {
if let size = requestedSize {
let blocks = size / UInt64(self.blockSize)
// JBD2_MIN_JOURNAL_BLOCKS: the kernel refuses to mount with fewer.
// blocks == 0 would also cause a UInt32 underflow in the caller.
guard blocks >= EXT4.MinJournalBlocks else {
throw EXT4.Formatter.Error.journalTooSmall(size)
}
// Safe: any journal large enough to overflow UInt32 (>16 TiB at 4 KiB block size)
// would fail at the I/O layer before this conversion is reached.
return UInt32(blocks)
}
// Default sizing: scale with the usable content area, with a floor determined by
// JBD2_MIN_JOURNAL_BLOCKS and a ceiling that follows e2fsprogs convention: 128 MiB for
// filesystems up to 128 GiB, and 1 GiB for larger filesystems. The larger ceiling was
// introduced in e2fsprogs 1.43.2:
// https://e2fsprogs.sourceforge.net/e2fsprogs-release.html#1.43.2
let usableBytes = UInt64(usableBlocks) * UInt64(self.blockSize)
let scaledBytes = usableBytes / 64 // 1/64th of the usable area, matching e2fsprogs defaults
let minBytes: UInt64 = UInt64(EXT4.MinJournalBlocks) * UInt64(self.blockSize)
let maxBytes: UInt64 = usableBytes > 128.gib() ? 1.gib() : 128.mib()
let clampedBytes = min(max(scaledBytes, minBytes), maxBytes)
// Safe: clampedBytes ≤ 1 GiB and blockSize ≥ 1, so the quotient fits in UInt32.
return UInt32(clampedBytes / UInt64(self.blockSize))
}
private func writeJournalSuperblock(
journalBlocks: UInt32,
filesystemUUID: (
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8
)
) throws {
// Safe: blockSize is UInt32; widening to Int (64-bit on all supported platforms) never truncates.
var buf = [UInt8](repeating: 0, count: Int(self.blockSize))
// JBD2 is a big-endian format regardless of host byte order (§3.6.1).
func writeU32BigEndian(_ value: UInt32, at offset: Int) {
buf[offset] = UInt8((value >> 24) & 0xFF)
buf[offset + 1] = UInt8((value >> 16) & 0xFF)
buf[offset + 2] = UInt8((value >> 8) & 0xFF)
buf[offset + 3] = UInt8(value & 0xFF)
}
// JBD2 block header (§3.6.3): https://www.kernel.org/doc/html/latest/filesystems/ext4/journal.html#block-header
writeU32BigEndian(EXT4.JournalMagic, at: 0x00) // h_magic
writeU32BigEndian(4, at: 0x04) // h_blocktype = superblock v2
writeU32BigEndian(1, at: 0x08) // h_sequence
// JBD2 superblock body (§3.6.4): https://www.kernel.org/doc/html/latest/filesystems/ext4/journal.html#super-block
writeU32BigEndian(self.blockSize, at: 0x0C) // s_blocksize
writeU32BigEndian(journalBlocks, at: 0x10) // s_maxlen
writeU32BigEndian(1, at: 0x14) // s_first (first usable block)
writeU32BigEndian(1, at: 0x18) // s_sequence
// 0x1C s_start: left zero — kernel treats zero as "journal empty, begin at s_first"
// 0x20 s_errno: left zero — no prior abort error
// 0x24 s_feature_compat: left zero — no optional features (e.g. data-block checksums)
// 0x28 s_feature_incompat: left zero — non-zero unrecognised flags would cause mount refusal
// 0x2C s_feature_ro_compat: left zero — no flags defined by the spec
// s_uuid at 0x30 (16 bytes)
let uuidBytes = [
filesystemUUID.0, filesystemUUID.1, filesystemUUID.2, filesystemUUID.3,
filesystemUUID.4, filesystemUUID.5, filesystemUUID.6, filesystemUUID.7,
filesystemUUID.8, filesystemUUID.9, filesystemUUID.10, filesystemUUID.11,
filesystemUUID.12, filesystemUUID.13, filesystemUUID.14, filesystemUUID.15,
]
buf[0x30..<0x40] = uuidBytes[...]
writeU32BigEndian(1, at: 0x40) // s_nr_users
let maxTrans = min(journalBlocks / 4, 32768)
writeU32BigEndian(maxTrans, at: 0x48) // s_max_transaction
writeU32BigEndian(maxTrans, at: 0x4C) // s_max_trans_data
// s_users[0] at 0x100 (first entry of 768-byte users array)
buf[0x100..<0x110] = uuidBytes[...]
try self.handle.write(contentsOf: buf)
}
private func zeroJournalBlocks(count: UInt32) throws {
guard count > 0 else { return }
let chunkSize = 1.mib()
// Safe: both operands are UInt32, so their product peaks at ~17 TiB, which fits
// in Int64 (the width of Int on all 64-bit Apple platforms).
let totalBytes = Int(count) * Int(self.blockSize)
let zeroBuf = [UInt8](repeating: 0, count: min(Int(chunkSize), totalBytes))
var remaining = totalBytes
while remaining > 0 {
let toWrite = min(zeroBuf.count, remaining)
try self.handle.write(contentsOf: zeroBuf[0..<toWrite])
remaining -= toWrite
}
}
private func setupJournalInode(startBlock: UInt32, blockCount: UInt32) throws {
var journalInode = EXT4.Inode()
journalInode.mode = EXT4.Inode.Mode(.S_IFREG, 0o600)
journalInode.uid = 0
journalInode.gid = 0
let size = UInt64(blockCount) * UInt64(self.blockSize)
journalInode.sizeLow = size.lo
journalInode.sizeHigh = size.hi
let now = Date().fs()
journalInode.atime = now.lo
journalInode.atimeExtra = now.hi
journalInode.ctime = now.lo
journalInode.ctimeExtra = now.hi
journalInode.mtime = now.lo
journalInode.mtimeExtra = now.hi
journalInode.crtime = now.lo
journalInode.crtimeExtra = now.hi
journalInode.linksCount = 1
journalInode.extraIsize = UInt16(EXT4.ExtraIsize)
journalInode.flags = EXT4.InodeFlag.extents.rawValue | EXT4.InodeFlag.hugeFile.rawValue
// Journal is one contiguous allocation → numExtents = 1 → extent tree fits inline
// in the inode, so writeExtents needs no extra disk I/O for extent index blocks.
// Safe: blockCount is at most UInt32.max and startBlock ≥ 0, so the addition could
// theoretically overflow — but zeroJournalBlocks would have already failed with an
// I/O error if the journal extended past the end of the filesystem image.
journalInode = try self.writeExtents(journalInode, (startBlock, startBlock + blockCount))
self.inodes[Int(EXT4.JournalInode) - 1].pointee = journalInode
}
func journalInodeBlockBackup() -> (
UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32,
UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32,
UInt32
) {
let ji = self.inodes[Int(EXT4.JournalInode) - 1].pointee
// s_jnl_blocks layout (§4.1.2): first 15 words = i_block[] extent-tree data,
// 16th word (index 15) = i_size_high, 17th word (index 16) = i_size.
var words = [UInt32](repeating: 0, count: 17)
withUnsafeBytes(of: ji.block) { bytes in
for i in 0..<15 {
words[i] = bytes.load(fromByteOffset: i * 4, as: UInt32.self)
}
}
words[15] = ji.sizeHigh // i_size_high (16th element per spec)
words[16] = ji.sizeLow // i_size (17th element per spec)
return (
words[0], words[1], words[2], words[3],
words[4], words[5], words[6], words[7],
words[8], words[9], words[10], words[11],
words[12], words[13], words[14], words[15],
words[16]
)
}
}
@@ -0,0 +1,36 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
extension EXT4 {
class Ptr<T> {
let underlying: UnsafeMutablePointer<T>
var pointee: T {
get { underlying.pointee }
set { underlying.pointee = newValue }
}
init(_ value: T) {
self.underlying = .allocate(capacity: 1)
self.underlying.initialize(to: value)
}
deinit {
underlying.deinitialize(count: 1)
underlying.deallocate()
}
}
}
@@ -0,0 +1,277 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import Foundation
import SystemPackage
extension EXT4 {
/// The `EXT4Reader` opens a block device, parses the superblock, and loads group descriptors & inodes.
public class EXT4Reader {
public var superBlock: EXT4.SuperBlock {
self._superBlock
}
let handle: FileHandle
let _superBlock: EXT4.SuperBlock
private var groupDescriptors: [UInt32: EXT4.GroupDescriptor] = [:]
private var inodes: [InodeNumber: EXT4.Inode] = [:]
var hardlinks: [FilePath: InodeNumber] = [:]
var tree: EXT4.FileTree = EXT4.FileTree(EXT4.RootInode, ".")
var blockSize: UInt64 { UInt64(_superBlock.blockSize) }
private var groupDescriptorSize: UInt16 {
if _superBlock.featureIncompat & EXT4.IncompatFeature.bit64.rawValue != 0 {
return _superBlock.descSize
}
return UInt16(MemoryLayout<EXT4.GroupDescriptor>.size)
}
public init(blockDevice: FilePath) throws {
guard FileManager.default.fileExists(atPath: blockDevice.description) else {
throw EXT4.Error.notFound(blockDevice.description)
}
guard let fileHandle = FileHandle(forReadingAtPath: blockDevice) else {
throw Error.notFound(blockDevice.description)
}
self.handle = fileHandle
try handle.seek(toOffset: EXT4.SuperBlockOffset)
let superBlockSize = MemoryLayout<EXT4.SuperBlock>.size
guard let data = try? self.handle.read(upToCount: superBlockSize) else {
throw EXT4.Error.couldNotReadSuperBlock(blockDevice.description, EXT4.SuperBlockOffset, superBlockSize)
}
let sb = data.withUnsafeBytes { ptr in
ptr.loadLittleEndian(as: EXT4.SuperBlock.self)
}
guard sb.magic == EXT4.SuperBlockMagic else {
throw EXT4.Error.invalidSuperBlock
}
self._superBlock = sb
var items: [(item: Ptr<EXT4.FileTree.FileTreeNode>, inode: InodeNumber)] = [
(self.tree.root, EXT4.RootInode)
]
while items.count > 0 {
guard let item = items.popLast() else {
break
}
let (itemPtr, inodeNum) = item
let childItems = try self.children(of: inodeNum)
let root = itemPtr.pointee
for (itemName, itemInodeNum) in childItems {
if itemName == "." || itemName == ".." {
continue
}
if self.inodes[itemInodeNum] != nil {
// we have seen this inode before, we will hard link this file to it
guard let parentPath = itemPtr.pointee.path else {
continue
}
let path = parentPath.join(itemName)
self.hardlinks[path] = itemInodeNum
continue
}
let blocks = try self.getExtents(inode: itemInodeNum)
let itemTreeNode = FileTree.FileTreeNode(
inode: itemInodeNum,
name: itemName,
parent: itemPtr,
children: []
)
if let blocks {
if blocks.count > 1 {
itemTreeNode.additionalBlocks = Array(blocks.dropFirst())
}
itemTreeNode.blocks = blocks.first
}
let itemTreeNodePtr = Ptr(itemTreeNode)
root.children.append(itemTreeNodePtr)
itemPtr.pointee = root
let itemInode = try self.getInode(number: itemInodeNum)
if itemInode.mode.isDir() {
items.append((itemTreeNodePtr, itemInodeNum))
}
}
}
}
deinit {
try? self.handle.close()
}
private func readGroupDescriptor(_ number: UInt32) throws -> GroupDescriptor {
let bs = self.blockSize
let offset = bs + UInt64(number) * UInt64(self.groupDescriptorSize)
try self.handle.seek(toOffset: offset)
guard let data = try? self.handle.read(upToCount: MemoryLayout<EXT4.GroupDescriptor>.size) else {
throw EXT4.Error.couldNotReadGroup(number)
}
let gd = data.withUnsafeBytes { ptr in
ptr.loadLittleEndian(as: EXT4.GroupDescriptor.self)
}
return gd
}
private func readInode(_ number: UInt32) throws -> Inode {
let inodeGroupNumber = ((number - 1) / self._superBlock.inodesPerGroup)
let numberInGroup = UInt64((number - 1) % self._superBlock.inodesPerGroup)
let gd = try getGroupDescriptor(inodeGroupNumber)
let inodeTableStart = UInt64(gd.inodeTableLow) * self.blockSize
let inodeOffset: UInt64 = inodeTableStart + numberInGroup * UInt64(_superBlock.inodeSize)
try self.handle.seek(toOffset: inodeOffset)
guard let inodeData = try self.handle.read(upToCount: MemoryLayout<EXT4.Inode>.size) else {
throw EXT4.Error.couldNotReadInode(number)
}
let inode = inodeData.withUnsafeBytes { ptr in
ptr.loadLittleEndian(as: EXT4.Inode.self)
}
return inode
}
private func getDirTree(_ number: InodeNumber) throws -> [(String, InodeNumber)] {
var children: [(String, InodeNumber)] = []
let extents = try getExtents(inode: number) ?? []
for (start, end) in extents {
try self.seek(block: start)
for i in 0..<(end - start) {
guard let dirEntryBlock = try self.handle.read(upToCount: Int(self.blockSize)) else {
throw EXT4.Error.couldNotReadBlock(start + i)
}
let childEntries = try getDirEntries(dirTree: dirEntryBlock)
children.append(contentsOf: childEntries)
}
}
return children.sorted { a, b in
a.0 < b.0
}
}
private func getDirEntries(dirTree: Data) throws -> [(String, InodeNumber)] {
var children: [(String, InodeNumber)] = []
var offset = 0
let entrySize = MemoryLayout<DirectoryEntry>.size
while offset < dirTree.count {
let dirEntry = dirTree.subdata(in: offset..<offset + entrySize).withUnsafeBytes {
$0.loadLittleEndian(as: DirectoryEntry.self)
}
guard dirEntry.recordLength >= entrySize else {
break
}
if dirEntry.inode == 0 {
offset += Int(dirEntry.recordLength)
continue
}
let nameData = dirTree.subdata(in: offset + 8..<offset + 8 + Int(dirEntry.nameLength))
let name = String(data: nameData, encoding: .utf8) ?? ""
children.append((name, dirEntry.inode))
offset += Int(dirEntry.recordLength)
}
return children.sorted { a, b in
a.0 < b.0
}
}
func getExtents(inode: InodeNumber) throws -> [(start: UInt32, end: UInt32)]? {
let inode = try self.getInode(number: inode)
let inodeBlock = Data(tupleToArray(inode.block))
var offset = 0
var extents: [(start: UInt32, end: UInt32)] = []
let extentHeaderSize = MemoryLayout<ExtentHeader>.size
let extentIndexSize = MemoryLayout<ExtentIndex>.size
let extentLeafSize = MemoryLayout<ExtentLeaf>.size
// read extent header
let header = inodeBlock.subdata(in: offset..<offset + extentHeaderSize).withUnsafeBytes {
$0.loadLittleEndian(as: ExtentHeader.self)
}
guard header.magic == EXT4.ExtentHeaderMagic else {
return []
}
offset += extentHeaderSize // Jump to entries
switch header.depth {
case 0:
// When depth is 0 the extent header is followed by extent leaves
for _ in 0..<header.entries {
let leaf = inodeBlock.subdata(in: offset..<offset + extentLeafSize).withUnsafeBytes {
$0.loadLittleEndian(as: ExtentLeaf.self)
}
extents.append((leaf.startLow, leaf.startLow + UInt32(leaf.length)))
offset += extentLeafSize
}
case 1:
// When depth is 1 the extent header is followed by extent indices which point to leaves
for _ in 0..<header.entries {
let indexNode = inodeBlock.subdata(in: offset..<offset + extentIndexSize).withUnsafeBytes {
$0.loadLittleEndian(as: ExtentIndex.self)
}
try self.seek(block: indexNode.leafLow)
guard let block = try self.handle.read(upToCount: Int(self.blockSize)) else {
throw EXT4.Error.couldNotReadBlock(indexNode.leafLow)
}
var blockOffset = 0
let leafHeader = block.subdata(in: blockOffset..<extentHeaderSize).withUnsafeBytes {
$0.loadLittleEndian(as: ExtentHeader.self)
}
guard leafHeader.magic == EXT4.ExtentHeaderMagic else {
throw Error.invalidExtents
}
blockOffset += extentHeaderSize
for _ in 0..<leafHeader.entries {
let leaf = block.subdata(in: blockOffset..<blockOffset + extentLeafSize).withUnsafeBytes {
$0.loadLittleEndian(as: ExtentLeaf.self)
}
extents.append((leaf.startLow, leaf.startLow + UInt32(leaf.length)))
blockOffset += extentLeafSize
}
offset += extentIndexSize
}
default:
throw Error.deepExtentsUnimplemented
}
return extents
}
// MARK: Internal functions
func getInode(number: UInt32) throws -> Inode {
if let inode = self.inodes[number] {
return inode
}
let inode = try readInode(number)
self.inodes[number] = inode
return inode
}
func getGroupDescriptor(_ number: UInt32) throws -> GroupDescriptor {
if let gd = self.groupDescriptors[number] {
return gd
}
let gd = try readGroupDescriptor(number)
self.groupDescriptors[number] = gd
return gd
}
func children(of number: EXT4.InodeNumber) throws -> [(String, InodeNumber)] {
try getDirTree(number)
}
}
}
@@ -0,0 +1,633 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
// swiftlint:disable large_tuple
import Foundation
extension EXT4 {
public struct SuperBlock {
public var inodesCount: UInt32 = 0
public var blocksCountLow: UInt32 = 0
public var reservedBlocksCountLow: UInt32 = 0
public var freeBlocksCountLow: UInt32 = 0
public var freeInodesCount: UInt32 = 0
public var firstDataBlock: UInt32 = 0
public var logBlockSize: UInt32 = 0
public var logClusterSize: UInt32 = 0
public var blockSize: UInt32 { 1024 << logBlockSize }
public var blocksPerGroup: UInt32 = 0
public var clustersPerGroup: UInt32 = 0
public var inodesPerGroup: UInt32 = 0
public var mtime: UInt32 = 0
public var wtime: UInt32 = 0
public var mountCount: UInt16 = 0
public var maxMountCount: UInt16 = 0
public var magic: UInt16 = 0
public var state: UInt16 = 0
public var errors: UInt16 = 0
public var minorRevisionLevel: UInt16 = 0
public var lastCheck: UInt32 = 0
public var checkInterval: UInt32 = 0
public var creatorOS: UInt32 = 0
public var revisionLevel: UInt32 = 0
public var defaultReservedUid: UInt16 = 0
public var defaultReservedGid: UInt16 = 0
public var firstInode: UInt32 = 0
public var inodeSize: UInt16 = 0
public var blockGroupNr: UInt16 = 0
public var featureCompat: UInt32 = 0
public var featureIncompat: UInt32 = 0
public var featureRoCompat: UInt32 = 0
public var uuid:
(
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8
) = (
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0
)
public var volumeName:
(
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8
) = (
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0
)
public var lastMounted:
(
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8
) = (
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0
)
public var algorithmUsageBitmap: UInt32 = 0
public var preallocBlocks: UInt8 = 0
public var preallocDirBlocks: UInt8 = 0
public var reservedGdtBlocks: UInt16 = 0
public var journalUUID:
(
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8
) = (
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0
)
public var journalInum: UInt32 = 0
public var journalDev: UInt32 = 0
public var lastOrphan: UInt32 = 0
public var hashSeed: (UInt32, UInt32, UInt32, UInt32) = (0, 0, 0, 0)
public var defHashVersion: UInt8 = 0
public var journalBackupType: UInt8 = 0
public var descSize: UInt16 = UInt16(MemoryLayout<GroupDescriptor>.size)
public var defaultMountOpts: UInt32 = 0
public var firstMetaBg: UInt32 = 0
public var mkfsTime: UInt32 = 0
public var journalBlocks:
(
UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32,
UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32,
UInt32
) = (
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0
)
public var blocksCountHigh: UInt32 = 0
public var reservedBlocksCountHigh: UInt32 = 0
public var freeBlocksCountHigh: UInt32 = 0
public var minExtraIsize: UInt16 = 0
public var wantExtraIsize: UInt16 = 0
public var flags: UInt32 = 0
public var raidStride: UInt16 = 0
public var mmpInterval: UInt16 = 0
public var mmpBlock: UInt64 = 0
public var raidStripeWidth: UInt32 = 0
public var logGroupsPerFlex: UInt8 = 0
public var checksumType: UInt8 = 0
public var reservedPad: UInt16 = 0
public var kbytesWritten: UInt64 = 0
public var snapshotInum: UInt32 = 0
public var snapshotID: UInt32 = 0
public var snapshotRBlocksCount: UInt64 = 0
public var snapshotList: UInt32 = 0
public var errorCount: UInt32 = 0
public var firstErrorTime: UInt32 = 0
public var firstErrorInode: UInt32 = 0
public var firstErrorBlock: UInt64 = 0
public var firstErrorFunc:
(
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8
) = (
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0
)
public var firstErrorLine: UInt32 = 0
public var lastErrorTime: UInt32 = 0
public var lastErrorInode: UInt32 = 0
public var lastErrorLine: UInt32 = 0
public var lastErrorBlock: UInt64 = 0
public var lastErrorFunc:
(
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8
) = (
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0
)
public var mountOpts:
(
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8
) = (
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0
)
public var userQuotaInum: UInt32 = 0
public var groupQuotaInum: UInt32 = 0
public var overheadBlocks: UInt32 = 0
public var backupBgs: (UInt32, UInt32) = (0, 0)
public var encryptAlgos: (UInt8, UInt8, UInt8, UInt8) = (0, 0, 0, 0)
public var encryptPwSalt:
(
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8
) = (
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0
)
public var lpfInode: UInt32 = 0
public var projectQuotaInum: UInt32 = 0
public var checksumSeed: UInt32 = 0
public var wtimeHigh: UInt8 = 0
public var mtimeHigh: UInt8 = 0
public var mkfsTimeHigh: UInt8 = 0
public var lastcheckHigh: UInt8 = 0
public var firstErrorTimeHigh: UInt8 = 0
public var lastErrorTimeHigh: UInt8 = 0
public var pad: (UInt8, UInt8) = (0, 0)
public var reserved:
(
UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32,
UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32,
UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32,
UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32,
UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32,
UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32,
UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32,
UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32,
UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32,
UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32,
UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32,
UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32, UInt32
) = (
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0
)
public var checksum: UInt32 = 0
}
static let JournalMagic: UInt32 = 0xC03B_3998
static let JournalInode: InodeNumber = 8
static let MinJournalBlocks: UInt32 = 1024 // JBD2_MIN_JOURNAL_BLOCKS
struct DefaultMountOpts {
static let journalData: UInt32 = 0x0020 // data=journal
static let journalOrdered: UInt32 = 0x0040 // data=ordered
static let journalWriteback: UInt32 = 0x0060 // data=writeback
}
struct CompatFeature {
let rawValue: UInt32
static let dirPrealloc = CompatFeature(rawValue: 0x1)
static let imagicInodes = CompatFeature(rawValue: 0x2)
static let hasJournal = CompatFeature(rawValue: 0x4)
static let extAttr = CompatFeature(rawValue: 0x8)
static let resizeInode = CompatFeature(rawValue: 0x10)
static let dirIndex = CompatFeature(rawValue: 0x20)
static let lazyBg = CompatFeature(rawValue: 0x40)
static let excludeInode = CompatFeature(rawValue: 0x80)
static let excludeBitmap = CompatFeature(rawValue: 0x100)
static let sparseSuper2 = CompatFeature(rawValue: 0x200)
}
struct IncompatFeature {
let rawValue: UInt32
static let compression = IncompatFeature(rawValue: 0x1)
static let filetype = IncompatFeature(rawValue: 0x2)
static let recover = IncompatFeature(rawValue: 0x4)
static let journalDev = IncompatFeature(rawValue: 0x8)
static let metaBg = IncompatFeature(rawValue: 0x10)
static let extents = IncompatFeature(rawValue: 0x40)
static let bit64 = IncompatFeature(rawValue: 0x80)
static let mmp = IncompatFeature(rawValue: 0x100)
static let flexBg = IncompatFeature(rawValue: 0x200)
static let eaInode = IncompatFeature(rawValue: 0x400)
static let dirdata = IncompatFeature(rawValue: 0x1000)
static let csumSeed = IncompatFeature(rawValue: 0x2000)
static let largedir = IncompatFeature(rawValue: 0x4000)
static let inlineData = IncompatFeature(rawValue: 0x8000)
static let encrypt = IncompatFeature(rawValue: 0x10000)
}
struct RoCompatFeature {
let rawValue: UInt32
static let sparseSuper = RoCompatFeature(rawValue: 0x1)
static let largeFile = RoCompatFeature(rawValue: 0x2)
static let btreeDir = RoCompatFeature(rawValue: 0x4)
static let hugeFile = RoCompatFeature(rawValue: 0x8)
static let gdtCsum = RoCompatFeature(rawValue: 0x10)
static let dirNlink = RoCompatFeature(rawValue: 0x20)
static let extraIsize = RoCompatFeature(rawValue: 0x40)
static let hasSnapshot = RoCompatFeature(rawValue: 0x80)
static let quota = RoCompatFeature(rawValue: 0x100)
static let bigalloc = RoCompatFeature(rawValue: 0x200)
static let metadataCsum = RoCompatFeature(rawValue: 0x400)
static let replica = RoCompatFeature(rawValue: 0x800)
static let readonly = RoCompatFeature(rawValue: 0x1000)
static let project = RoCompatFeature(rawValue: 0x2000)
}
struct BlockGroupFlag {
let rawValue: UInt16
static let inodeUninit = BlockGroupFlag(rawValue: 0x1)
static let blockUninit = BlockGroupFlag(rawValue: 0x2)
static let inodeZeroed = BlockGroupFlag(rawValue: 0x4)
}
struct GroupDescriptor {
let blockBitmapLow: UInt32
let inodeBitmapLow: UInt32
let inodeTableLow: UInt32
let freeBlocksCountLow: UInt16
let freeInodesCountLow: UInt16
let usedDirsCountLow: UInt16
let flags: UInt16
let excludeBitmapLow: UInt32
let blockBitmapCsumLow: UInt16
let inodeBitmapCsumLow: UInt16
let itableUnusedLow: UInt16
let checksum: UInt16
}
struct GroupDescriptor64 {
let groupDescriptor: GroupDescriptor
let blockBitmapHigh: UInt32
let inodeBitmapHigh: UInt32
let inodeTableHigh: UInt32
let freeBlocksCountHigh: UInt16
let freeInodesCountHigh: UInt16
let usedDirsCountHigh: UInt16
let itableUnusedHigh: UInt16
let excludeBitmapHigh: UInt32
let blockBitmapCsumHigh: UInt16
let inodeBitmapCsumHigh: UInt16
let reserved: UInt32
}
public struct FileModeFlag: Sendable {
let rawValue: UInt16
public static let S_IXOTH = FileModeFlag(rawValue: 0x1)
public static let S_IWOTH = FileModeFlag(rawValue: 0x2)
public static let S_IROTH = FileModeFlag(rawValue: 0x4)
public static let S_IXGRP = FileModeFlag(rawValue: 0x8)
public static let S_IWGRP = FileModeFlag(rawValue: 0x10)
public static let S_IRGRP = FileModeFlag(rawValue: 0x20)
public static let S_IXUSR = FileModeFlag(rawValue: 0x40)
public static let S_IWUSR = FileModeFlag(rawValue: 0x80)
public static let S_IRUSR = FileModeFlag(rawValue: 0x100)
public static let S_ISVTX = FileModeFlag(rawValue: 0x200)
public static let S_ISGID = FileModeFlag(rawValue: 0x400)
public static let S_ISUID = FileModeFlag(rawValue: 0x800)
public static let S_IFIFO = FileModeFlag(rawValue: 0x1000)
public static let S_IFCHR = FileModeFlag(rawValue: 0x2000)
public static let S_IFDIR = FileModeFlag(rawValue: 0x4000)
public static let S_IFBLK = FileModeFlag(rawValue: 0x6000)
public static let S_IFREG = FileModeFlag(rawValue: 0x8000)
public static let S_IFLNK = FileModeFlag(rawValue: 0xA000)
public static let S_IFSOCK = FileModeFlag(rawValue: 0xC000)
public static let TypeMask = FileModeFlag(rawValue: 0xF000)
}
public typealias InodeNumber = UInt32
public struct Inode {
public var mode: UInt16 = 0
public var uid: UInt16 = 0
public var sizeLow: UInt32 = 0
public var atime: UInt32 = 0
public var ctime: UInt32 = 0
public var mtime: UInt32 = 0
public var dtime: UInt32 = 0
public var gid: UInt16 = 0
public var linksCount: UInt16 = 0
public var blocksLow: UInt32 = 0
public var flags: UInt32 = 0
public var version: UInt32 = 0
public var block:
(
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8
) = (
0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0
)
public var generation: UInt32 = 0
public var xattrBlockLow: UInt32 = 0
public var sizeHigh: UInt32 = 0
public var obsoleteFragmentAddr: UInt32 = 0
public var blocksHigh: UInt16 = 0
public var xattrBlockHigh: UInt16 = 0
public var uidHigh: UInt16 = 0
public var gidHigh: UInt16 = 0
public var checksumLow: UInt16 = 0
public var reserved: UInt16 = 0
public var extraIsize: UInt16 = 0
public var checksumHigh: UInt16 = 0
public var ctimeExtra: UInt32 = 0
public var mtimeExtra: UInt32 = 0
public var atimeExtra: UInt32 = 0
public var crtime: UInt32 = 0
public var crtimeExtra: UInt32 = 0
public var versionHigh: UInt32 = 0
public var projid: UInt32 = 0 // Size until this point is 160 bytes
public var inlineXattrs:
( // 96 bytes for extended attributes
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8, UInt8,
UInt8, UInt8, UInt8, UInt8, UInt8, UInt8
) = (
0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0
)
public static func Mode(_ mode: FileModeFlag, _ perm: UInt16) -> UInt16 {
mode.rawValue | perm
}
}
struct InodeFlag {
let rawValue: UInt32
static let secRm = InodeFlag(rawValue: 0x1)
static let unRm = InodeFlag(rawValue: 0x2)
static let compressed = InodeFlag(rawValue: 0x4)
static let sync = InodeFlag(rawValue: 0x8)
static let immutable = InodeFlag(rawValue: 0x10)
static let append = InodeFlag(rawValue: 0x20)
static let noDump = InodeFlag(rawValue: 0x40)
static let noAtime = InodeFlag(rawValue: 0x80)
static let dirtyCompressed = InodeFlag(rawValue: 0x100)
static let compressedClusters = InodeFlag(rawValue: 0x200)
static let noCompress = InodeFlag(rawValue: 0x400)
static let encrypted = InodeFlag(rawValue: 0x800)
static let hashedIndex = InodeFlag(rawValue: 0x1000)
static let magic = InodeFlag(rawValue: 0x2000)
static let journalData = InodeFlag(rawValue: 0x4000)
static let noTail = InodeFlag(rawValue: 0x8000)
static let dirSync = InodeFlag(rawValue: 0x10000)
static let topDir = InodeFlag(rawValue: 0x20000)
static let hugeFile = InodeFlag(rawValue: 0x40000)
static let extents = InodeFlag(rawValue: 0x80000)
static let eaInode = InodeFlag(rawValue: 0x200000)
static let eofBlocks = InodeFlag(rawValue: 0x400000)
static let snapfile = InodeFlag(rawValue: 0x0100_0000)
static let snapfileDeleted = InodeFlag(rawValue: 0x0400_0000)
static let snapfileShrunk = InodeFlag(rawValue: 0x0800_0000)
static let inlineData = InodeFlag(rawValue: 0x1000_0000)
static let projectIDInherit = InodeFlag(rawValue: 0x2000_0000)
static let reserved = InodeFlag(rawValue: 0x8000_0000)
}
struct ExtentHeader {
let magic: UInt16
let entries: UInt16
let max: UInt16
let depth: UInt16
let generation: UInt32
}
struct ExtentIndex {
let block: UInt32
let leafLow: UInt32
let leafHigh: UInt16
let unused: UInt16
}
struct ExtentLeaf {
let block: UInt32
let length: UInt16
let startHigh: UInt16
let startLow: UInt32
}
struct ExtentTail {
let checksum: UInt32
}
struct ExtentIndexNode {
var header: ExtentHeader
var indices: [ExtentIndex]
}
struct ExtentLeafNode {
var header: ExtentHeader
var leaves: [ExtentLeaf]
}
struct DirectoryEntry {
let inode: InodeNumber
let recordLength: UInt16
let nameLength: UInt8
let fileType: UInt8
// let name: [UInt8]
}
enum FileType: UInt8 {
case unknown = 0x0
case regular = 0x1
case directory = 0x2
case character = 0x3
case block = 0x4
case fifo = 0x5
case socket = 0x6
case symbolicLink = 0x7
}
struct DirectoryEntryTail {
let reservedZero1: UInt32
let recordLength: UInt16
let reservedZero2: UInt8
let fileType: UInt8
let checksum: UInt32
}
struct DirectoryTreeRoot {
let dot: DirectoryEntry
let dotName: [UInt8]
let dotDot: DirectoryEntry
let dotDotName: [UInt8]
let reservedZero: UInt32
let hashVersion: UInt8
let infoLength: UInt8
let indirectLevels: UInt8
let unusedFlags: UInt8
let limit: UInt16
let count: UInt16
let block: UInt32
// let entries: [DirectoryTreeEntry]
}
struct DirectoryTreeNode {
let fakeInode: UInt32
let fakeRecordLength: UInt16
let nameLength: UInt8
let fileType: UInt8
let limit: UInt16
let count: UInt16
let block: UInt32
// let entries: [DirectoryTreeEntry]
}
struct DirectoryTreeEntry {
let hash: UInt32
let block: UInt32
}
struct DirectoryTreeTail {
let reserved: UInt32
let checksum: UInt32
}
struct XAttrEntry {
let nameLength: UInt8
let nameIndex: UInt8
let valueOffset: UInt16
let valueInum: UInt32
let valueSize: UInt32
let hash: UInt32
}
struct XAttrHeader {
let magic: UInt32
let referenceCount: UInt32
let blocks: UInt32
let hash: UInt32
let checksum: UInt32
let reserved: [UInt32]
}
}
extension EXT4.Inode {
public static func Root() -> EXT4.Inode {
var inode = Self() // inode
inode.mode = Self.Mode(.S_IFDIR, 0o755)
inode.linksCount = 2
inode.uid = 0
inode.gid = 0
// time
let now = Date().fs()
let now_lo: UInt32 = now.lo
let now_hi: UInt32 = now.hi
inode.atime = now_lo
inode.atimeExtra = now_hi
inode.ctime = now_lo
inode.ctimeExtra = now_hi
inode.mtime = now_lo
inode.mtimeExtra = now_hi
inode.crtime = now_lo
inode.crtimeExtra = now_hi
inode.flags = EXT4.InodeFlag.hugeFile.rawValue
inode.extraIsize = UInt16(EXT4.ExtraIsize)
return inode
}
}
@@ -0,0 +1,317 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import Foundation
/*
* Note: Both the entries and values for the attributes need to occupy a size that is a multiple of 4,
* meaning, in cases where the attribute name or value is not a multiple of 4, it is padded with 0
* until it reaches that size.
*/
extension EXT4 {
public struct ExtendedAttribute {
public static let prefixMap: [Int: String] = [
1: "user.",
2: "system.posix_acl_access",
3: "system.posix_acl_default",
4: "trusted.",
6: "security.",
7: "system.",
8: "system.richacl",
]
let name: String
let index: UInt8
let value: [UInt8]
var sizeValue: UInt32 {
UInt32((value.count + 3) & ~3)
}
var sizeEntry: UInt32 {
UInt32((name.count + 3) & ~3 + 16) // 16 bytes are needed to store other metadata for the xattr entry
}
var size: UInt32 {
sizeEntry + sizeValue
}
var fullName: String {
Self.decompressName(id: Int(index), suffix: name)
}
var hash: UInt32 {
var hash: UInt32 = 0
for char in name {
hash = (hash << 5) ^ (hash >> 27) ^ UInt32(char.asciiValue!)
}
var i = 0
while i + 3 < value.count {
let s = value[i..<i + 4]
let v = s.withUnsafeBytes { $0.loadLittleEndian(as: UInt32.self) }
hash = (hash << 16) ^ (hash >> 16) ^ v
i += 4
}
if value.count % 4 != 0 {
let last = value.count & ~3
var buff: [UInt8] = [0, 0, 0, 0]
for (i, byte) in value[last...].enumerated() {
buff[i] = byte
}
let v = buff.withUnsafeBytes { $0.loadLittleEndian(as: UInt32.self) }
hash = (hash << 16) ^ (hash >> 16) ^ v
}
return hash
}
init(name: String, value: [UInt8]) {
let compressed = Self.compressName(name)
self.name = compressed.str
self.index = compressed.id
self.value = value
}
init(idx: UInt8, compressedName name: String, value: [UInt8]) {
self.name = name
self.index = idx
self.value = value
}
// MARK: Class methods
public static func compressName(_ name: String) -> (id: UInt8, str: String) {
for (id, prefix) in prefixMap.sorted(by: { $1.1.count < $0.1.count }) where name.hasPrefix(prefix) {
return (UInt8(id), String(name.dropFirst(prefix.count)))
}
return (0, name)
}
public static func decompressName(id: Int, suffix: String) -> String {
guard let prefix = prefixMap[id] else {
return suffix
}
return "\(prefix)\(suffix)"
}
}
public struct FileXattrsState {
private let inodeCapacity: UInt32
private let blockCapacity: UInt32
private let inode: UInt32 // the inode number for which we are tracking these xattrs
var inlineAttributes: [ExtendedAttribute] = []
var blockAttributes: [ExtendedAttribute] = []
private var usedSizeInline: UInt32 = 0
private var usedSizeBlock: UInt32 = 0
private var inodeFreeBytes: UInt32 {
self.inodeCapacity - EXT4.XattrInodeHeaderSize - usedSizeInline - 4 // need to have 4 null bytes b/w xattr entries and values
}
private var blockFreeBytes: UInt32 {
self.blockCapacity - EXT4.XattrBlockHeaderSize - usedSizeBlock - 4
}
init(inode: UInt32, inodeXattrCapacity: UInt32, blockCapacity: UInt32) {
self.inode = inode
self.inodeCapacity = inodeXattrCapacity
self.blockCapacity = blockCapacity
}
public mutating func add(_ attribute: ExtendedAttribute) throws {
let size = attribute.size
if size <= inodeFreeBytes {
usedSizeInline += size
inlineAttributes.append(attribute)
return
}
if size <= blockFreeBytes {
usedSizeBlock += size
blockAttributes.append(attribute)
return
}
throw Error.insufficientSpace(Int(self.inode))
}
public func writeInlineAttributes(buffer: inout [UInt8]) throws {
var idx = 0
withUnsafeLittleEndianBytes(
of: EXT4.XAttrHeaderMagic,
body: { bytes in
for byte in bytes {
buffer[idx] = byte
idx += 1
}
})
try Self.write(buffer: &buffer, attrs: self.inlineAttributes, start: UInt16(idx), delta: 0, inline: true)
}
public func writeBlockAttributes(buffer: inout [UInt8]) throws {
var idx = 0
for val in [EXT4.XAttrHeaderMagic, 1, 1] {
withUnsafeLittleEndianBytes(
of: UInt32(val),
body: { bytes in
for byte in bytes {
buffer[idx] = byte
idx += 1
}
})
}
while idx != 32 {
buffer[idx] = 0
idx += 1
}
var attributes = self.blockAttributes
attributes.sort(by: {
if $0.index != $1.index {
return $0.index < $1.index
}
if $0.name.count != $1.name.count {
return $0.name.count < $1.name.count
}
return $0.name < $1.name
})
try Self.write(buffer: &buffer, attrs: attributes, start: UInt16(idx), delta: UInt16(idx), inline: false)
}
/// Writes the specified list of extended attribute entries and their values to the provided
/// This method does not fill in any headers (Inode inline / block level) that may be required to parse these attributes
///
/// - Parameters:
/// - buffer: An array of [UInt8] where the data will be written into
/// - attrs: The list of ExtendedAttributes to write
/// - start: the index from where data should be put into the buffer - useful when if you dont want this method to be overwriting existing data
/// - delta: index from where the begin the offset calculations
/// - inline: if the byte buffer being written into is an inline data block for an inode: Determines the hash calculation
private static func write(
buffer: inout [UInt8], attrs: [ExtendedAttribute], start: UInt16, delta: UInt16, inline: Bool
) throws {
var offset: UInt16 = UInt16(buffer.count) + delta - start
var front = Int(start)
var end = buffer.count
for attribute in attrs {
guard end - front >= 4 else {
throw Error.malformedXattrBuffer
}
var out: [UInt8] = []
let v = attribute.sizeValue
offset -= UInt16(v)
out.append(UInt8(attribute.name.count))
out.append(attribute.index)
withUnsafeLittleEndianBytes(
of: UInt16(offset),
body: { bytes in
out.append(contentsOf: bytes)
})
out.append(contentsOf: [0, 0, 0, 0]) // these next four bytes indicate that the attr values are in the same block
withUnsafeLittleEndianBytes(
of: UInt32(attribute.value.count),
body: { bytes in
out.append(contentsOf: bytes)
})
if !inline {
withUnsafeLittleEndianBytes(
of: UInt32(attribute.hash),
body: { bytes in
out.append(contentsOf: bytes)
})
} else {
out.append(contentsOf: [0, 0, 0, 0])
}
guard let name = attribute.name.data(using: .ascii) else {
throw Error.convertAsciiString(attribute.name)
}
out.append(contentsOf: [UInt8](name))
while out.count < Int(attribute.sizeEntry) { // ensure that xattr entry size is a multiple of 4
out.append(0)
}
for (i, byte) in out.enumerated() {
buffer[front + i] = byte
}
front += out.count
end -= Int(attribute.sizeValue)
for (i, byte) in attribute.value.enumerated() {
buffer[end + i] = byte
}
}
}
public static func read(buffer: [UInt8], start: Int, offset: Int) throws -> [ExtendedAttribute] {
var i = start
var attribs: [ExtendedAttribute] = []
// 16 is the size of 1 XAttrEntry
while i + 16 <= buffer.count {
let attributeStart = i
let rawXattrEntry = Array(buffer[i..<i + 16])
let xattrEntry = try EXT4.XAttrEntry(using: rawXattrEntry)
i += 16
var endIndex = i + Int(xattrEntry.nameLength)
guard endIndex <= buffer.count else {
continue
}
let rawName = buffer[i..<endIndex]
guard let name = String(bytes: rawName, encoding: .ascii) else {
throw Error.nonAsciiXattrName
}
let valueStart = Int(xattrEntry.valueOffset) + offset
let valueEnd = Int(xattrEntry.valueOffset) + Int(xattrEntry.valueSize) + offset
guard valueEnd <= buffer.count else {
break
}
let value = [UInt8](buffer[valueStart..<valueEnd])
let xattr = ExtendedAttribute(idx: xattrEntry.nameIndex, compressedName: name, value: value)
attribs.append(xattr)
i = attributeStart + xattr.sizeEntry
// The next 4 bytes being null indicate that there are no more attributes to read
endIndex = i + 3
guard endIndex < buffer.count else {
continue
}
if Array(buffer[i...i + 3]) == [0, 0, 0, 0] {
break
}
}
return attribs
}
public enum Error: CustomStringConvertible, Swift.Error {
case insufficientSpace(_ inode: Int)
case malformedXattrBuffer
case convertAsciiString(_ s: String)
case nonAsciiXattrName
case missingXAttrHeader
public var description: String {
switch self {
case .insufficientSpace(let inode):
return "cannot fit xattr for inode \(inode)"
case .malformedXattrBuffer:
return "malformed extended attribute buffer"
case .convertAsciiString(let s):
return "cannot convert string \(s) to a list of ASCII characters"
case .nonAsciiXattrName:
return "extended attribute name contains non-ASCII bytes"
case .missingXAttrHeader:
return "missing header for extended attribute entry"
}
}
}
}
}
+354
View File
@@ -0,0 +1,354 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationOS
import Foundation
/**
```
# EXT4 Filesystem Layout
The EXT4 filesystem divides the disk into an upfront metadata section followed by several logical groups known as block groups. The
metadata section looks like this:
+--------------------------+
| Boot Sector (1024) |
+--------------------------+
| Superblock (1024) |
+--------------------------+
| Empty (2048) |
+--------------------------+
| |
| [Block Group Descriptors]|
| |
| - Free/used block bitmap |
| - Free/used inode bitmap |
| - Inode table pointer |
| - Other metadata |
| |
+--------------------------+
## Block Groups
Each block group optionally stores a copy of the superblock and group descriptor table for disaster recovery.
The rest of the block group comprises of data blocks. The size of each block group is dynamically decided
while formatting, based on total amount of space available on the disk.
+--------------------------+
| Block Group 0 |
| +------------------+ |
| | Super Block | |
| +------------------+ |
| | Group Desc. | |
| +------------------+ |
| | Data Blocks | |
| | | |
| +------------------+ |
+--------------------------+
| Block Group 1 |
| +------------------+ |
| | Super Block | |
| +------------------+ |
| | Group Desc. | |
| +------------------+ |
| | Data Blocks | |
| | | |
| +------------------+ |
+--------------------------+
| ... |
+--------------------------+
| Block Group N |
| +------------------+ |
| | Super Block | |
| +------------------+ |
| | Group Desc. | |
| +------------------+ |
| | Data Blocks | |
| | | |
| +------------------+ |
+--------------------------+
The descriptor for each block group contain the following information:
- Block Bitmap
- Inode Bitmap
- Pointer to Inode Table
- other metadata such as used block count, num. dirs etc.
### Block Bitmap
A sequence of bits, where each bit represents a block in the block group.
1: In use block
0: Free block
+---------------+---------------+
| Block |
| Bitmap |
+---------------+---------------+
| 1 0 1 0 1 1 0 0 |
+---------------+---------------+
| | | | | | | |
| | | | | | | |
v v v v v v v v
+---+---+---+---+---+---+---+---+
| B | | B | | B | B | | |
+---+---+---+---+---+---+---+---+
Whenever a file is created, free data blocks are identified by using this table.
When it is deleted, the corresponding data blocks are marked as free.
### Inode Bitmap
A sequence of bits, where each bit represents an inode in the block group. Since
inodes per group is a fixed number, this bitmap is made to be of sufficient length
to accommodate that many inodes
1: In use inode
0: Free inode
+---------------+---------------+
| Inode |
| Bitmap |
+---------------+---------------+
| 1 0 1 0 1 1 0 0 |
+---------------+---------------+
| | | | | | | |
| | | | | | | |
v v v v v v v v
+---+---+---+---+---+---+---+---+
| I | | I | | I | I | | |
+---+---+---+---+---+---+---+---+
## Inode table
Inode table provides a mapping from Inode -> Data blocks. In this implementation, inode size is set to 256 bytes.
Inode table uses extents to efficiently describe the mapping.
+-----------------------+
| Inode Table |
+-----------------------+
| Inode | Metadata |
+-------+---------------+
| 1 | permissions |
| | size |
| | user ID |
| | group ID |
| | timestamps |
| | block |
| | blocks count |
+-------+---------------+
| 2 | ... |
+-------+---------------+
| ... | ... |
+-------+---------------+
The length of `block` field in the inode table is 60 bytes. This field contains an extent tree
that holds information about ranges of blocks used by the file. For smaller files, the entire extent
tree can be stored within this field.
+-----------------------+
| Inode |
+-----------------------+
| Metadata |
+-----------------------+
| Extent Tree |
| +-------------------+ |
| | Extent Leaf Node | |
| +-------------------+ |
| | - Start Block | |
| | - Block Count | |
| | - ... | |
| +-------------------+ |
+-----------------------+
For larger files which span across multiple non-contiguous blocks, extent tree's root points to extent
blocks, which in-turn point to the blocks used by the file
+-----------------------+
| Extent Tree |
| +-------------------+ |
| | Extent Root | |
| +-------------------+ |
| | - Pointers to | |
| | Extent Blocks | |
| +-------------------+ |
+-----------------------+
|
v
+-----------------------+
| Extent Block |
+-----------------------+
| +-------------------+ |
| | Extent Leaf Node | |
| +-------------------+ |
| | - Start Block | |
| | - Block Count | |
| | - ... | |
| +-------------------+ |
| +-------------------+ |
| | Extent Leaf Node | |
| +-------------------+ |
| | - Start Block | |
| | - Block Count | |
| | - ... | |
| +-------------------+ |
+-----------------------+
## Directory entries
The data blocks for directory inodes point to a list of directory entries. Each entry
consists of only a name and inode number. The name and inode number correspond to the
name and inode number of the children of the directory
+-------------------------+
| Directory Entry |
+-------------------------+
| inode | rec_len | name |
+-------------------------+
| 2 | 1 | "." |
+-------------------------+
| Directory Entry |
+-------------------------+
| inode | rec_len | name |
+-------------------------+
| 1 | 2 | ".." |
+-------------------------+
| Directory Entry |
+-------------------------+
| inode | rec_len | name |
+-------------------------+
| 11 | 10 | lost& |
| | | found |
+-------------------------+
More details can be found here https://ext4.wiki.kernel.org/index.php/Ext4_Disk_Layout
```
*/
/// A type for interacting with ext4 file systems.
///
/// The `Ext4` class provides functionality to read the superblock of an existing ext4 block device
/// and format a new block device with the ext4 file system.
///
/// Usage:
/// - To read the superblock of an existing ext4 block device, create an instance of `Ext4` with the
/// path to the block device
/// - To format a new block device with ext4, create an instance of `Ext4.Formatter` with the path to the block
/// device and call the `close()` method.
///
/// Example 1: Read an existing block device
/// ```swift
/// let blockDevice = URL(filePath: "/dev/sdb")
/// // succeeds if a valid ext4 fs is found at path
/// let ext4 = try Ext4(blockDevice: blockDevice)
/// print("Block size: \(ext4.blockSize)")
/// print("Total size: \(ext4.size)")
///
/// // Reading the superblock
/// let superblock = ext4.superblock
/// print("Superblock information:")
/// print("Total blocks: \(superblock.blocksCountLow)")
/// ```
///
/// Example 2: Format a new block device (Refer [`EXT4.Formatter`](x-source-tag://EXT4.Formatter) for more info)
/// ```swift
/// let devicePath = URL(filePath: "/dev/sdc")
/// let formatter = try EXT4.Formatter(devicePath, blockSize: 4096)
/// try formatter.close()
/// ```
public enum EXT4 {
public static let SuperBlockMagic: UInt16 = 0xef53
static let ExtentHeaderMagic: UInt16 = 0xf30a
static let XAttrHeaderMagic: UInt32 = 0xea02_0000
static let DefectiveBlockInode: InodeNumber = 1
static let RootInode: InodeNumber = 2
static let FirstInode: InodeNumber = 11
static let LostAndFoundInode: InodeNumber = 11
static let InodeActualSize: UInt32 = 160 // 160 bytes used by metadata
static let InodeExtraSize: UInt32 = 96 // 96 bytes for inline xattrs
static let InodeSize: UInt32 = UInt32(MemoryLayout<Inode>.size) // 256 bytes. This is the max size of an inode
static let XattrInodeHeaderSize: UInt32 = 4
static let XattrBlockHeaderSize: UInt32 = 32
static let ExtraIsize: UInt16 = UInt16(InodeActualSize) - 128
static let MaxLinks: UInt32 = 65000
static let MaxBlocksPerExtent: UInt32 = 0x8000
static let MaxFileSize: UInt64 = 128.gib()
static let SuperBlockOffset: UInt64 = 1024
public struct JournalConfig: Sendable {
public var size: UInt64?
public var defaultMode: JournalMode?
public enum JournalMode: Sendable {
case writeback
case ordered
case journal
}
public init(size: UInt64? = nil, defaultMode: JournalMode? = nil) {
self.size = size
self.defaultMode = defaultMode
}
public static let `default` = JournalConfig()
}
}
extension EXT4 {
// `EXT4` errors.
public enum Error: Swift.Error, CustomStringConvertible, Sendable, Equatable {
case notFound(_ path: String)
case couldNotReadSuperBlock(_ path: String, _ offset: UInt64, _ size: Int)
case invalidSuperBlock
case deepExtentsUnimplemented
case invalidExtents
case invalidXattrEntry
case couldNotReadBlock(_ block: UInt32)
case invalidPathEncoding(_ path: String)
case couldNotReadInode(_ inode: UInt32)
case couldNotReadGroup(_ group: UInt32)
public var description: String {
switch self {
case .notFound(let path):
return "file at path \(path) not found"
case .couldNotReadSuperBlock(let path, let offset, let size):
return "could not read \(size) bytes of superblock from \(path) at offset \(offset)"
case .invalidSuperBlock:
return "not a valid EXT4 superblock"
case .deepExtentsUnimplemented:
return "deep extents are not supported"
case .invalidExtents:
return "extents invalid or corrupted"
case .invalidXattrEntry:
return "invalid extended attribute entry"
case .couldNotReadBlock(let block):
return "could not read block \(block)"
case .invalidPathEncoding(let path):
return "path encoding for '\(path)' is invalid, must be ascii or utf8"
case .couldNotReadInode(let inode):
return "could not read inode \(inode)"
case .couldNotReadGroup(let group):
return "could not read group descriptor \(group)"
}
}
}
}
@@ -0,0 +1,215 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationArchive
import Foundation
import SystemPackage
extension EXT4.EXT4Reader {
public func export(archive: FilePath) throws {
let config = ArchiveWriterConfiguration(
format: .paxRestricted, filter: .none, options: [Options.xattrformat(.schily)])
let writer = try ArchiveWriter(configuration: config)
try writer.open(file: archive.url)
var items = self.tree.root.pointee.children
let hardlinkedInodes = Set(self.hardlinks.values)
var hardlinkTargets: [EXT4.InodeNumber: FilePath] = [:]
while items.count > 0 {
let itemPtr = items.removeFirst()
let item = itemPtr.pointee
let inode = try self.getInode(number: item.inode)
let entry = WriteEntry()
let mode = inode.mode
let size: UInt64 = (UInt64(inode.sizeHigh) << 32) | UInt64(inode.sizeLow)
entry.permissions = mode_t(mode)
guard let path = item.path else {
continue
}
if hardlinkedInodes.contains(item.inode) {
hardlinkTargets[item.inode] = path
}
guard self.hardlinks[path] == nil else {
continue
}
var attributes: [EXT4.ExtendedAttribute] = []
let buffer: [UInt8] = EXT4.tupleToArray(inode.inlineXattrs)
if !buffer.allZeros {
try attributes.append(contentsOf: Self.readInlineExtendedAttributes(from: buffer))
}
if inode.xattrBlockLow != 0 {
let block = inode.xattrBlockLow
try self.seek(block: block)
guard let buffer = try self.handle.read(upToCount: Int(self.blockSize)) else {
throw EXT4.Error.couldNotReadBlock(block)
}
try attributes.append(contentsOf: Self.readBlockExtendedAttributes(from: [UInt8](buffer)))
}
var xattrs: [String: Data] = [:]
for attribute in attributes {
guard attribute.fullName != "system.data" else {
continue
}
xattrs[attribute.fullName] = Data(attribute.value)
}
let pathStr = path.description
entry.path = pathStr
entry.size = Int64(size)
entry.group = gid_t(inode.gidHigh) << 16 | gid_t(inode.gid)
entry.owner = uid_t(inode.uidHigh) << 16 | uid_t(inode.uid)
entry.creationDate = Date(fsTimestamp: UInt64(inode.crtimeExtra) << 32 | UInt64(inode.crtime))
entry.modificationDate = Date(fsTimestamp: UInt64(inode.mtimeExtra) << 32 | UInt64(inode.mtime))
entry.contentAccessDate = Date(fsTimestamp: UInt64(inode.atimeExtra) << 32 | UInt64(inode.atime))
entry.xattrs = xattrs
if mode.isDir() {
entry.fileType = .directory
for child in item.children {
items.append(child)
}
if pathStr == "" {
continue
}
try writer.writeEntry(entry: entry, data: nil)
} else if mode.isReg() {
entry.fileType = .regular
var data = Data()
var remaining: UInt64 = size
if let block = item.blocks {
for dataBlock in block.start..<block.end {
try self.seek(block: dataBlock)
var count: UInt64
if remaining > self.blockSize {
count = self.blockSize
} else {
count = remaining
}
guard let dataBytes = try self.handle.read(upToCount: Int(count)) else {
throw EXT4.Error.couldNotReadBlock(dataBlock)
}
data.append(dataBytes)
remaining -= UInt64(dataBytes.count)
}
}
if let additionalBlocks = item.additionalBlocks {
for block in additionalBlocks {
for dataBlock in block.start..<block.end {
try self.seek(block: dataBlock)
var count: UInt64
if remaining > self.blockSize {
count = self.blockSize
} else {
count = remaining
}
guard let dataBytes = try self.handle.read(upToCount: Int(count)) else {
throw EXT4.Error.couldNotReadBlock(dataBlock)
}
data.append(dataBytes)
remaining -= UInt64(dataBytes.count)
}
}
}
try writer.writeEntry(entry: entry, data: data)
} else if mode.isLink() {
entry.fileType = .symbolicLink
if size < 60 {
let linkBytes = EXT4.tupleToArray(inode.block)
entry.symlinkTarget = String(bytes: linkBytes.prefix(Int(size)), encoding: .utf8) ?? ""
} else {
if let block = item.blocks {
try self.seek(block: block.start)
guard let linkBytes = try self.handle.read(upToCount: Int(size)) else {
throw EXT4.Error.couldNotReadBlock(block.start)
}
entry.symlinkTarget = String(bytes: linkBytes, encoding: .utf8) ?? ""
}
}
try writer.writeEntry(entry: entry, data: nil)
} else { // do not process sockets, fifo, character and block devices
continue
}
}
for (path, number) in self.hardlinks {
guard let targetPath = hardlinkTargets[number] else {
continue
}
let inode = try self.getInode(number: number)
let entry = WriteEntry()
entry.path = path.description
entry.hardlink = targetPath.description
entry.permissions = mode_t(inode.mode)
entry.group = gid_t(inode.gidHigh) << 16 | gid_t(inode.gid)
entry.owner = uid_t(inode.uidHigh) << 16 | uid_t(inode.uid)
entry.creationDate = Date(fsTimestamp: UInt64(inode.crtimeExtra) << 32 | UInt64(inode.crtime))
entry.modificationDate = Date(fsTimestamp: UInt64(inode.mtimeExtra) << 32 | UInt64(inode.mtime))
entry.contentAccessDate = Date(fsTimestamp: UInt64(inode.atimeExtra) << 32 | UInt64(inode.atime))
try writer.writeEntry(entry: entry, data: nil)
}
try writer.finishEncoding()
}
@available(*, deprecated, renamed: "readInlineExtendedAttributes(from:)")
public static func readInlineExtenedAttributes(from buffer: [UInt8]) throws -> [EXT4.ExtendedAttribute] {
try readInlineExtendedAttributes(from: buffer)
}
public static func readInlineExtendedAttributes(from buffer: [UInt8]) throws -> [EXT4.ExtendedAttribute] {
let header = buffer[0..<4].withUnsafeBytes { $0.loadLittleEndian(as: UInt32.self) }
if header != EXT4.XAttrHeaderMagic {
throw EXT4.FileXattrsState.Error.missingXAttrHeader
}
return try EXT4.FileXattrsState.read(buffer: buffer, start: 4, offset: 4)
}
@available(*, deprecated, renamed: "readBlockExtendedAttributes(from:)")
public static func readBlockExtenedAttributes(from buffer: [UInt8]) throws -> [EXT4.ExtendedAttribute] {
try readBlockExtendedAttributes(from: buffer)
}
public static func readBlockExtendedAttributes(from buffer: [UInt8]) throws -> [EXT4.ExtendedAttribute] {
let header = buffer[0..<4].withUnsafeBytes { $0.loadLittleEndian(as: UInt32.self) }
if header != EXT4.XAttrHeaderMagic {
throw EXT4.FileXattrsState.Error.missingXAttrHeader
}
return try EXT4.FileXattrsState.read(buffer: [UInt8](buffer), start: 32, offset: 0)
}
func seek(block: UInt32) throws {
try self.handle.seek(toOffset: UInt64(block) * blockSize)
}
}
extension Date {
init(fsTimestamp: UInt64) {
if fsTimestamp == 0 {
self = Date.distantPast
return
}
// 32 bits - base: seconds since January 1, 1970, signed (negative for pre-1970 dates)
// 2 bits - epoch: overflow counter (0-3), how many times the 32-bit seconds field has wrapped
// 30 bits - nanoseconds (0-999,999,999)
let base = Int32(truncatingIfNeeded: fsTimestamp)
let epoch = Int64(fsTimestamp & 0x3_0000_0000)
let seconds = Int64(base) + epoch
let nanoseconds = Double(fsTimestamp >> 34) / 1_000_000_000
self = Date(timeIntervalSince1970: Double(seconds) + nanoseconds)
}
}
@@ -0,0 +1,495 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import Foundation
import SystemPackage
extension EXT4 {
public enum PathIOError: Swift.Error, CustomStringConvertible {
case notFound(String)
case notAFile(String)
case isDirectory(String)
case notADirectory(String)
case symlinkLoop(String)
case invalidPath(String)
public var description: String {
switch self {
case .notFound(let p): return "no such file or directory: \(p)"
case .notAFile(let p): return "not a regular file: \(p)"
case .isDirectory(let p): return "is a directory: \(p)"
case .notADirectory(let p): return "not a directory: \(p)"
case .symlinkLoop(let p): return "symlink loop while resolving: \(p)"
case .invalidPath(let p): return "invalid path: \(p)"
}
}
}
}
// MARK: - Public API
extension EXT4.EXT4Reader {
/// Return true if a path exists (file or directory) in this ext4 device.
public func exists(_ path: FilePath, followSymlinks: Bool = true) -> Bool {
(try? resolvePath(path, followSymlinks: followSymlinks).inode) != nil
}
/// Get the total number of blocks in the filesystem
private var totalBlocks: UInt64 {
let lo = UInt64(_superBlock.blocksCountLow)
let hi = UInt64(_superBlock.blocksCountHigh)
return lo | (hi << 32)
}
/// Validate that a physical block address is within device bounds
private func validateBlockAddress(_ block: UInt32) throws {
guard UInt64(block) < totalBlocks else {
throw EXT4.PathIOError.invalidPath("block address \(block) exceeds device bounds (\(totalBlocks) blocks)")
}
}
/// Metadata (inode + inode number) for a path.
public func stat(_ path: FilePath, followSymlinks: Bool = true) throws -> (inodeNumber: EXT4.InodeNumber, inode: EXT4.Inode) {
let resolved = try resolvePath(path, followSymlinks: followSymlinks)
return (resolved.inodeNum, try getInode(number: resolved.inodeNum))
}
/// List a directory's entries (names only). Does not include "." or "..".
public func listDirectory(_ path: FilePath) throws -> [String] {
let (inoNum, ino) = try stat(path)
guard ino.mode.isDir() else {
throw EXT4.PathIOError.notADirectory(path.description)
}
let children = try children(of: inoNum)
return
children
.map { $0.0 }
.filter { $0 != "." && $0 != ".." }
.sorted()
}
/// Read bytes from a regular file at `path` into `buffer`, starting at `offset`.
/// Returns the number of bytes written to `buffer` (may be less than `buffer.count` at EOF).
/// - Note: Semantics mirror `read(2)`: partial reads are possible.
@discardableResult
public func readFile(
at path: FilePath,
into buffer: UnsafeMutableRawBufferPointer,
offset: UInt64 = 0,
followSymlinks: Bool = true
) throws -> Int {
let context = try prepareRead(path: path, offset: offset, followSymlinks: followSymlinks)
if buffer.count == 0 || context.maxReadable == 0 {
return 0
}
let want = min(UInt64(buffer.count), context.maxReadable)
return try performRead(
inodeNumber: context.inodeNumber,
start: context.start,
wantedBytes: want,
into: buffer
)
}
/// Read bytes from a regular file at `path` starting at `offset`.
/// If `count` is nil, reads to EOF. Returns exactly the requested bytes (or less at EOF).
public func readFile(
at path: FilePath,
offset: UInt64 = 0,
count: Int? = nil,
followSymlinks: Bool = true
) throws -> Data {
let context = try prepareRead(path: path, offset: offset, followSymlinks: followSymlinks)
let want = count.map { min(UInt64($0), context.maxReadable) } ?? context.maxReadable
if want == 0 {
return Data()
}
var out = Data(count: Int(want))
let wrote = try out.withUnsafeMutableBytes {
try performRead(
inodeNumber: context.inodeNumber,
start: context.start,
wantedBytes: want,
into: $0
)
}
if wrote < Int(want) {
out.removeSubrange(wrote..<Int(want))
}
return out
}
private func prepareRead(
path: FilePath,
offset: UInt64,
followSymlinks: Bool
) throws -> (inodeNumber: EXT4.InodeNumber, start: UInt64, maxReadable: UInt64) {
let (inoNum, inode) = try stat(path, followSymlinks: followSymlinks)
if inode.mode.isDir() {
throw EXT4.PathIOError.isDirectory(path.description)
}
if !inode.mode.isReg() {
throw EXT4.PathIOError.notAFile(path.description)
}
let fileSize: UInt64 = inodeFileSize(inode)
let start = min(offset, fileSize)
let maxReadable = fileSize - start
return (inodeNumber: inoNum, start: start, maxReadable: maxReadable)
}
private func performRead(
inodeNumber: EXT4.InodeNumber,
start: UInt64,
wantedBytes: UInt64,
into buffer: UnsafeMutableRawBufferPointer
) throws -> Int {
if wantedBytes == 0 {
return 0
}
guard let extents = try self.getExtents(inode: inodeNumber), !extents.isEmpty else {
return 0
}
for (physStartBlk, physEndBlk) in extents {
try validateBlockAddress(physStartBlk)
if physEndBlk > physStartBlk {
try validateBlockAddress(physEndBlk - 1)
}
}
guard let base = buffer.baseAddress else {
return 0
}
let desiredBytes = Int(min(wantedBytes, UInt64(buffer.count)))
if desiredBytes == 0 {
return 0
}
let blockSizeBytes = self.blockSize
let reqStart = start
let reqEnd = start + UInt64(desiredBytes)
var logicalOffset: UInt64 = 0
var bytesWritten = 0
for (physStartBlk, physEndBlk) in extents {
let extentBytes = UInt64(physEndBlk - physStartBlk) * blockSizeBytes
let logicalEnd = logicalOffset + extentBytes
if logicalEnd <= reqStart {
logicalOffset = logicalEnd
continue
}
if logicalOffset >= reqEnd {
break
}
let overlapStart = max(logicalOffset, reqStart)
let overlapEnd = min(logicalEnd, reqEnd)
var remaining = overlapEnd - overlapStart
if remaining == 0 {
logicalOffset = logicalEnd
continue
}
let offsetIntoExtent = overlapStart - logicalOffset
let absoluteByteOffset = (UInt64(physStartBlk) * blockSizeBytes) + offsetIntoExtent
do {
try self.handle.seek(toOffset: absoluteByteOffset)
} catch {
if bytesWritten > 0 {
return bytesWritten
}
throw EXT4.PathIOError.invalidPath("failed to seek to offset \(absoluteByteOffset): \(error)")
}
while remaining > 0 && bytesWritten < desiredBytes {
let chunk = min(desiredBytes - bytesWritten, Int(min(remaining, UInt64(1 << 20))))
let dest = UnsafeMutableRawBufferPointer(
start: base.advanced(by: bytesWritten),
count: chunk
)
do {
guard let data = try self.handle.read(upToCount: chunk) else {
return bytesWritten
}
if data.count == 0 {
return bytesWritten
}
// Copy the data to the destination buffer
data.withUnsafeBytes { sourceBytes in
dest.copyMemory(from: UnsafeRawBufferPointer(sourceBytes))
}
bytesWritten += data.count
remaining -= UInt64(data.count)
if data.count < chunk && remaining > 0 {
return bytesWritten
}
} catch {
if bytesWritten > 0 {
return bytesWritten
}
throw error
}
}
logicalOffset = logicalEnd
if bytesWritten >= desiredBytes {
break
}
}
return bytesWritten
}
// MARK: - Internals inside EXT4Reader
public struct ResolvedPath {
let inodeNum: EXT4.InodeNumber
let inode: EXT4.Inode
}
/// Resolve a path to an inode (optionally following symlinks).
/// Paths may be absolute ("/...") or relative (from "/").
public func resolvePath(_ path: FilePath, followSymlinks: Bool, maxSymlinks: Int = 40) throws -> ResolvedPath {
var components: [String] = normalize(path: path)
var current: EXT4.InodeNumber = EXT4.RootInode
var parentStack: [EXT4.InodeNumber] = [] // Track parent chain for proper ".." handling
var symlinkHops = 0
// Process components one at a time to handle symlinks in the middle of paths
var componentIndex = 0
while componentIndex < components.count {
let name = components[componentIndex]
if name == "." {
componentIndex += 1
continue
}
if name == ".." {
// Handle parent directory traversal
if current == EXT4.RootInode {
// At root, ".." points to itself
componentIndex += 1
continue
}
// Use parent stack if available
if !parentStack.isEmpty {
current = parentStack.removeLast()
} else {
// Fallback: look up ".." entry in filesystem
let entries = try children(of: current)
if let parent = entries.first(where: { $0.0 == ".." })?.1 {
current = parent
}
}
componentIndex += 1
continue
}
// Regular component: verify current is a directory and look up child
let currentInode = try getInode(number: current)
guard currentInode.mode.isDir() else {
throw EXT4.PathIOError.notADirectory(name)
}
let entries = try children(of: current)
guard let child = entries.first(where: { $0.0 == name }) else {
throw EXT4.PathIOError.notFound(name)
}
// Check if child is a symlink
let childInode = try getInode(number: child.1)
if childInode.mode.isLink() && followSymlinks {
// Enforce max symlink depth
symlinkHops += 1
if symlinkHops > maxSymlinks {
throw EXT4.PathIOError.symlinkLoop(FilePath(components.joined(separator: "/")).description)
}
// Read symlink target
let linkBytes = try readFileFromInode(inodeNum: child.1)
guard let linkTarget = String(data: linkBytes, encoding: .utf8), !linkTarget.isEmpty else {
throw EXT4.PathIOError.invalidPath("empty symlink target")
}
// Parse symlink target into components
let targetComponents = normalize(path: FilePath(linkTarget))
// Replace current component with symlink target components and continue
if linkTarget.hasPrefix("/") {
// Absolute symlink: reset to root
current = EXT4.RootInode
parentStack = []
// Replace the symlink component with target components + remaining path
components = targetComponents + Array(components[(componentIndex + 1)...])
componentIndex = 0 // Start from beginning with new path
} else {
// Relative symlink: continue from current directory
// Replace the symlink component with target components + remaining path
components = Array(components[0..<componentIndex]) + targetComponents + Array(components[(componentIndex + 1)...])
// Don't change componentIndex - continue from same position with expanded path
}
} else {
// Not a symlink or not following symlinks - descend into directory
parentStack.append(current)
current = child.1
componentIndex += 1
}
}
// All components processed - return final inode
let finalInode = try getInode(number: current)
return ResolvedPath(inodeNum: current, inode: finalInode)
}
/// Normalize a path into components, handling absolute and relative paths.
private func normalize(path: FilePath) -> [String] {
let s = path.description
let trimmed = s.hasPrefix("/") ? String(s.dropFirst()) : s
if trimmed.isEmpty { return [] }
return trimmed.split(separator: "/").map(String.init)
}
/// Read entire file content of a regular file given an inode (used for symlink targets).
private func readFileFromInode(inodeNum: EXT4.InodeNumber) throws -> Data {
let ino = try getInode(number: inodeNum)
guard ino.mode.isReg() || ino.mode.isLink() else {
return Data()
}
let size = inodeFileSize(ino)
if size == 0 { return Data() }
// Handle fast symlinks (target stored directly in inode block field)
if ino.mode.isLink() && size < 60 {
// Extract target from inode block field
let blockData = withUnsafeBytes(of: ino.block) { Data($0) }
return blockData.prefix(Int(size))
}
return try readFileBytesFromExtents(inodeNum: inodeNum, offset: 0, count: size)
}
/// Low-level read using extents, with explicit offset & length (in bytes).
private func readFileBytesFromExtents(inodeNum: EXT4.InodeNumber, offset: UInt64, count: UInt64) throws -> Data {
guard let extents = try self.getExtents(inode: inodeNum), !extents.isEmpty else {
return Data()
}
// Validate all extent blocks are within device bounds
for (startBlk, endBlk) in extents {
try validateBlockAddress(startBlk)
if endBlk > startBlk {
try validateBlockAddress(endBlk - 1)
}
}
var out = Data(capacity: Int(count))
var logicalOffset: UInt64 = 0
var bytesReadSuccessfully: Int = 0
let reqStart = offset
let reqEnd = offset + count
let bs = self.blockSize
for (startBlk, endBlk) in extents {
let extentBytes = UInt64(endBlk - startBlk) * bs
let logicalEnd = logicalOffset + extentBytes
if logicalEnd <= reqStart {
logicalOffset = logicalEnd
continue
}
if logicalOffset >= reqEnd { break }
let ovlStart = max(logicalOffset, reqStart)
let ovlEnd = min(logicalEnd, reqEnd)
let ovlLen = ovlEnd - ovlStart
if ovlLen == 0 {
logicalOffset = logicalEnd
continue
}
let offsetIntoExtent = ovlStart - logicalOffset
let absByteOffset = UInt64(startBlk) * bs + offsetIntoExtent
do {
try self.handle.seek(toOffset: absByteOffset)
} catch {
if bytesReadSuccessfully > 0 {
// Return partial data that was successfully read
return out
}
throw EXT4.PathIOError.invalidPath("failed to seek to offset \(absByteOffset): \(error)")
}
var left = ovlLen
while left > 0 {
let chunk = Int(min(left, 1 << 20))
do {
guard let data = try self.handle.read(upToCount: chunk) else {
let blk = UInt32(absByteOffset / bs)
throw EXT4.Error.couldNotReadBlock(blk)
}
out.append(data)
bytesReadSuccessfully += data.count
left -= UInt64(data.count)
if data.count < chunk && left > 0 {
// Partial read - return what we have
return out
}
} catch {
if bytesReadSuccessfully > 0 {
// Return partial data on error
return out
}
throw error
}
}
logicalOffset = logicalEnd
if out.count >= Int(count) { break }
}
if out.count > Int(count) { out.removeSubrange(Int(count)..<out.count) }
return out
}
/// Compute 64-bit file size from the inode fields (i_size).
/// ext4 stores low 32 bits in i_size_lo and the high 32 bits in i_size_high when 64-bit sizes are enabled.
private func inodeFileSize(_ inode: EXT4.Inode) -> UInt64 {
// The Containerization EXT4 Inode struct exposes mode and block fields; size fields
// are commonly named sizeLo/sizeHigh in this codebase.
// EXT4 supports 64-bit file sizes - always use both low and high parts.
let lo = UInt64(inode.sizeLow)
let hi = UInt64(inode.sizeHigh)
return lo | (hi << 32)
}
}
@@ -0,0 +1,113 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import Foundation
import SystemPackage
extension FilePath {
public static let Separator: String = "/"
public var bytes: [UInt8] {
self.withCString { cstr in
var ptr = cstr
var rawBytes: [UInt8] = []
while UInt(bitPattern: ptr) != 0 {
if ptr.pointee == 0x00 { break }
rawBytes.append(UInt8(bitPattern: ptr.pointee))
ptr = ptr.successor()
}
return rawBytes
}
}
public var base: String {
self.lastComponent?.string ?? "/"
}
public var dir: FilePath {
self.removingLastComponent()
}
public var url: URL {
URL(fileURLWithPath: self.string)
}
public var items: [String] {
self.components.map { $0.string }
}
public var isRoot: Bool { // platform agnostic
self.removingRoot().isEmpty
}
public init(_ url: URL) {
self.init(url.path(percentEncoded: false))
}
public func join(_ path: FilePath) -> FilePath {
self.pushing(path)
}
public func join(_ path: String) -> FilePath {
self.join(FilePath(path))
}
public func split() -> (dir: FilePath, base: String) {
(self.dir, self.base)
}
public func clean() -> FilePath {
self.lexicallyNormalized()
}
public static func rel(_ basepath: String, _ targpath: String) -> FilePath {
let base = FilePath(basepath)
let targ = FilePath(targpath)
if base == targ {
return "."
}
let baseComponents = base.items
let targComponents = targ.items
var commonPrefix = 0
while commonPrefix < min(baseComponents.count, targComponents.count)
&& baseComponents[commonPrefix] == targComponents[commonPrefix]
{
commonPrefix += 1
}
let upCount = baseComponents.count - commonPrefix
let relComponents = Array(repeating: "..", count: upCount) + targComponents[commonPrefix...]
return FilePath(relComponents.joined(separator: Self.Separator))
}
}
extension FileHandle {
public convenience init?(forWritingTo path: FilePath) {
self.init(forWritingAtPath: path.description)
}
public convenience init?(forReadingAtPath path: FilePath) {
self.init(forReadingAtPath: path.description)
}
public convenience init?(forReadingFrom path: FilePath) {
self.init(forReadingAtPath: path.description)
}
}
@@ -0,0 +1,67 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import Foundation
public struct FileTimestamps {
public var access: Date
public var modification: Date
public var creation: Date
public var now: Date
public var accessLo: UInt32 {
access.fs().lo
}
public var accessHi: UInt32 {
access.fs().hi
}
public var modificationLo: UInt32 {
modification.fs().lo
}
public var modificationHi: UInt32 {
modification.fs().hi
}
public var creationLo: UInt32 {
creation.fs().lo
}
public var creationHi: UInt32 {
creation.fs().hi
}
public var nowLo: UInt32 {
now.fs().lo
}
public var nowHi: UInt32 {
now.fs().hi
}
public init(access: Date?, modification: Date?, creation: Date?) {
now = Date()
self.access = access ?? now
self.modification = modification ?? now
self.creation = creation ?? now
}
public init() {
self.init(access: nil, modification: nil, creation: nil)
}
}
@@ -0,0 +1,255 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import ContainerizationArchive
import ContainerizationExtras
import ContainerizationOS
import Foundation
import SystemPackage
private typealias Hardlinks = [FilePath: FilePath]
extension EXT4.Formatter {
/// Unpack the provided archive on to the ext4 filesystem.
public func unpack(reader: ArchiveReader, progress: ProgressHandler? = nil) async throws {
try await self.unpackEntries(reader: reader, progress: progress)
}
/// Unpack an archive at the source URL on to the ext4 filesystem.
public func unpack(
source: URL,
format: ContainerizationArchive.Format = .paxRestricted,
compression: ContainerizationArchive.Filter = .gzip,
progress: ProgressHandler? = nil
) async throws {
// For zstd, decompress once and reuse for both passes to avoid double decompression.
let fileToRead: URL
let readerFilter: ContainerizationArchive.Filter
var decompressedFile: URL?
if progress != nil && compression == .zstd {
let decompressed = try ArchiveReader.decompressZstd(source)
fileToRead = decompressed
readerFilter = .none
decompressedFile = decompressed
} else {
fileToRead = source
readerFilter = compression
}
defer {
if let decompressedFile {
ArchiveReader.cleanUpDecompressedZstd(decompressedFile)
}
}
if let progress {
// First pass: scan headers to get totals (fast, metadata only)
let totals = try Self.scanArchiveHeaders(format: format, filter: readerFilter, file: fileToRead)
var totalEvents: [ProgressEvent] = []
if totals.size > 0 {
totalEvents.append(.addTotalSize(totals.size))
}
if totals.items > 0 {
totalEvents.append(.addTotalItems(totals.items))
}
if !totalEvents.isEmpty {
await progress(totalEvents)
}
}
// Unpack pass
let reader = try ArchiveReader(
format: format,
filter: readerFilter,
file: fileToRead
)
try await self.unpackEntries(reader: reader, progress: progress)
}
/// Scan archive headers to count the total number of bytes in regular files
/// and the total number of entries.
public static func scanArchiveHeaders(
format: ContainerizationArchive.Format,
filter: ContainerizationArchive.Filter,
file: URL
) throws -> (size: Int64, items: Int) {
let reader = try ArchiveReader(format: format, filter: filter, file: file)
var totalSize: Int64 = 0
var totalItems: Int = 0
for (entry, _) in reader.makeStreamingIterator() {
try Task.checkCancellation()
guard entry.path != nil else { continue }
totalItems += 1
if entry.fileType == .regular, entry.hardlink == nil, let size = entry.size {
totalSize += Int64(size)
}
}
return (size: totalSize, items: totalItems)
}
/// Core unpack logic. When `progress` is nil the handler calls are skipped.
private func unpackEntries(reader: ArchiveReader, progress: ProgressHandler?) async throws {
var hardlinks: Hardlinks = [:]
// Allocate a single 128KiB reusable buffer for all files to minimize allocations
// and reduce the number of read calls to libarchive.
let bufferSize = 128 * 1024
let reusableBuffer = UnsafeMutableBufferPointer<UInt8>.allocate(capacity: bufferSize)
defer { reusableBuffer.deallocate() }
for (entry, streamReader) in reader.makeStreamingIterator() {
try Task.checkCancellation()
guard var pathEntry = entry.path else {
continue
}
pathEntry = preProcessPath(s: pathEntry)
let path = FilePath(pathEntry)
if path.base.hasPrefix(".wh.") {
if path.base == ".wh..wh..opq" { // whiteout directory
try self.unlink(path: path.dir, directoryWhiteout: true)
if let progress {
await progress([.addItems(1)])
}
continue
}
let startIndex = path.base.index(path.base.startIndex, offsetBy: ".wh.".count)
let filePath = String(path.base[startIndex...])
let dir: FilePath = path.dir
try self.unlink(path: dir.join(filePath))
if let progress {
await progress([.addItems(1)])
}
continue
}
if let hardlink = entry.hardlink {
let hl = preProcessPath(s: hardlink)
hardlinks[path] = FilePath(hl)
if let progress {
await progress([.addItems(1)])
}
continue
}
let ts = FileTimestamps(
access: entry.contentAccessDate, modification: entry.modificationDate, creation: entry.creationDate)
switch entry.fileType {
case .directory:
try self.create(
path: path, mode: EXT4.Inode.Mode(.S_IFDIR, UInt16(entry.permissions)), ts: ts, uid: entry.owner,
gid: entry.group,
xattrs: entry.xattrs)
case .regular:
try self.create(
path: path, mode: EXT4.Inode.Mode(.S_IFREG, UInt16(entry.permissions)), ts: ts, buf: streamReader,
uid: entry.owner,
gid: entry.group, xattrs: entry.xattrs, fileBuffer: reusableBuffer)
if let progress, let size = entry.size {
await progress([.addSize(Int64(size))])
}
case .symbolicLink:
var symlinkTarget: FilePath?
if let target = entry.symlinkTarget {
symlinkTarget = FilePath(target)
}
try self.create(
path: path, link: symlinkTarget, mode: EXT4.Inode.Mode(.S_IFLNK, UInt16(entry.permissions)), ts: ts,
uid: entry.owner,
gid: entry.group, xattrs: entry.xattrs)
default:
if let progress {
await progress([.addItems(1)])
}
continue
}
if let progress {
await progress([.addItems(1)])
}
}
guard hardlinks.acyclic else {
throw UnpackError.circularLinks
}
for (path, _) in hardlinks {
if let resolvedTarget = try hardlinks.resolve(path) {
try self.link(link: path, target: resolvedTarget)
}
}
}
private func preProcessPath(s: String) -> String {
var p = s
if p.hasPrefix("./") {
p = String(p.dropFirst())
}
if !p.hasPrefix("/") {
p = "/" + p
}
return p
}
}
/// Common errors for unpacking an archive onto an ext4 filesystem.
public enum UnpackError: Swift.Error, CustomStringConvertible, Sendable, Equatable {
/// The name is invalid.
case invalidName(_ name: String)
/// A circular link is found.
case circularLinks
/// The description of the error.
public var description: String {
switch self {
case .invalidName(let name):
return "'\(name)' is an invalid name"
case .circularLinks:
return "circular links found"
}
}
}
extension Hardlinks {
fileprivate var acyclic: Bool {
for (_, target) in self {
var visited: Set<FilePath> = [target]
var next = target
while let item = self[next] {
if visited.contains(item) {
return false
}
next = item
visited.insert(next)
}
}
return true
}
fileprivate func resolve(_ key: FilePath) throws -> FilePath? {
let target = self[key]
guard let target else {
return nil
}
var next = target
var visited: Set<FilePath> = [next]
while let item = self[next] {
if visited.contains(item) {
throw UnpackError.circularLinks
}
next = item
visited.insert(next)
}
return next
}
}
@@ -0,0 +1,131 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
extension UInt64 {
public var lo: UInt32 {
UInt32(self & 0xffff_ffff)
}
public var hi: UInt32 {
UInt32(self >> 32)
}
public static func - (lhs: Self, rhs: UInt32) -> UInt64 {
lhs - UInt64(rhs)
}
public static func % (lhs: Self, rhs: UInt32) -> UInt64 {
lhs % UInt64(rhs)
}
public static func / (lhs: Self, rhs: UInt32) -> UInt32 {
UInt32(lhs / UInt64(rhs))
}
public static func * (lhs: Self, rhs: UInt32) -> UInt64 {
lhs * UInt64(rhs)
}
public static func * (lhs: Self, rhs: Int) -> UInt64 {
lhs * UInt64(rhs)
}
}
extension UInt32 {
public var lo: UInt16 {
UInt16(self & 0xffff)
}
public var hi: UInt16 {
UInt16(self >> 16)
}
public static func + (lhs: Self, rhs: Int.IntegerLiteralType) -> UInt32 {
lhs + UInt32(rhs)
}
public static func - (lhs: Self, rhs: Int.IntegerLiteralType) -> UInt32 {
lhs - UInt32(rhs)
}
public static func / (lhs: Self, rhs: Int.IntegerLiteralType) -> UInt32 {
lhs / UInt32(rhs)
}
public static func - (lhs: Self, rhs: UInt16) -> UInt32 {
lhs - UInt32(rhs)
}
public static func * (lhs: Self, rhs: Int.IntegerLiteralType) -> Int {
Int(lhs) * rhs
}
}
extension Int {
public static func + (lhs: Self, rhs: UInt32) -> Int {
lhs + Int(rhs)
}
public static func + (lhs: Self, rhs: UInt32) -> UInt32 {
UInt32(lhs) + rhs
}
}
extension UInt16 {
func isDir() -> Bool {
self & EXT4.FileModeFlag.TypeMask.rawValue == EXT4.FileModeFlag.S_IFDIR.rawValue
}
func isLink() -> Bool {
self & EXT4.FileModeFlag.TypeMask.rawValue == EXT4.FileModeFlag.S_IFLNK.rawValue
}
func isReg() -> Bool {
self & EXT4.FileModeFlag.TypeMask.rawValue == EXT4.FileModeFlag.S_IFREG.rawValue
}
func fileType() -> UInt8 {
typealias FMode = EXT4.FileModeFlag
typealias FileType = EXT4.FileType
switch self & FMode.TypeMask.rawValue {
case FMode.S_IFREG.rawValue:
return FileType.regular.rawValue
case FMode.S_IFDIR.rawValue:
return FileType.directory.rawValue
case FMode.S_IFCHR.rawValue:
return FileType.character.rawValue
case FMode.S_IFBLK.rawValue:
return FileType.block.rawValue
case FMode.S_IFIFO.rawValue:
return FileType.fifo.rawValue
case FMode.S_IFSOCK.rawValue:
return FileType.socket.rawValue
case FMode.S_IFLNK.rawValue:
return FileType.symbolicLink.rawValue
default:
return FileType.unknown.rawValue
}
}
}
extension [UInt8] {
var allZeros: Bool {
for num in self where num != 0 {
return false
}
return true
}
}
+4
View File
@@ -0,0 +1,4 @@
# ``ContainerizationEXT4``
`ContainerizationEXT4` provides functionality to read the superblock of an existing ext4 block device and format a new block device with
the ext4 file system.
@@ -0,0 +1,78 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import CoreFoundation
// takes a pointer and converts its contents to native endian bytes
public func withUnsafeLittleEndianBytes<T, Result>(of value: T, body: (UnsafeRawBufferPointer) throws -> Result)
rethrows -> Result
{
switch Endian {
case .little:
return try withUnsafeBytes(of: value) { bytes in
try body(bytes)
}
case .big:
return try withUnsafeBytes(of: value) { buffer in
let reversedBuffer = Array(buffer.reversed())
return try reversedBuffer.withUnsafeBytes { buf in
try body(buf)
}
}
}
}
public func withUnsafeLittleEndianBuffer<T>(
of value: UnsafeRawBufferPointer, body: (UnsafeRawBufferPointer) throws -> T
) rethrows -> T {
switch Endian {
case .little:
return try body(value)
case .big:
let reversed = Array(value.reversed())
return try reversed.withUnsafeBytes { buf in
try body(buf)
}
}
}
extension UnsafeRawBufferPointer {
// loads littleEndian raw data, converts it native endian format and calls UnsafeRawBufferPointer.load
public func loadLittleEndian<T>(as type: T.Type) -> T {
switch Endian {
case .little:
return self.load(as: T.self)
case .big:
let buffer = Array(self.reversed())
return buffer.withUnsafeBytes { ptr in
ptr.load(as: T.self)
}
}
}
}
public enum Endianness {
case little
case big
}
// returns current endianness
public var Endian: Endianness {
var value: UInt32 = 0x0102_0304
return withUnsafeBytes(of: &value) { buffer in
buffer.first == 0x04 ? .little : .big
}
}
@@ -0,0 +1,176 @@
//===----------------------------------------------------------------------===//
// Copyright © 2025-2026 Apple Inc. and the Containerization project authors.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// https://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//===----------------------------------------------------------------------===//
import Foundation
/// The core error type for Containerization.
///
/// Most API surfaces for the core container/process/agent types will
/// return a ContainerizationError.
public struct ContainerizationError: Swift.Error, Sendable {
/// A code describing the error encountered.
public var code: Code
/// A description of the error.
public var message: String
/// The original error which led to this error being thrown.
public var cause: (any Error)?
/// Creates a new error.
///
/// - Parameters:
/// - code: The error code.
/// - message: A description of the error.
/// - cause: The original error which led to this error being thrown.
public init(_ code: Code, message: String, cause: (any Error)? = nil) {
self.code = code
self.message = message
self.cause = cause
}
/// Creates a new error.
///
/// - Parameters:
/// - rawCode: The error code value as a String.
/// - message: A description of the error.
/// - cause: The original error which led to this error being thrown.
public init(_ rawCode: String, message: String, cause: (any Error)? = nil) {
self.code = Code(rawValue: rawCode)
self.message = message
self.cause = cause
}
/// Provides a unique hash of the error.
public func hash(into hasher: inout Hasher) {
hasher.combine(self.code)
hasher.combine(self.message)
}
/// Equality operator for the error. Uses the code and message.
public static func == (lhs: Self, rhs: Self) -> Bool {
lhs.code == rhs.code && lhs.message == rhs.message
}
/// Checks if the given error has the provided code.
public func isCode(_ code: Code) -> Bool {
self.code == code
}
}
extension ContainerizationError: CustomStringConvertible {
/// Description of the error.
public var description: String {
guard let cause = self.cause else {
return "\(self.code): \"\(self.message)\""
}
return "\(self.code): \"\(self.message)\" (cause: \"\(cause)\")"
}
}
extension ContainerizationError: LocalizedError {
/// A localized message describing what error occurred.
public var errorDescription: String? {
guard let cause = self.cause else {
return message
}
return "\(message) (cause: \"\(cause)\")"
}
}
extension ContainerizationError {
/// Codes for a `ContainerizationError`.
public struct Code: Sendable, Hashable {
private enum Value: Hashable, Sendable, CaseIterable {
case unknown
case invalidArgument
case internalError
case exists
case notFound
case cancelled
case invalidState
case empty
case timeout
case unsupported
case interrupted
}
private var value: Value
private init(_ value: Value) {
self.value = value
}
init(rawValue: String) {
let values = Value.allCases.reduce(into: [String: Value]()) {
$0[String(describing: $1)] = $1
}
let match = values[rawValue]
guard let match else {
fatalError("invalid code value \(rawValue)")
}
self.value = match
}
public static var unknown: Self {
Self(.unknown)
}
public static var invalidArgument: Self {
Self(.invalidArgument)
}
public static var internalError: Self {
Self(.internalError)
}
public static var exists: Self {
Self(.exists)
}
public static var notFound: Self {
Self(.notFound)
}
public static var cancelled: Self {
Self(.cancelled)
}
public static var invalidState: Self {
Self(.invalidState)
}
public static var empty: Self {
Self(.empty)
}
public static var timeout: Self {
Self(.timeout)
}
public static var unsupported: Self {
Self(.unsupported)
}
public static var interrupted: Self {
Self(.interrupted)
}
}
}
extension ContainerizationError.Code: CustomStringConvertible {
public var description: String {
String(describing: self.value)
}
}

Some files were not shown because too many files have changed in this diff Show More