diff --git a/.github/workflows/build-trigger.yml b/.github/workflows/build-trigger.yml index d95d866..363fded 100644 --- a/.github/workflows/build-trigger.yml +++ b/.github/workflows/build-trigger.yml @@ -32,6 +32,7 @@ jobs: - 'VERSION' dockerfile: - 'Dockerfile' + - 'Dockerfile.freebsd' c: - 'Makefile' - 'main.c' diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index 794a3b4..01a41da 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -93,8 +93,68 @@ jobs: -a "ref=${{github.sha}}" \ -a "author=Nubificus LTD" + build-freebsd: + # Cross-compiled on a Linux amd64 runner (see Dockerfile.freebsd); the + # resulting image is stamped os=freebsd and amended into the manifest list + # below. Required: a FreeBSD build failure blocks the pipeline. + runs-on: ${{ format('{0}-{1}', join(fromJSON(inputs.runner), '-'), 'amd64') }} + permissions: + id-token: write # to complete the identity challenge with sigstore/fulcio when running outside of PRs + env: + IMAGE_NAME: ${{ inputs.registry }}/${{ github.repository }} + steps: + - name: Checkout the repo + uses: actions/checkout@8e8c483db84b4bee98b60c0593521ed34d9990e8 # v6.0.1 + + - name: Set short SHA + run: echo "SHA_SHORT=${GITHUB_SHA::7}" >> $GITHUB_ENV + + - name: Set up Docker Buildx + uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3.12.0 + + - name: Log into registry ${{ inputs.registry }} + uses: docker/login-action@5e57cd118135c172c3672efd75eb46360885c0ef + with: + registry: ${{ inputs.registry }} + username: ${{ secrets.harbor_user }} + password: ${{ secrets.harbor_secret }} + + - name: Extract Docker metadata + id: meta + uses: docker/metadata-action@318604b99e75e41977312d83839a89be02ca4893 + with: + images: ${{ env.IMAGE_NAME }} + tags: | + type=sha,prefix=freebsd-amd64- + + - name: Build and push urunit FreeBSD image + id: build-and-push + uses: docker/build-push-action@9e436ba9f2d7bcd1d038c8e55d039d37896ddc5d # master + with: + context: . + file: "Dockerfile.freebsd" + platforms: freebsd/amd64 + tags: ${{ steps.meta.outputs.tags }} + labels: ${{ steps.meta.outputs.labels }} + push: true + provenance: false + + - name: Install cosign + uses: sigstore/cosign-installer@d58896d6a1865668819e1d91763c7751a165e159 # main + + - name: Sign the published Docker image + env: + COSIGN_EXPERIMENTAL: "true" + DIGEST: ${{steps.build-and-push.outputs.digest}} + run: | + cosign sign --yes ${{ env.IMAGE_NAME }}@$DIGEST \ + -a "repo=${{github.repository}}" \ + -a "workflow=${{github.workflow}}" \ + -a "ref=${{github.sha}}" \ + -a "author=Nubificus LTD" + manifest: - needs: [build-all] + needs: [build-all, build-freebsd] runs-on: base-dind-2204-amd64 permissions: id-token: write # to complete the identity challenge with sigstore/fulcio when running outside of PRs @@ -140,6 +200,10 @@ jobs: amend_command+=" --amend ${{ env.IMAGE_NAME }}:$arch-${{ env.SHA_SHORT }}" done + # Include the FreeBSD image (build-freebsd is a required dependency, + # so if we got here it was built and pushed). + amend_command+=" --amend ${{ env.IMAGE_NAME }}:freebsd-amd64-${{ env.SHA_SHORT }}" + echo "-------------------- Amend command constructed -------------------" echo "$amend_command" diff --git a/Dockerfile.freebsd b/Dockerfile.freebsd new file mode 100644 index 0000000..016c791 --- /dev/null +++ b/Dockerfile.freebsd @@ -0,0 +1,38 @@ +# Builds the FreeBSD static urunit and packages it into a scratch image. +# +# The build runs on a Linux builder and cross-compiles for FreeBSD with clang + +# lld against a sysroot extracted from the FreeBSD base distribution. The final +# scratch stage only copies the binary, so nothing FreeBSD is ever executed at +# build time. Stamp the image OS with buildx, e.g.: +# +# docker buildx build --platform freebsd/amd64 -f Dockerfile.freebsd . +# +# which sets os=freebsd in the resulting image config. + +FROM --platform=$BUILDPLATFORM alpine AS builder + +ARG FREEBSD_VERSION=14.3-RELEASE +ARG TARGET_TRIPLE=x86_64-unknown-freebsd14.3 + +WORKDIR /urunit + +RUN apk add --no-cache clang lld make curl tar xz + +# FreeBSD sysroot: headers plus the (static) libraries and startup objects from +# the base distribution. Only the parts a sysroot needs are extracted. +RUN mkdir -p /sysroot && \ + curl -fsSL "https://download.freebsd.org/releases/amd64/amd64/${FREEBSD_VERSION}/base.txz" \ + | tar -xJf - -C /sysroot ./lib ./usr/lib ./usr/include + +COPY Makefile . +COPY src src + +# TARGET_OS must be explicit: the Makefile would otherwise infer the OS from the +# Linux builder's uname. Everything the cross toolchain needs rides on CC. +RUN make TARGET_OS=FreeBSD \ + CC="clang --target=${TARGET_TRIPLE} --sysroot=/sysroot -fuse-ld=lld" \ + static + +FROM scratch +COPY --from=builder /urunit/dist/urunit_static /urunit +ENTRYPOINT ["/urunit"] diff --git a/Makefile b/Makefile index 23e3800..b55d777 100644 --- a/Makefile +++ b/Makefile @@ -77,6 +77,9 @@ endif ifeq ($(TARGET_OS),Linux) URUNIT_SRC += ${SOURCE_DIR}/linux.c endif +ifeq ($(TARGET_OS),FreeBSD) + URUNIT_SRC += ${SOURCE_DIR}/freebsd.c +endif # Main Building rules # diff --git a/README.md b/README.md index a38e61a..7f87e52 100644 --- a/README.md +++ b/README.md @@ -70,8 +70,12 @@ following information: - The list of the environment variables to set for the application - The configuration for the process execution environment. +- The application command to execute, when `urunit` is started without a + command line (optional). - The list of mounts of block devices. Each block device is defined by its serial id and it will get mounted in the defined mountpoint. +- The network configuration (IP address, gateway and netmask) for guests + that can not get it from the kernel command line (optional). The file can be specified to `urunit` setting the `URUNIT_CONFIG` environment variable with the path to the configuration file. @@ -86,14 +90,32 @@ UCS UID: GID: WD: +ARC: +ARV: +... UCE UBS ID: MP: ... UBE +UNS +IP: +GW: +MSK: +UNE ``` +Inside the `UCS` section, the application command is optional: `ARC` holds the +number of arguments and is followed by exactly that many `ARV` lines, one per +argument, each taken verbatim. It is used when `urunit` is started without a +command line, in which case the command from the configuration is executed. + +The `UNS` section is optional and provides the network configuration for +guests that can not get it from the kernel command line: `IP` is the IPv4 +address, `GW` the gateway and `MSK` the netmask. An empty value (e.g. `IP:`) +leaves the field unset. + ## Installation Using one of the following methods, we can install `urunit` either in a diff --git a/src/common.h b/src/common.h index 4c48ba1..55ab3cb 100644 --- a/src/common.h +++ b/src/common.h @@ -33,7 +33,7 @@ #define DEBUG_PRINTF(fmt, ...) \ do { if (SHOW_DEBUG) fprintf(stderr, "[DEBUG] " fmt, __VA_ARGS__); } while (0) -#define DEBUG_PRINT(fmt, ...) \ +#define DEBUG_PRINT(fmt) \ do { if (SHOW_DEBUG) fprintf(stderr, "[DEBUG] " fmt); } while (0) struct block_config { @@ -41,6 +41,12 @@ struct block_config { char *mountpoint; }; +struct net_config { + char *ip; + char *gateway; + char *mask; +}; + int ensure_dir(const char *path); int mkdir_all(const char *path, mode_t mode, char *first_dir); int rm_empty_dirs(const char *dir, const char *top_dir); @@ -49,14 +55,20 @@ int is_block_fs(const char *fs_type); int is_network_fs(const char *fs_type); int is_cloud_storage_fs(const char *fs_type); +// Platform specific functions +int setup_console(void); +int remount_root_rw(void); +char *get_boot_var(const char *name); +char *read_raw_device(int fd, size_t *size); +int configure_network(struct net_config *net); int read_block_dev_serial(const char *device_name, char *serial, const size_t size); int find_vblock_device_by_order(const uint32_t n, char *device_path); int find_vblock_device_by_serial(const char *target_serial, char *device_path); -int mount_special_fs(); +int mount_special_fs(void); int mount_block_vols(struct block_config **vols); -int set_default_route(); -void unmount_external(); -int set_subreaper(); -void request_reboot(); +int set_default_route(void); +void unmount_external(void); +int set_subreaper(void); +void request_reboot(void); #endif // COMMON_H diff --git a/src/freebsd.c b/src/freebsd.c new file mode 100644 index 0000000..cfe52d1 --- /dev/null +++ b/src/freebsd.c @@ -0,0 +1,741 @@ +// Copyright (c) 2023-2026, Nubificus LTD +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// FreeBSD implementation of the platform specific parts of urunit. +// +// Differences from Linux that shape this file: +// - The kernel starts init (us) without arguments and without open file +// descriptors. The application and its arguments come from the urunit +// configuration and the console has to be opened by hand. +// - Boot arguments are kernel environment variables (kenv(2)), not process +// environment variables. URUNIT_CONFIG and URUNIT_DEFROUTE are read from +// there when they are not in the environment. +// - There is no ip= boot parameter, so the network configuration is part of +// the urunit configuration (UNS/UNE section) and applied here. +// - The urunit configuration is handed over as a raw virtio block device +// (disks are character devices with st_size == 0, so their size comes +// from DIOCGMEDIASIZE). +// - Virtio block devices are /dev/vtbdN and, when the device has a serial, +// /dev/diskid/DISK- (GEOM_LABEL). + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "common.h" + +#define VTNET_IF "vtnet0" +#define CONSOLE_DEV "/dev/console" +#define CONFIG_MAX_SZ (1024 * 1024) +#define DEFAULT_MASK "255.255.255.0" + +// setup_console: PID 1 started by the FreeBSD kernel has no open file +// descriptors and no controlling terminal. Become a session leader, open the +// console and use it for stdin/stdout/stderr. When we already have a stdin +// (e.g. started from a shell) nothing is done. +// +// Return value: +// On success 0 is returned. Otherwise 1 is returned. +int setup_console(void) { + struct stat st; + int fd; + int ret; + + ret = fstat(STDIN_FILENO, &st); + if (ret == 0) { + DEBUG_PRINT("stdin is open, keeping the current console\n"); + return 0; + } + + // PID 1 is not a session leader; ignore EPERM if we already are one. + ret = setsid(); + if (ret < 0 && errno != EPERM) { + perror("setsid"); + } + + fd = open(CONSOLE_DEV, O_RDWR); + if (fd < 0) { + // Nothing we can print to. Try to continue silently. + return 1; + } + // Make the console our controlling terminal so that the application + // gets job control and signals (^C) from the serial console. + ret = ioctl(fd, TIOCSCTTY, 0); + if (ret < 0) { + perror("TIOCSCTTY"); + } + ret = dup2(fd, STDIN_FILENO) < 0 || dup2(fd, STDOUT_FILENO) < 0 || + dup2(fd, STDERR_FILENO) < 0; + if (ret) { + perror("dup2 console"); + close(fd); + return 1; + } + if (fd > STDERR_FILENO) { + close(fd); + } + + return 0; +} + +// get_boot_var: Returns the value of a kernel environment variable, which is +// how boot arguments (name=value) reach the guest on the PVH boot path. +// +// Arguments: +// 1. name: The name of the variable +// +// Return value: +// On success a dynamically allocated string with the value is returned. +// If the variable does not exist, NULL is returned. +char *get_boot_var(const char *name) { + char buf[KENV_MVALLEN + 1]; + int ret; + + ret = kenv(KENV_GET, name, buf, sizeof(buf) - 1); + if (ret < 0) { + return NULL; + } + buf[sizeof(buf) - 1] = '\0'; + DEBUG_PRINTF("kenv %s=%s\n", name, buf); + + return strdup(buf); +} + +// read_raw_device: Reads the contents of a raw disk device (the urunit +// configuration is attached as a tiny virtio block device). The device is read +// in sector-sized chunks and the buffer is NUL terminated. The caller is +// responsible to free the returned buffer. +// +// Arguments: +// 1. fd: An open file descriptor of the device +// 2. size: Will hold the amount of bytes read +// +// Return value: +// On success a buffer with the contents of the device is returned. +// On failure, NULL is returned. +char *read_raw_device(int fd, size_t *size) { + off_t media_size = 0; + u_int sector_size = DEV_BSIZE; + char *buffer = NULL; + off_t off = 0; + int ret = 0; + + ret = ioctl(fd, DIOCGMEDIASIZE, &media_size); + if (ret < 0) { + perror("DIOCGMEDIASIZE"); + return NULL; + } + ret = ioctl(fd, DIOCGSECTORSIZE, §or_size); + if (ret < 0 || sector_size == 0) { + sector_size = DEV_BSIZE; + } + if (media_size <= 0) { + fprintf(stderr, "Configuration device is empty\n"); + return NULL; + } + if (media_size > CONFIG_MAX_SZ) { + media_size = CONFIG_MAX_SZ; + } + // Raw device I/O has to be a multiple of the sector size. + media_size -= media_size % sector_size; + DEBUG_PRINTF("Reading %jd bytes from configuration device (sector %u)\n", + (intmax_t)media_size, sector_size); + + buffer = malloc(media_size + 1); + if (!buffer) { + fprintf(stderr, "Failed to allocate memory for the configuration device\n"); + return NULL; + } + + while (off < media_size) { + ssize_t n = pread(fd, buffer + off, media_size - off, off); + if (n < 0) { + if (errno == EINTR) { + continue; + } + perror("read configuration device"); + free(buffer); + return NULL; + } + if (n == 0) { + break; + } + off += n; + } + buffer[off] = '\0'; + *size = (size_t)off; + + return buffer; +} + +// remount_root_rw: The kernel always mounts the root filesystem read-only +// (vfs.root.mountfrom.options=rw is ignored by mountroot); rc(8) normally +// remounts it read-write. Do the same for the application. +// +// Return value: +// On success 0 is returned. Otherwise a non-zero value is returned. +int remount_root_rw(void) { + struct statfs sfs; + struct iovec iov[10]; + int ret = 0; + + ret = statfs("/", &sfs); + if (ret != 0) { + perror("statfs /"); + return 1; + } + if ((sfs.f_flags & MNT_RDONLY) == 0) { + return 0; + } + DEBUG_PRINTF("Remounting / (%s from %s) read-write\n", sfs.f_fstypename, sfs.f_mntfromname); + iov[0].iov_base = (void *)"fstype"; + iov[0].iov_len = sizeof("fstype"); + iov[1].iov_base = sfs.f_fstypename; + iov[1].iov_len = strlen(sfs.f_fstypename) + 1; + iov[2].iov_base = (void *)"fspath"; + iov[2].iov_len = sizeof("fspath"); + iov[3].iov_base = (void *)"/"; + iov[3].iov_len = sizeof("/"); + iov[4].iov_base = (void *)"from"; + iov[4].iov_len = sizeof("from"); + iov[5].iov_base = sfs.f_mntfromname; + iov[5].iov_len = strlen(sfs.f_mntfromname) + 1; + // Flag options are name/value pairs with an empty value. + iov[6].iov_base = (void *)"update"; + iov[6].iov_len = sizeof("update"); + iov[7].iov_base = NULL; + iov[7].iov_len = 0; + iov[8].iov_base = (void *)"rw"; + iov[8].iov_len = sizeof("rw"); + iov[9].iov_base = NULL; + iov[9].iov_len = 0; + ret = nmount(iov, 10, MNT_UPDATE); + if (ret != 0) { + perror("remount / read-write"); + return 1; + } + + return 0; +} + +// mount_special_fs: devfs is mounted on /dev by the kernel. Mount a tmpfs on +// /tmp, and procfs on /proc if the directory exists. Neither failure is fatal. +// +// Return value: +// It always returns 0. +int mount_special_fs(void) { + struct iovec iov[6]; + struct stat st; + int ret = 0; + + // A writable /tmp is expected by most applications. urunit does not + // itself depend on it, so a failure here is not fatal. + ret = ensure_dir("/tmp"); + if (ret == 0) { + iov[0].iov_base = (void *)"fstype"; + iov[0].iov_len = sizeof("fstype"); + iov[1].iov_base = (void *)"tmpfs"; + iov[1].iov_len = sizeof("tmpfs"); + iov[2].iov_base = (void *)"fspath"; + iov[2].iov_len = sizeof("fspath"); + iov[3].iov_base = (void *)"/tmp"; + iov[3].iov_len = sizeof("/tmp"); + iov[4].iov_base = (void *)"from"; + iov[4].iov_len = sizeof("from"); + iov[5].iov_base = (void *)"tmpfs"; + iov[5].iov_len = sizeof("tmpfs"); + ret = nmount(iov, 6, 0); + if (ret < 0) { + DEBUG_PRINTF("mount /tmp: %s\n", strerror(errno)); + } else { + // tmpfs takes the mode of the covered directory; make /tmp + // world-writable with the sticky bit like a normal /tmp. + ret = chmod("/tmp", 01777); + if (ret < 0) { + perror("chmod /tmp"); + } + } + } + + // procfs is optional and only mounted when /proc exists. + ret = stat("/proc", &st); + if (ret != 0) { + return 0; + } + iov[0].iov_base = (void *)"fstype"; + iov[0].iov_len = sizeof("fstype"); + iov[1].iov_base = (void *)"procfs"; + iov[1].iov_len = sizeof("procfs"); + iov[2].iov_base = (void *)"fspath"; + iov[2].iov_len = sizeof("fspath"); + iov[3].iov_base = (void *)"/proc"; + iov[3].iov_len = sizeof("/proc"); + iov[4].iov_base = (void *)"from"; + iov[4].iov_len = sizeof("from"); + iov[5].iov_base = (void *)"procfs"; + iov[5].iov_len = sizeof("procfs"); + ret = nmount(iov, 6, 0); + if (ret < 0 && errno != EBUSY) { + DEBUG_PRINTF("mount /proc: %s\n", strerror(errno)); + } + + return 0; +} + +// find_vblock_device_by_order: Returns the nth (zero-based) virtio block +// device (/dev/vtbdN) if it exists. +// +// Arguments: +// 1. n: Zero-based index of the virtio block device to return. +// 2. device_path: The buffer that will store the path to the block device. +// +// Return value: +// On success 0 is returned and device_path holds the device path. +// Otherwise -1 is returned. +int find_vblock_device_by_order(const uint32_t n, char *device_path) { + char device_name[PATH_MAX]; + int len = 0; + int ret = 0; + + len = snprintf(device_name, sizeof(device_name), "/dev/vtbd%u", n); + if (len < 0 || (size_t)len >= sizeof(device_name)) { + return -1; + } + ret = access(device_name, F_OK); + if (ret != 0) { + return -1; + } + memcpy(device_path, device_name, (size_t)len + 1); + + return 0; +} + +// find_vblock_device_by_serial: Finds a disk by its serial id through the +// GEOM_LABEL diskid provider (/dev/diskid/DISK-). +// +// Arguments: +// 1. target_serial: The serial ID to search for +// 2. device_path: The buffer that will store the path to the block device +// +// Return value: +// On success 0 is returned and device_path holds the device path. +// Otherwise -1 is returned. +int find_vblock_device_by_serial(const char *target_serial, char *device_path) { + char device_name[PATH_MAX]; + int len = 0; + int ret = 0; + + len = snprintf(device_name, sizeof(device_name), "/dev/diskid/DISK-%s", target_serial); + if (len < 0 || (size_t)len >= sizeof(device_name)) { + return -1; + } + ret = access(device_name, F_OK); + if (ret != 0) { + return -1; + } + memcpy(device_path, device_name, (size_t)len + 1); + + return 0; +} + +// do_mount: Mounts a filesystem with nmount(2). +static int do_mount(const char *fstype, const char *from, const char *fspath, int flags) { + struct iovec iov[6]; + + iov[0].iov_base = (void *)"fstype"; + iov[0].iov_len = sizeof("fstype"); + iov[1].iov_base = (void *)fstype; + iov[1].iov_len = strlen(fstype) + 1; + iov[2].iov_base = (void *)"fspath"; + iov[2].iov_len = sizeof("fspath"); + iov[3].iov_base = (void *)fspath; + iov[3].iov_len = strlen(fspath) + 1; + iov[4].iov_base = (void *)"from"; + iov[4].iov_len = sizeof("from"); + iov[5].iov_base = (void *)from; + iov[5].iov_len = strlen(from) + 1; + + return nmount(iov, 6, flags); +} + +// mount_block_vols: Mounts all block devices using their info from the +// block_config parameter. Volumes are tried as ext2fs first (urunc creates +// ext2/ext4 block volumes on the host) and as ufs afterwards. +// +// Arguments: +// 1. vols: An array of struct block_config with information to mount +// block volumes +// +// Return value: +// On success 0 is returned. +// Otherwise 1 is returned. +int mount_block_vols(struct block_config **vols) { + struct block_config **iter_bc = NULL; + char first_new_dir[PATH_MAX] = { 0 }; + uint32_t blk_count = 0; + + if (vols == NULL) { + DEBUG_PRINT("No block volumes to mount, nothing to do\n"); + return 0; + } + + for (iter_bc = vols; *iter_bc != NULL; iter_bc++) { + struct block_config *tmp_bc = *iter_bc; + char block_dev[PATH_MAX] = { 0 }; + int ret = 0; + + blk_count++; + first_new_dir[0] = '\0'; + DEBUG_PRINTF("Searching block device with serial ID %s\n", tmp_bc->id); + // Firecracker has no drive serials: the volumes follow the rootfs + // in the order they were attached. + if (strlen(tmp_bc->id) > 2 && tmp_bc->id[0] == 'F' && tmp_bc->id[1] == 'C') { + ret = find_vblock_device_by_order(blk_count, block_dev); + } else { + ret = find_vblock_device_by_serial(tmp_bc->id, block_dev); + } + if (ret) { + fprintf(stderr, "Could not find any virtio block device with serial ID %s\n", tmp_bc->id); + continue; + } + DEBUG_PRINTF("Found device %s\n", block_dev); + DEBUG_PRINTF("Setup the mountpoint %s\n", tmp_bc->mountpoint); + ret = mkdir_all(tmp_bc->mountpoint, 0755, first_new_dir); + if (ret != 0 ) { + fprintf(stderr, "Failed to create %s\n",tmp_bc->mountpoint); + continue; + } + DEBUG_PRINT("Mount device as ext2fs\n"); + ret = do_mount("ext2fs", block_dev, tmp_bc->mountpoint, 0); + if (ret != 0) { + DEBUG_PRINT("Mount device as ufs\n"); + ret = do_mount("ufs", block_dev, tmp_bc->mountpoint, 0); + } + if (ret != 0) { + perror("mount"); + if (first_new_dir[0] != '\0') { + ret = rm_empty_dirs(tmp_bc->mountpoint, first_new_dir); + if (ret < 0) { + fprintf(stderr, "WARNING: Could not remove %s and its subdirs\n", tmp_bc->mountpoint); + } + } + } + } + + return 0; +} + +#define RT_ROUNDUP(a) \ + ((a) > 0 ? (1 + (((a) - 1) | (sizeof(long) - 1))) : sizeof(long)) + +// rt_add: Adds a route through the routing socket. +// +// Arguments: +// 1. dst: The destination +// 2. gw: The gateway (an IPv4 address or the link address of an interface) +// 3. mask: The netmask, or NULL for a host route +// 4. flags: RTF_* flags +// +// Return value: +// On success 0 is returned. Otherwise -1 is returned. +static int rt_add(struct sockaddr *dst, struct sockaddr *gw, struct sockaddr *mask, int flags) { + struct { + struct rt_msghdr hdr; + char space[512]; + } msg; + char *cp = msg.space; + int sock; + ssize_t n; + + memset(&msg, 0, sizeof(msg)); + msg.hdr.rtm_type = RTM_ADD; + msg.hdr.rtm_version = RTM_VERSION; + msg.hdr.rtm_flags = flags; + msg.hdr.rtm_seq = 1; + msg.hdr.rtm_addrs = RTA_DST | RTA_GATEWAY; + + memcpy(cp, dst, dst->sa_len); + cp += RT_ROUNDUP(dst->sa_len); + memcpy(cp, gw, gw->sa_len); + cp += RT_ROUNDUP(gw->sa_len); + if (mask) { + msg.hdr.rtm_addrs |= RTA_NETMASK; + memcpy(cp, mask, mask->sa_len); + cp += RT_ROUNDUP(mask->sa_len); + } + msg.hdr.rtm_msglen = (u_short)(cp - (char *)&msg); + + sock = socket(PF_ROUTE, SOCK_RAW, AF_INET); + if (sock < 0) { + perror("routing socket"); + return -1; + } + n = write(sock, &msg, msg.hdr.rtm_msglen); + close(sock); + if (n < 0) { + if (errno == EEXIST) { + return 0; + } + perror("RTM_ADD"); + return -1; + } + + return 0; +} + +// get_link_addr: Returns the AF_LINK address of an interface. +static int get_link_addr(const char *ifname, struct sockaddr_dl *sdl) { + struct ifaddrs *ifap = NULL, *ifa = NULL; + int found = 0; + int ret = 0; + + ret = getifaddrs(&ifap); + if (ret != 0) { + perror("getifaddrs"); + return -1; + } + for (ifa = ifap; ifa != NULL; ifa = ifa->ifa_next) { + size_t len = 0; + + if (ifa->ifa_addr == NULL || ifa->ifa_addr->sa_family != AF_LINK) { + continue; + } + if (strcmp(ifa->ifa_name, ifname) != 0) { + continue; + } + // The kernel-supplied AF_LINK sockaddr can be larger than our + // fixed struct (56 vs 54 bytes on 64-bit FreeBSD). Clamp the copy + // to sizeof(*sdl) so we do not overflow it, and cap sdl_len to the + // bytes we actually hold so rt_add does not later over-read it. + len = ifa->ifa_addr->sa_len; + if (len > sizeof(*sdl)) { + len = sizeof(*sdl); + } + memcpy(sdl, ifa->ifa_addr, len); + sdl->sdl_len = (u_char)len; + found = 1; + break; + } + freeifaddrs(ifap); + + return found ? 0 : -1; +} + +static void sin_init(struct sockaddr_in *sin, in_addr_t addr) { + memset(sin, 0, sizeof(*sin)); + sin->sin_len = sizeof(*sin); + sin->sin_family = AF_INET; + sin->sin_addr.s_addr = addr; +} + +// configure_network: Configures vtnet0 with the address from the urunit +// configuration and installs the default route. If the gateway is not in the +// interface subnet (e.g. Kubernetes CNI setups), a host route to the gateway +// through the interface is added first. +// +// Arguments: +// 1. net: The network configuration (IP, gateway, mask) +// +// Return value: +// On success 0 is returned. Otherwise a non-zero value is returned. +int configure_network(struct net_config *net) { + struct in_aliasreq ifra; + struct ifreq ifr; + struct in_addr ip, mask, gw; + int sock; + int ret = 0; + + if (net == NULL || net->ip == NULL || net->ip[0] == '\0') { + DEBUG_PRINT("No network configuration\n"); + return 0; + } + + ret = inet_pton(AF_INET, net->ip, &ip); + if (ret != 1) { + fprintf(stderr, "Invalid IP address %s\n", net->ip); + return 1; + } + if (net->mask == NULL || net->mask[0] == '\0' || + inet_pton(AF_INET, net->mask, &mask) != 1) { + DEBUG_PRINTF("Invalid or empty netmask, using %s\n", DEFAULT_MASK); + inet_pton(AF_INET, DEFAULT_MASK, &mask); + } + + sock = socket(AF_INET, SOCK_DGRAM, 0); + if (sock < 0) { + perror("socket"); + return 1; + } + + memset(&ifra, 0, sizeof(ifra)); + strlcpy(ifra.ifra_name, VTNET_IF, sizeof(ifra.ifra_name)); + sin_init(&ifra.ifra_addr, ip.s_addr); + sin_init(&ifra.ifra_mask, mask.s_addr); + sin_init(&ifra.ifra_broadaddr, (ip.s_addr & mask.s_addr) | ~mask.s_addr); + DEBUG_PRINTF("Setting %s address %s mask %s\n", VTNET_IF, net->ip, inet_ntoa(mask)); + ret = ioctl(sock, SIOCAIFADDR, &ifra); + if (ret < 0) { + perror("SIOCAIFADDR"); + close(sock); + return 1; + } + + memset(&ifr, 0, sizeof(ifr)); + strlcpy(ifr.ifr_name, VTNET_IF, sizeof(ifr.ifr_name)); + ret = ioctl(sock, SIOCGIFFLAGS, &ifr); + if (ret == 0) { + ifr.ifr_flags |= IFF_UP; + ret = ioctl(sock, SIOCSIFFLAGS, &ifr); + if (ret < 0) { + perror("SIOCSIFFLAGS"); + ret = 1; + } + } else { + perror("SIOCGIFFLAGS"); + ret = 1; + } + close(sock); + + if (net->gateway == NULL || net->gateway[0] == '\0') { + DEBUG_PRINT("No gateway, skipping default route\n"); + return ret; + } + if (inet_pton(AF_INET, net->gateway, &gw) != 1) { + fprintf(stderr, "Invalid gateway address %s\n", net->gateway); + return 1; + } + + if ((gw.s_addr & mask.s_addr) != (ip.s_addr & mask.s_addr)) { + // The gateway is not directly reachable through the subnet: + // add a host route to it through the interface (link address). + struct sockaddr_in dst; + struct sockaddr_dl sdl; + + DEBUG_PRINTF("Gateway %s is outside the subnet, adding a host route\n", net->gateway); + if (get_link_addr(VTNET_IF, &sdl) != 0) { + fprintf(stderr, "Could not find the link address of %s\n", VTNET_IF); + return 1; + } + sin_init(&dst, gw.s_addr); + if (rt_add((struct sockaddr *)&dst, (struct sockaddr *)&sdl, NULL, + RTF_UP | RTF_HOST | RTF_STATIC) != 0) { + return 1; + } + } + + { + struct sockaddr_in dst, gateway, netmask; + + DEBUG_PRINTF("Setting default route via %s\n", net->gateway); + sin_init(&dst, INADDR_ANY); + sin_init(&gateway, gw.s_addr); + sin_init(&netmask, INADDR_ANY); + if (rt_add((struct sockaddr *)&dst, (struct sockaddr *)&gateway, + (struct sockaddr *)&netmask, + RTF_UP | RTF_GATEWAY | RTF_STATIC) != 0) { + return 1; + } + } + + return ret; +} + +// set_default_route: On FreeBSD the default route is installed by +// configure_network (the gateway is part of the urunit configuration). +// +// Return value: +// It always returns 0. +int set_default_route(void) { + return 0; +} + +// unmount_external: Unmounts all block based filesystems except the root. +void unmount_external(void) { + struct statfs *mntbuf = NULL; + int count = 0; + int i = 0; + int ret = 0; + + count = getmntinfo(&mntbuf, MNT_NOWAIT); + if (count <= 0) { + perror("getmntinfo"); + return; + } + // Walk backwards so that nested mounts go first. + for (i = count - 1; i >= 0; i--) { + const char *type = mntbuf[i].f_fstypename; + const char *path = mntbuf[i].f_mntonname; + + if (strcmp(path, "/") == 0) { + continue; + } + if (strcmp(type, "ufs") == 0 || strcmp(type, "ext2fs") == 0 || + strcmp(type, "msdosfs") == 0 || strcmp(type, "cd9660") == 0 || + is_block_fs(type) || is_network_fs(type) || is_cloud_storage_fs(type)) { + DEBUG_PRINTF("Trying to unmount %s (%s)\n", path, type); + ret = unmount(path, MNT_FORCE); + if (ret != 0) { + perror("unmount"); + } + } + } +} + +// set_subreaper: PID 1 is the reaper of the system by default; for any other +// PID acquire the reaper role of our subtree. +int set_subreaper(void) { + int ret; + + ret = procctl(P_PID, getpid(), PROC_REAP_ACQUIRE, NULL); + if (ret < 0) { + if (errno == EBUSY) { + return 0; + } + return -1; + } + + return 0; +} + +// request_reboot: Resets the VM. Firecracker terminates when the guest +// resets; QEMU does the same when started with -no-reboot. +void request_reboot(void) { + sync(); + reboot(RB_AUTOBOOT); +} diff --git a/src/linux.c b/src/linux.c index d8b822d..bb93094 100644 --- a/src/linux.c +++ b/src/linux.c @@ -12,10 +12,13 @@ // See the License for the specific language governing permissions and // limitations under the License. -#include +#include #include #include #include +#include +#include +#include #include #include @@ -146,14 +149,14 @@ int find_vblock_device_by_serial(const char *target_serial, char *device_path) { } // mount_special_fs: Mounts the special filesystems procfs and sysfs in /proc and -// /sys respectively. +// /sys respectively, plus a tmpfs on /tmp. // // Arguments: // No arguments. // // Return value: // It returns 0 in success. Otherwise it returns 1. -int mount_special_fs() { +int mount_special_fs(void) { int ret = 0; ret = ensure_dir("/proc"); @@ -176,6 +179,16 @@ int mount_special_fs() { return 1; } + // A writable /tmp is expected by most applications. Unlike /proc and + // /sys, urunit does not itself depend on it, so a failure is not fatal. + ret = ensure_dir("/tmp"); + if (ret == 0) { + ret = mount("tmpfs", "/tmp", "tmpfs", MS_NOSUID|MS_NODEV, "mode=1777"); + if (ret < 0) { + perror("mount /tmp"); + } + } + return 0; } @@ -250,7 +263,7 @@ int mount_block_vols(struct block_config **vols) { // // Return value: // On success 0 is returned. Otherwise a non-zero value is returned. -int set_default_route() { +int set_default_route(void) { int sockfd; struct rtentry rt; struct sockaddr_in addr; @@ -298,7 +311,7 @@ int set_default_route() { // Arguments: // // Return value: -void unmount_external() { +void unmount_external(void) { FILE *mount_info_f = NULL; char line[1024] = { 0 }; @@ -363,11 +376,85 @@ void unmount_external() { fclose(mount_info_f); } -int set_subreaper() { +int set_subreaper(void) { return prctl(PR_SET_CHILD_SUBREAPER, 1, 0, 0, 0); } -void request_reboot() { +// setup_console: On Linux the kernel opens the console for init. +int setup_console(void) { + return 0; +} + +// remount_root_rw: On Linux the root filesystem is already writable by the time +// urunit runs (urunc/the kernel set it up), so there is nothing to do here. +int remount_root_rw(void) { + return 0; +} + +// get_boot_var: On Linux, name=value boot parameters reach init as +// environment variables, so there is nothing else to look at. +char *get_boot_var(const char *name) { + (void)name; // just to suppress the unused warning + return NULL; +} + +// read_raw_device: Reads the whole contents of a block device. The caller is +// responsible to free the returned buffer. +char *read_raw_device(int fd, size_t *size) { + uint64_t dev_size = 0; + char *buffer = NULL; + size_t off = 0; + int ret = 0; + // BLKGETSIZE64 does not fit in an int. glibc's ioctl takes an unsigned + // long request, but musl's takes an int, so passing the constant + // directly overflows under -Werror there. Hold it in an unsigned long + // so the (musl) narrowing is a runtime conversion, not a constant one. + unsigned long request = BLKGETSIZE64; + + ret = ioctl(fd, request, &dev_size); + if (ret < 0) { + perror("BLKGETSIZE64"); + return NULL; + } + if (dev_size == 0 || dev_size > (1024 * 1024)) { + fprintf(stderr, "Unexpected configuration device size %llu\n", + (unsigned long long)dev_size); + return NULL; + } + buffer = malloc(dev_size + 1); + if (!buffer) { + fprintf(stderr, "Failed to allocate memory for the configuration device\n"); + return NULL; + } + while (off < dev_size) { + ssize_t n = 0; + + n = pread(fd, buffer + off, dev_size - off, (off_t)off); + if (n < 0) { + if (errno == EINTR) + continue; + perror("read configuration device"); + free(buffer); + return NULL; + } + if (n == 0) + break; + off += (size_t)n; + } + buffer[off] = '\0'; + *size = off; + + return buffer; +} + +// configure_network: On Linux the network is configured by the kernel +// through the ip= boot parameter. +int configure_network(struct net_config *net) { + (void)net; + return 0; +} + +void request_reboot(void) { syscall(SYS_reboot, LINUX_REBOOT_MAGIC1, LINUX_REBOOT_MAGIC2, LINUX_REBOOT_CMD_RESTART, NULL); } diff --git a/src/main.c b/src/main.c index 19b0a37..5f5d5f8 100644 --- a/src/main.c +++ b/src/main.c @@ -43,6 +43,8 @@ #include #include +#include +#include #include #include #include @@ -57,6 +59,8 @@ struct process_config { uint32_t uid; uint32_t gid; char *wdir; + uint32_t argc; + char **argv; }; struct app_exec_config { @@ -64,6 +68,7 @@ struct app_exec_config { char *path_env; struct process_config *pr_conf; struct block_config **blk_conf; + struct net_config *net_conf; }; extern char **environ; @@ -124,14 +129,19 @@ int isolate_child(void) { // 2. sz: The amount of bytes to read // // Return value: -// On success it returns a buffer of sz size with all bytes read. +// On success it returns a buffer of sz + 1 bytes: the sz bytes read from the +// file followed by a trailing NUL, so callers can treat it as a string (the +// same contract as the raw-device path). The caller frees it. // On failure, it returns NULL. char *read_exact_size(FILE *f, size_t sz) { size_t total_read = 0; size_t bytes_read = 0; char *buffer = NULL; - buffer = malloc(sz); + // Allocate one extra byte for a trailing NUL so the buffer is a valid + // C string; the section parsers (strtok) and the memcmp probes in + // get_config_from_file rely on it, as read_raw_device already does. + buffer = malloc(sz + 1); if (!buffer) { fprintf(stderr, "Failed to allocate memory for file contents\n"); return NULL; @@ -160,6 +170,8 @@ char *read_exact_size(FILE *f, size_t sz) { goto read_exact_error; } + buffer[sz] = '\0'; + return buffer; read_exact_error: @@ -167,22 +179,20 @@ char *read_exact_size(FILE *f, size_t sz) { return NULL; } -// read_file_and_size: Reads the file from arguments and returns buffer -// with all the contents of the file. Furthermore, it stores in the size argument -// the total size of the file. +// read_file_and_size: Opens , determines whether it is a regular file or +// a block device and reads it accordingly. // // Arguments: // 1. file: The file to read // 2. size: The total size of the file // // Return value: -// On success it returns a buffer with all the contents of the file and updates the -// size argument to contain the total size of the file. +// On success it returns a buffer with all the contents of the file and updates +// the size argument to contain the total size of the file. // On failure, it returns NULL. char *read_file_and_size(char *file, size_t *size) { FILE *fp = NULL; struct stat st = { 0 }; - int ret = 0; char *buf = NULL; DEBUG_PRINTF("Read configuration file %s\n", file); @@ -194,24 +204,30 @@ char *read_file_and_size(char *file, size_t *size) { // Find the total size of the file in order to read the whole file // and have a limit to search in the buffer. - ret = fstat(fileno(fp), &st); - if (ret != 0) { + if (fstat(fileno(fp), &st) != 0) { perror("Getting configuration file size"); - goto exit_read_file; + fclose(fp); + return NULL; } - DEBUG_PRINTF("Total size of configuration file %ld\n", st.st_size); - // Make sure to read the whole file in one buffer. - buf = read_exact_size(fp, st.st_size); + // A raw block device has no size in st_size, so it is read by + // the platform code (which only borrows the descriptor); a regular file + // is read through the stdio stream. Only one of the two interfaces is + // used per call, so the stdio buffer and the raw reads never disagree. + if (S_ISCHR(st.st_mode) || S_ISBLK(st.st_mode)) { + DEBUG_PRINT("Configuration is a block device\n"); + buf = read_raw_device(fileno(fp), size); + } else { + DEBUG_PRINTF("Total size of configuration file %ld\n", (long)st.st_size); + buf = read_exact_size(fp, st.st_size); + if (buf) + *size = st.st_size; + } + fclose(fp); if (!buf) { - fprintf(stderr, "Could not read whole configuration file\n"); - goto exit_read_file; + fprintf(stderr, "Could not read configuration %s\n", file); } - DEBUG_PRINTF("Contents of configuration file\n%s\n", buf); - *size = st.st_size; -exit_read_file: - fclose(fp); return buf; } @@ -420,6 +436,8 @@ int get_string_val(char *str, char **value) { // UID: // GID: // WD: +// ARC: (optional, followed by ARC lines of) +// ARV: (taken verbatim) // UCE // It is important to note, that this function will alter the given list, // replacing the new line characters with the end of string '\0' character. @@ -438,6 +456,7 @@ int get_string_val(char *str, char **value) { struct process_config *parse_process_config(char **string_area, size_t max_sz) { struct process_config *conf = NULL; char *tmp_field = NULL; + uint32_t found_argv = 0; conf = malloc(sizeof(struct process_config)); if (!conf) { @@ -446,6 +465,8 @@ struct process_config *parse_process_config(char **string_area, size_t max_sz) { } memset(conf, 0, sizeof(struct process_config)); conf->wdir = NULL; // Sanity + conf->argv = NULL; + conf->argc = 0; tmp_field = strtok(*string_area, "\n"); // Discard the first string since it is the special string "UCS" @@ -455,7 +476,7 @@ struct process_config *parse_process_config(char **string_area, size_t max_sz) { while (tmp_field && ((size_t)(tmp_field - *string_area) < max_sz)) { int ret = 0; - if (memcmp(tmp_field, "UID", 3) == 0) { + if (memcmp(tmp_field, "UID:", 4) == 0) { ret = get_uint_val(tmp_field, &(conf->uid)); if (ret != 0) { fprintf(stderr, "Failed to retreive UID information from %s\n", tmp_field); @@ -473,7 +494,44 @@ struct process_config *parse_process_config(char **string_area, size_t max_sz) { fprintf(stderr, "Failed to retreive WD information from %s\n", tmp_field); break; } + } else if (memcmp(tmp_field, "ARC:", 4) == 0) { + // Number of arguments of the application command. It + // must precede the ARV entries. + uint32_t argc = 0; + + ret = get_uint_val(tmp_field, &argc); + if (ret != 0 || conf->argv != NULL) { + fprintf(stderr, "Failed to retrieve ARC information from %s\n", tmp_field); + break; + } + // Compute the element count in size_t. argc is uint32_t, so + // "argc + 1" alone is 32-bit unsigned and wraps to 0 for + // argc == UINT32_MAX, handing calloc a zero-size allocation + // that the following ARV writes would then overflow. In size_t + // the +1 cannot wrap, and an oversized count fails calloc below. + conf->argv = calloc((size_t)argc + 1, sizeof(char *)); + if (!conf->argv) { + fprintf(stderr, "Failed to allocate memory for the application arguments\n"); + break; + } + conf->argc = argc; + found_argv = 0; + } else if (memcmp(tmp_field, "ARV:", 4) == 0) { + // One argument of the application command, taken verbatim + // (it may contain spaces or be empty). + if (conf->argv == NULL || found_argv >= conf->argc) { + fprintf(stderr, "Unexpected ARV entry %s\n", tmp_field); + break; + } + // We keep any argument verbatim because anything can be + // an argument (except a new line which is not supported) + conf->argv[found_argv++] = tmp_field + 4; + DEBUG_PRINTF("Found argument %s\n", tmp_field + 4); } else if (memcmp(tmp_field, "UCE", 3) == 0) { + if (conf->argv != NULL && found_argv != conf->argc) { + fprintf(stderr, "Expected %u arguments, found %u\n", conf->argc, found_argv); + break; + } *string_area = tmp_field + 4; // 4 bytes for the "UCE" string return conf; } @@ -481,6 +539,80 @@ struct process_config *parse_process_config(char **string_area, size_t max_sz) { tmp_field = strtok(NULL, "\n"); } + free(conf->argv); + free(conf); + return NULL; +} + +// parse_net_config: Parses a list with the following format: +// UNS +// IP: +// GW: +// MSK: +// UNE +// It is used by guests that can not get the network configuration from the +// kernel command line (e.g. FreeBSD). It alters the given list, replacing the +// new line characters with '\0'. The caller is responsible to free the +// returned memory. +// +// Arguments: +// 1. string_area: The list with in the aformentioned format. +// 2. max_sz: The max possible size of the list. +// +// Return value: +// On success it returns a pointer to a dynamically allocated net_config. +// Otherwise, NULL is returned +struct net_config *parse_net_config(char **string_area, size_t max_sz) { + struct net_config *conf = NULL; + char *tmp_field = NULL; + + conf = malloc(sizeof(struct net_config)); + if (!conf) { + fprintf(stderr, "Failed to allocate memory for network config\n"); + return NULL; + } + memset(conf, 0, sizeof(struct net_config)); + // just for snaity + conf->ip = NULL; + conf->gateway = NULL; + conf->mask = NULL; + + tmp_field = strtok(*string_area, "\n"); + // Discard the first string since it is the special string "UNS" + tmp_field = strtok(NULL, "\n"); + while (tmp_field && ((size_t)(tmp_field - *string_area) < max_sz)) { + int ret = 0; + + // An empty value (e.g. "IP:") means the field is not set. + if (memcmp(tmp_field, "IP:", 3) == 0) { + ret = get_string_val(tmp_field, &(conf->ip)); + if (ret != 0) { + conf->ip = NULL; + fprintf(stderr, "Failed to retreive IP information from %s\n", tmp_field); + } + } else if (memcmp(tmp_field, "GW:", 3) == 0) { + ret = get_string_val(tmp_field, &(conf->gateway)); + if (ret != 0) { + conf->gateway = NULL; + fprintf(stderr, "Failed to retreive GW information from %s\n", tmp_field); + } + } else if (memcmp(tmp_field, "MSK:", 4) == 0) { + ret = get_string_val(tmp_field, &(conf->mask)); + if (ret != 0) { + conf->mask = NULL; + fprintf(stderr, "Failed to retreive MSK information from %s\n", tmp_field); + } + } else if (memcmp(tmp_field, "UNE", 3) == 0) { + *string_area = tmp_field + 4; // 4 bytes for the "UNE" string + DEBUG_PRINTF("Found network config ip=%s gw=%s mask=%s\n", + conf->ip ? conf->ip : "", conf->gateway ? conf->gateway : "", + conf->mask ? conf->mask : ""); + return conf; + } + + tmp_field = strtok(NULL, "\n"); + } + free(conf); return NULL; } @@ -634,6 +766,7 @@ struct app_exec_config *get_config_from_file(char *file, char **sbuf) { struct app_exec_config *econf = NULL; struct process_config *pconf = NULL; struct block_config **bconf = NULL; + struct net_config *nconf = NULL; char *conf_area = NULL; buf = read_file_and_size(file, &size); @@ -647,7 +780,7 @@ struct app_exec_config *get_config_from_file(char *file, char **sbuf) { // Check if the special string "UES" is present // which means that now starts the environment variable // list. - if (memcmp(conf_area, "UES", 3) == 0) { + if (size >= 3 && memcmp(conf_area, "UES", 3) == 0) { char *init_conf_area = conf_area; // Extract the environment variables from the list env_vars = parse_envs(&conf_area, size, &path_env); @@ -671,7 +804,7 @@ struct app_exec_config *get_config_from_file(char *file, char **sbuf) { // Check if the special string "UCS" is present // which means that now starts the configuration for the application // execution environment - if (memcmp(conf_area, "UCS", 3) == 0) { + if (size >= 3 && memcmp(conf_area, "UCS", 3) == 0) { char *init_conf_area = conf_area; // Extract the environment variables from the list pconf = parse_process_config(&conf_area, size); @@ -694,7 +827,7 @@ struct app_exec_config *get_config_from_file(char *file, char **sbuf) { DEBUG_PRINT("Checking for block volumes mount configuration\n"); // Check if the special string "UBS" is present // which means that now starts the configuration for the block mounts - if (memcmp(conf_area, "UBS", 3) == 0) { + if (size >= 3 && memcmp(conf_area, "UBS", 3) == 0) { char *init_conf_area = conf_area; // Extract the block configuration bconf = parse_block_config(&conf_area, size); @@ -712,6 +845,23 @@ struct app_exec_config *get_config_from_file(char *file, char **sbuf) { size -= conf_area - init_conf_area; } + DEBUG_PRINT("Checking for network configuration\n"); + // Check if the special string "UNS" is present + // which means that now starts the network configuration + if (size >= 3 && memcmp(conf_area, "UNS", 3) == 0) { + char *init_conf_area = conf_area; + + nconf = parse_net_config(&conf_area, size); + if (!nconf) { + fprintf(stderr, "Warning: No network configuration was found\n"); + } + if (conf_area == init_conf_area) { + fprintf(stderr, "Invalid format of network configuration\n"); + goto get_env_vars_error_free; + } + size -= conf_area - init_conf_area; + } + econf = malloc(sizeof(struct app_exec_config)); if (!econf) { fprintf(stderr, "Could not allocate memory for app exec config struct\n"); @@ -723,13 +873,68 @@ struct app_exec_config *get_config_from_file(char *file, char **sbuf) { econf->path_env = path_env; econf->pr_conf = pconf; econf->blk_conf = bconf; + econf->net_conf = nconf; return econf; get_env_vars_error_free: + free(env_vars); + if (pconf) + free(pconf->argv); + free(pconf); + free(nconf); free(buf); return NULL; } +// load_app_config: Reads the urunit configuration, if any. The configuration +// file is given with the URUNIT_CONFIG environment variable or, when the +// guest passes boot parameters through the kernel environment (FreeBSD), with +// the URUNIT_CONFIG kernel environment variable. +// +// Arguments: +// 1. config: Will hold the parsed configuration, or stay untouched if +// there is no configuration file. +// 2. config_buf: Will hold the backing buffer of the configuration, which +// the caller is responsible to free. +// +// Return value: +// 0 if there is no configuration or it was loaded successfully. +// Otherwise 1 is returned. +int load_app_config(struct app_exec_config **config, char **config_buf) { + char *config_file = NULL; + int ret = 0; + uint8_t free_boot_var = 0; + + config_file = getenv("URUNIT_CONFIG"); + if (!config_file) { + config_file = get_boot_var("URUNIT_CONFIG"); + free_boot_var = 1; + } + if (!config_file) { + DEBUG_PRINT("No configuration file\n"); + ret = 0; + goto exit_load_app_config; + } + + // We need to mount sysfs to read the data from retained initrd + ret = mount_special_fs(); + if (ret != 0) { + fprintf(stderr, "Failed to mount special filesystems\n"); + ret = 1; + goto exit_load_app_config; + } + *config = get_config_from_file(config_file, config_buf); + if (!*config) { + fprintf(stderr, "Failed to read the configuration from %s\n", config_file); + ret = 1; + } + +exit_load_app_config: + if (free_boot_var) + free(config_file); + return ret; +} + // manual_execvpe: Tries to implement in a simple way execvpe, since execvpe is // only supported by glibc. The rational is to combine every path in env_path // (which is the PATH) with the file_bin (the executable) and try to execve. @@ -916,10 +1121,12 @@ int setup_exec_env(struct process_config *process_conf) { return 0; } -int child_func(char *argv[]) { - char *config_file = NULL; - char *config_buf = NULL; +int child_func(int argc, char *argv[]) { struct app_exec_config *app_config = NULL; + char *app_config_buf = NULL; + // The command line was already reassembled by spawn_app. When there is + // none, the application command comes from the configuration instead. + char **exec_argv = argv; int ret = 0; DEBUG_PRINT("Isolating child\n"); @@ -929,17 +1136,46 @@ int child_func(char *argv[]) { return 1; } - // Check if we need to read any configuration for the app execution - config_file = getenv("URUNIT_CONFIG"); - if (config_file) { - // We need to mount sysfs to read the data from retained initrd - ret = mount_special_fs(); - if (ret != 0) { - fprintf(stderr, "Failed to mount special filesystems\n"); - return 1; + // Load the configuration (if any) and apply everything that depends on + // it here in the child. + ret = load_app_config(&app_config, &app_config_buf); + if (ret != 0) { + fprintf(stderr, "Failed to load the configuration\n"); + return 1; + } + + // Configure the network if the configuration provides it. + if (app_config && app_config->net_conf) { + DEBUG_PRINT("Configuring the network\n"); + if (configure_network(app_config->net_conf) != 0) { + fprintf(stderr, "Failed to configure the network\n"); } - app_config = get_config_from_file(config_file, &config_buf); } + + // No command line (e.g. started by the FreeBSD kernel as init): take the + // application command from the configuration, where every argument is + // already separated and the array is NULL terminated. + if (argc < 2 && app_config && app_config->pr_conf && + app_config->pr_conf->argv && app_config->pr_conf->argc > 0) { + DEBUG_PRINT("Taking the application command from the configuration\n"); + exec_argv = app_config->pr_conf->argv; + } + + if (exec_argv[0] == NULL) { + fprintf(stderr, "No application execute\n"); + ret = 1; + goto child_func_free; + } +#ifdef DEBUG + printf("Starting app %s with the following arguments\n", exec_argv[0]); + for (int i = 1; exec_argv[i] != NULL; i++) { + printf("%s\n", exec_argv[i]); + } + printf("Environment variables\n"); + for (char **env = environ; *env != NULL; env++) { + printf("%s\n", *env); + } +#endif if (app_config) { ret = mount_block_vols(app_config->blk_conf); if (ret != 0) { @@ -951,17 +1187,22 @@ int child_func(char *argv[]) { fprintf(stderr, "Failed to set up the process execution environment\n"); goto child_func_free; } - ret = manual_execvpe(app_config->path_env, argv[0], argv, app_config->envs); + ret = manual_execvpe(app_config->path_env, exec_argv[0], exec_argv, app_config->envs); } else { DEBUG_PRINT("No configuration, simply execvp\n"); - ret = manual_execvpe(NULL, argv[0], argv, NULL); + ret = manual_execvpe(NULL, exec_argv[0], exec_argv, NULL); } // If we returned something went wrong child_func_free: - free(config_buf); - free(app_config->envs); - free(app_config->pr_conf); - free(app_config); + if (app_config) { + free(app_config->envs); + if (app_config->pr_conf) + free(app_config->pr_conf->argv); + free(app_config->pr_conf); + free(app_config->net_conf); + free(app_config); + } + free(app_config_buf); return ret; } @@ -1069,32 +1310,17 @@ int spawn_app(int argc, char *argv[], pid_t *child_pid) { } new_argv[new_argc] = NULL; - if (new_argc <= 0 || new_argv[0] == NULL) { - fprintf(stderr, "No application execute\n"); - return 1; - } -#ifdef DEBUG - printf("Starting app %s with the following arguments\n", new_argv[0]); - for (int i = 1; i < new_argc; i++) { - printf("%s\n", new_argv[i]); - } - printf("Environment variables\n"); - for (char **env = environ; *env != NULL; env++) { - printf("%s\n", *env); - } -#endif pid = fork(); if (pid < 0) { perror("fork"); return 1; } else if (pid == 0) { - return child_func(new_argv); - } else { - *child_pid = pid; - return 0; + return child_func(argc, new_argv); } - return 1; + *child_pid = pid; + + return 0; } int reap(const pid_t child_pid, int *child_exitcode_ptr) { @@ -1152,8 +1378,20 @@ int main(int argc, char *argv[]) { int ret = 0; int app_exitcode = -1; char *should_set_def_route = NULL; + uint8_t free_boot_var = 0; + + // When started by the kernel as init we may have no console yet. + setup_console(); + + // The kernel may mount the root read-only (FreeBSD does); remount it + // read-write for the application, regardless of the configuration. + remount_root_rw(); should_set_def_route = getenv("URUNIT_DEFROUTE"); + if (!should_set_def_route) { + should_set_def_route = get_boot_var("URUNIT_DEFROUTE"); + free_boot_var = 1; + } if (should_set_def_route) { DEBUG_PRINT("URUNIT_DEFROUTE was set\n"); ret = set_default_route(); @@ -1161,6 +1399,9 @@ int main(int argc, char *argv[]) { fprintf(stderr, "Failed to set default route\n"); } } + if (free_boot_var) { + free(should_set_def_route); + } DEBUG_PRINT("Setting subreaper\n"); ret = set_subreaper();