// Copyright (C) 2026 Viorel Munteanu // // This program is free software; you can redistribute it and/or modify // it under the terms of the GNU General Public License as published by // the Free Software Foundation; either version 2 of the License, or // (at your option) any later version. // // This program is distributed in the hope that it will be useful, // but WITHOUT ANY WARRANTY; without even the implied warranty of // MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the // GNU General Public License for more details. // // You should have received a copy of the GNU General Public License along // with this program; if not, write to the Free Software Foundation, Inc., // 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA. #include "bwrap.h" #include #include #include #include #include #include #include #include #include #include #include #include #include #include "process.h" namespace { struct NamespaceProbe { int clone_flag; const char* bwrap_arg; const char* name; }; constexpr std::array kNamespaceProbes = {{ {CLONE_NEWUSER, "--unshare-user", "user"}, {CLONE_NEWIPC, "--unshare-ipc", "ipc"}, {CLONE_NEWPID, "--unshare-pid", "pid"}, {CLONE_NEWNET, "--unshare-net", "net"}, {CLONE_NEWUTS, "--unshare-uts", "uts"}, {CLONE_NEWCGROUP, "--unshare-cgroup", "cgroup"}, }}; // unshare(2) affects the calling process's own namespaces, so support for each // namespace type is probed in a throwaway forked child rather than the caller. bool kernel_supports_namespace(int clone_flag) { pid_t pid = fork(); if (pid < 0) { return false; } if (pid == 0) { _exit(unshare(clone_flag) == 0 ? 0 : 1); } int status = 0; waitpid(pid, &status, 0); return WIFEXITED(status) && WEXITSTATUS(status) == 0; } // Scans /proc for a fuse-overlayfs process whose command line references // `merged_path`, mirroring `ps aux | grep fuse-overlayfs`. /proc is inherently racy // (processes come and go while it's being scanned), so filesystem errors from a // vanished entry are treated as "not this one" rather than propagated. std::optional find_fuse_overlayfs_pid(const std::string& merged_path) { std::error_code ec; auto it = std::filesystem::directory_iterator("/proc", ec); if (ec) { return std::nullopt; } for (const auto& entry : it) { const std::string name = entry.path().filename().string(); if (!std::all_of(name.begin(), name.end(), [](unsigned char c) { return std::isdigit(c); })) { continue; } std::error_code exe_ec; auto exe_target = std::filesystem::read_symlink(entry.path() / "exe", exe_ec); if (exe_ec || exe_target.filename() != "fuse-overlayfs") { continue; } std::ifstream cmdline_file(entry.path() / "cmdline", std::ios::binary); std::string cmdline((std::istreambuf_iterator(cmdline_file)), std::istreambuf_iterator()); size_t start = 0; while (start < cmdline.size()) { size_t end = cmdline.find('\0', start); if (end == std::string::npos) { end = cmdline.size(); } if (cmdline.compare(start, end - start, merged_path) == 0) { return static_cast(std::stoi(name)); } start = end + 1; } } return std::nullopt; } } // namespace std::vector detect_bwrap_unshare_args() { std::vector args; bool user_ns_supported = kernel_supports_namespace(CLONE_NEWUSER); spdlog::debug("namespace user: {}", user_ns_supported ? "supported" : "not supported"); // Root already has full privilege without a new user namespace, and bwrap's own // uid/gid mapping into one (a single trivial mapping, since we don't pass // --uid/--gid/subuid ranges) forces the kernel's unprivileged-userns // setgroups() restriction: every other supplementary group collapses to the // overflow gid (65534/"nobody"), and setgroups() calls inside the sandbox (e.g. // `su`) fail with "Operation not permitted". Observed directly: `id` inside a // root-launched sandbox showed "nobody" repeated once per real supplementary // group, and `su git` failed exactly this way. So only request --unshare-user // (and only combine the other probes with it) when not root. bool combine_with_user_ns = user_ns_supported && geteuid() != 0; if (combine_with_user_ns) { args.push_back("--unshare-user"); } for (const auto& probe : kNamespaceProbes) { if (probe.clone_flag == CLONE_NEWUSER) { continue; } // As a regular (non-root) user, most namespace types can only be unshared // together with a fresh user namespace (which supplies the capabilities // needed), not in isolation -- so combine them here whenever it applies. int flags = probe.clone_flag | (combine_with_user_ns ? CLONE_NEWUSER : 0); bool supported = kernel_supports_namespace(flags); spdlog::debug("namespace {}: {}", probe.name, supported ? "supported" : "not supported"); if (supported) { args.push_back(probe.bwrap_arg); } } return args; } std::vector build_bwrap_args(const std::string& root, const std::vector& command) { // --new-session detaches from the controlling terminal, which breaks job // control for an interactive foreground shell ("can't access tty"). Re-enable // once background/daemonized runs are implemented, where that's the point. std::vector args = {"bwrap", "--die-with-parent"}; auto unshare_args = detect_bwrap_unshare_args(); bool has_pid_ns = false; for (const auto& arg : unshare_args) { // Not requested yet: without any network setup (slirp4netns or similar), // unsharing it just leaves the sandbox with no network at all. Re-add once // network isolation is implemented. (detect_bwrap_unshare_args() still // probes/reports it, e.g. for -t/--test, since that's kernel capability, not // policy.) if (arg == "--unshare-net") { continue; } args.push_back(arg); if (arg == "--unshare-pid") { has_pid_ns = true; } } if (has_pid_ns) { args.push_back("--as-pid-1"); } std::vector filesystem_args = { "--bind", root, "/", "--proc", "/proc", "--dev", "/dev", "--perms", "01777", "--tmpfs", "/dev/shm", "--perms", "01777", "--tmpfs", "/tmp", "--chdir", "/", "--clearenv", "--setenv", "PATH", "/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin", "--setenv", "HOME", "/root", }; args.insert(args.end(), filesystem_args.begin(), filesystem_args.end()); if (const char* term = std::getenv("TERM")) { args.insert(args.end(), {"--setenv", "TERM", term}); } args.push_back("--"); args.insert(args.end(), command.begin(), command.end()); return args; } int run_bwrap(const std::string& root, const std::vector& command, bool use_nsenter) { std::vector argv; if (use_nsenter) { auto pid = find_fuse_overlayfs_pid(root); if (!pid) { spdlog::error("could not find the fuse-overlayfs process serving {}", root); return -1; } if (!find_in_path("nsenter")) { spdlog::error("nsenter not found in PATH"); return -1; } argv = {"nsenter", fmt::format("--user=/proc/{}/ns/user", *pid), fmt::format("--mount=/proc/{}/ns/mnt", *pid), "--"}; } auto bwrap_args = build_bwrap_args(root, command); argv.insert(argv.end(), bwrap_args.begin(), bwrap_args.end()); return run_process_foreground(argv); }