You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用CLONE_NEWUSER创建进程后pivot_root挂载/proc失败问题排查

问题:添加CLONE_NEWUSER后pivot_root挂载/proc失败(Operation not permitted)

在执行pivot_root后挂载/proc和/dev时,若子进程通过clone调用添加CLONE_NEWUSER标志创建,挂载操作会失败,报错信息为pivot_root_demo: mount-proc: Operation not permitted;移除该标志后所有操作正常。

问题代码示例

#define _GNU_SOURCE
#include <err.h>
#include <limits.h>
#include <sched.h>
#include <signal.h>
#include <stdio.h>
#include <stdlib.h>
#include <sys/mman.h>
#include <sys/mount.h>
#include <sys/stat.h>
#include <sys/syscall.h>
#include <sys/wait.h>
#include <unistd.h>

static int
pivot_root(const char *new_root, const char *put_old)
{
    return syscall(SYS_pivot_root, new_root, put_old);
}

#define STACK_SIZE (1024 * 1024)

static int              /* Startup function for cloned child */
child(void *arg)
{
    char        **args = arg;
    char        *new_root = args[0];

    /* Ensure that 'new_root' and its parent mount don't have
       shared propagation (which would cause pivot_root() to
       return an error), and prevent propagation of mount
       events to the initial mount namespace. */

    if (mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL) == -1)
        err(EXIT_FAILURE, "mount-MS_PRIVATE");

    /* Ensure that 'new_root' is a mount point. */

    if (mount(new_root, new_root, NULL, MS_BIND, NULL) == -1)
        err(EXIT_FAILURE, "mount-MS_BIND");


    /* And pivot the root filesystem. */

    if (pivot_root(new_root, new_root) == -1)
        err(EXIT_FAILURE, "pivot_root");

    /* Switch the current working directory to "/". */

    if (chdir("/") == -1)
        err(EXIT_FAILURE, "chdir");

    /* Unmount old root and remove mount point. */
    if (mount(NULL, "/", NULL, MS_REC | MS_SLAVE, NULL) == -1)
        err(EXIT_FAILURE, "mount-MS_PRIVATE");

    if (umount2("/", MNT_DETACH) == -1)
        perror("umount2");

    
    if (mount("proc", "/proc", "proc", MS_NOSUID | MS_NODEV | MS_NOEXEC, NULL) == -1)
        err(EXIT_FAILURE, "mount-proc");

    if (mount("tmpfs", "/dev", "tmpfs", MS_STRICTATIME | MS_NOSUID, NULL) == -1)
        err(EXIT_FAILURE, "mount-proc");


    /* Execute the command specified in argv[1]... */

    execv(args[1], &args[1]);
    err(EXIT_FAILURE, "execv");
}

int
main(int argc, char *argv[])
{
    char *stack;

    /* Create a child process in a new mount namespace. */
    stack = mmap(NULL, STACK_SIZE, PROT_READ | PROT_WRITE,
                 MAP_PRIVATE | MAP_ANONYMOUS | MAP_STACK, -1, 0);
    if (stack == MAP_FAILED)
        err(EXIT_FAILURE, "mmap");

    if (clone(child, stack + STACK_SIZE,
              CLONE_NEWNS | 
                          CLONE_NEWPID |
                          CLONE_NEWNET | 
                          CLONE_NEWUSER | 
                          SIGCHLD, &argv[1]) == -1)
        err(EXIT_FAILURE, "clone");

    /* Parent falls through to here; wait for child. */

    if (wait(NULL) == -1)
        err(EXIT_FAILURE, "wait");

    exit(EXIT_SUCCESS);
}

运行结果对比

无CLONE_NEWUSER时的正常输出

sudo ./pivot_root_demo /opt/chariot/containers/busybox/rootfs /bin/ls /proc
1                  consoles           dynamic_debug      ioports            kpagecgroup        meminfo            pagetypeinfo       softirqs           timer_list         zoneinfo
acpi               cpuinfo            execdomains        irq                kpagecount         misc               partitions         stat               tty
bootconfig         crypto             fb                 kallsyms           kpageflags         modules            pressure           swaps              uptime
buddyinfo          devices            filesystems        kcore              latency_stats      mounts             schedstat          sys                version
bus                diskstats          fs                 key-users          loadavg            mtd                scsi               sysrq-trigger      version_signature
cgroups            dma                interrupts         keys               locks              mtrr               self               sysvipc            vmallocinfo
cmdline            driver             iomem              kmsg               mdstat             net                slabinfo           thread-self        vmstat

添加CLONE_NEWUSER后的报错

sudo ./pivot_root_demo /opt/chariot/containers/busybox/rootfs /bin/ls /proc
pivot_root_demo: mount-proc: Operation not permitted

原因分析

使用CLONE_NEWUSER创建用户命名空间后,子进程在新命名空间内默认以无特权的nobody身份运行,即使父进程是root用户。挂载proc这类操作需要CAP_SYS_ADMIN能力,而新用户命名空间中默认没有该权限,导致挂载失败。

解决方法

需要在子进程中完成用户ID映射,将新命名空间内的UID/GID映射到宿主系统的特权UID/GID,并确保进程拥有所需能力。

修改后的代码

#define _GNU_SOURCE
#include <err.h>
#include <limits.h>
#include <sched.h>
#include <signal.h>
#include <stdio.h>
#include <stdlib.h>
#include <sys/mman.h>
#include <sys/mount.h>
#include <sys/stat.h>
#include <sys/syscall.h>
#include <sys/wait.h>
#include <unistd.h>
#include <fcntl.h>

static int
pivot_root(const char *new_root, const char *put_old)
{
    return syscall(SYS_pivot_root, new_root, put_old);
}

#define STACK_SIZE (1024 * 1024)

static void setup_uid_gid_mapping() {
    // 清空组列表,必须在写入gid_map之前调用
    if (setgroups(0, NULL) == -1)
        err(EXIT_FAILURE, "setgroups");

    // 写入UID映射:新命名空间的UID 0 映射到宿主的UID 0,范围1
    int fd = open("/proc/self/uid_map", O_WRONLY);
    if (fd == -1)
        err(EXIT_FAILURE, "open-uid_map");
    if (dprintf(fd, "0 %d 1\n", getuid()) == -1)
        err(EXIT_FAILURE, "write-uid_map");
    close(fd);

    // 写入GID映射:新命名空间的GID 0 映射到宿主的GID 0,范围1
    fd = open("/proc/self/gid_map", O_WRONLY);
    if (fd == -1)
        err(EXIT_FAILURE, "open-gid_map");
    if (dprintf(fd, "0 %d 1\n", getgid()) == -1)
        err(EXIT_FAILURE, "write-gid_map");
    close(fd);
}

static int              /* Startup function for cloned child */
child(void *arg)
{
    char        **args = arg;
    char        *new_root = args[0];

    // 先设置UID/GID映射,获取新命名空间内的特权
    setup_uid_gid_mapping();

    /* Ensure that 'new_root' and its parent mount don't have
       shared propagation (which would cause pivot_root() to
       return an error), and prevent propagation of mount
       events to the initial mount namespace. */

    if (mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL) == -1)
        err(EXIT_FAILURE, "mount-MS_PRIVATE");

    /* Ensure that 'new_root' is a mount point. */

    if (mount(new_root, new_root, NULL, MS_BIND, NULL) == -1)
        err(EXIT_FAILURE, "mount-MS_BIND");


    /* And pivot the root filesystem. */

    if (pivot_root(new_root, new_root) == -1)
        err(EXIT_FAILURE, "pivot_root");

    /* Switch the current working directory to "/". */

    if (chdir("/") == -1)
        err(EXIT_FAILURE, "chdir");

    /* Unmount old root and remove mount point. */
    if (mount(NULL, "/", NULL, MS_REC | MS_SLAVE, NULL) == -1)
        err(EXIT_FAILURE, "mount-MS_PRIVATE");

    if (umount2("/", MNT_DETACH) == -1)
        perror("umount2");

    
    if (mount("proc", "/proc", "proc", MS_NOSUID | MS_NODEV | MS_NOEXEC, NULL) == -1)
        err(EXIT_FAILURE, "mount-proc");

    if (mount("tmpfs", "/dev", "tmpfs", MS_STRICTATIME | MS_NOSUID, NULL) == -1)
        err(EXIT_FAILURE, "mount-dev");


    /* Execute the command specified in argv[1]... */

    execv(args[1], &args[1]);
    err(EXIT_FAILURE, "execv");
}

int
main(int argc, char *argv[])
{
    char *stack;

    /* Create a child process in a new mount namespace. */
    stack = mmap(NULL, STACK_SIZE, PROT_READ | PROT_WRITE,
                 MAP_PRIVATE | MAP_ANONYMOUS | MAP_STACK, -1, 0);
    if (stack == MAP_FAILED)
        err(EXIT_FAILURE, "mmap");

    if (clone(child, stack + STACK_SIZE,
              CLONE_NEWNS | 
                          CLONE_NEWPID |
                          CLONE_NEWNET | 
                          CLONE_NEWUSER | 
                          SIGCHLD, &argv[1]) == -1)
        err(EXIT_FAILURE, "clone");

    /* Parent falls through to here; wait for child. */

    if (wait(NULL) == -1)
        err(EXIT_FAILURE, "wait");

    exit(EXIT_SUCCESS);
}

修改说明

  1. 添加UID/GID映射:新增setup_uid_gid_mapping函数,负责将新用户命名空间内的根UID/GID(0)映射到宿主系统的当前UID/GID(父进程为root时映射到0)。
  2. 调用顺序:必须在执行任何需要特权的操作(如mount、pivot_root)之前调用映射函数,且setgroups(0, NULL)要在写入gid_map之前执行,否则会因权限问题失败。
  3. 能力获取:完成映射后,子进程在新用户命名空间内拥有CAP_SYS_ADMIN能力,足以执行挂载/proc和/dev的操作。

内容的提问来源于stack exchange,提问作者Klaus Ma

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.17 09:21:00