diff --git a/include/container.h b/include/container.h index 12e6cf8..75bf00a 100644 --- a/include/container.h +++ b/include/container.h @@ -1,3 +1,9 @@ +/* container.h — 容器核心共享定义 + * + * 包含所有模块共用的结构体、常量和系统头文件。 + * 每个模块的 .c 文件都 include 此头文件。 + */ + #ifndef CONTAINER_H #define CONTAINER_H @@ -24,18 +30,24 @@ #include #include +/* clone() 子进程栈大小 */ #define STACK_SIZE (1024 * 1024) + +/* User Namespace UID/GID 映射:容器内 0 映射到宿主机 10000~11999 */ #define USERNS_OFFSET 10000 #define USERNS_COUNT 2000 + +/* 容器内最大打开文件描述符数 */ #define FD_COUNT 64 +/* 子进程配置,由 main() 填充,传递给 child() */ struct child_config { - int argc; - uid_t uid; - int fd; - char *hostname; - char **argv; - char *mount_dir; + int argc; /* -c 后面命令的参数个数 */ + uid_t uid; /* -u 指定的容器内 UID */ + int fd; /* 父子进程通信用的 socket fd */ + char *hostname; /* 容器主机名 */ + char **argv; /* -c 后面的命令及参数 */ + char *mount_dir; /* -m 指定的 rootfs 路径 */ }; #endif diff --git a/src/capabilities.c b/src/capabilities.c index 5be3ce4..42faf90 100644 --- a/src/capabilities.c +++ b/src/capabilities.c @@ -1,3 +1,14 @@ +/* capabilities.c — Linux Capabilities 权限收窄模块 + * + * 将 root 的"万能权限"拆解为 30+ 个细粒度的 capability, + * 本模块 drop 掉容器不需要的 20 个危险 capability, + * 只保留最小权限集。 + * + * 两步操作: + * 1. prctl(PR_CAPBSET_DROP) — 从 bounding set 移除,子进程也无法获取 + * 2. cap_set_flag(CAP_INHERITABLE, CAP_CLEAR) — 清除 inheritable 集 + */ + #include #include #include @@ -7,29 +18,34 @@ int capabilities(void) { fprintf(stderr, "=> dropping capabilities..."); + /* 需要 drop 的危险 capability 列表 */ int drop_caps[] = { - CAP_AUDIT_CONTROL, - CAP_AUDIT_READ, - CAP_AUDIT_WRITE, - CAP_BLOCK_SUSPEND, - CAP_DAC_READ_SEARCH, - CAP_FSETID, - CAP_IPC_LOCK, - CAP_MAC_ADMIN, - CAP_MAC_OVERRIDE, - CAP_MKNOD, - CAP_SETFCAP, - CAP_SYSLOG, - CAP_SYS_ADMIN, - CAP_SYS_BOOT, - CAP_SYS_MODULE, - CAP_SYS_NICE, - CAP_SYS_RAWIO, - CAP_SYS_RESOURCE, - CAP_SYS_TIME, - CAP_WAKE_ALARM + CAP_AUDIT_CONTROL, /* 修改审计日志 */ + CAP_AUDIT_READ, /* 读取审计日志 */ + CAP_AUDIT_WRITE, /* 写入审计日志 */ + CAP_BLOCK_SUSPEND, /* 阻止系统挂起 */ + CAP_DAC_READ_SEARCH, /* 绕过文件读权限检查 */ + CAP_FSETID, /* 设置 setuid/setgid 位 */ + CAP_IPC_LOCK, /* 锁定共享内存 */ + CAP_MAC_ADMIN, /* 修改 MAC 策略(如 SELinux) */ + CAP_MAC_OVERRIDE, /* 覆盖 MAC 策略 */ + CAP_MKNOD, /* 创建设备文件 */ + CAP_SETFCAP, /* 设置文件 capability */ + CAP_SYSLOG, /* 操作内核日志 */ + CAP_SYS_ADMIN, /* 挂载、namespace 等大量特权操作 */ + CAP_SYS_BOOT, /* 重启系统 */ + CAP_SYS_MODULE, /* 加载/卸载内核模块 */ + CAP_SYS_NICE, /* 调整进程优先级 */ + CAP_SYS_RAWIO, /* 直接 I/O 端口操作 */ + CAP_SYS_RESOURCE, /* 修改资源限制 */ + CAP_SYS_TIME, /* 修改系统时间 */ + CAP_WAKE_ALARM /* 设置唤醒闹钟 */ }; size_t num_caps = sizeof(drop_caps) / sizeof(*drop_caps); + + /* 第一步:从 bounding set 中移除 + * bounding set 限制了进程能获取的最大能力集, + * 一旦 drop,即使后续 execve setuid 程序也无法获取 */ fprintf(stderr, "bounding..."); for (size_t i = 0; i < num_caps; i++) { if (prctl(PR_CAPBSET_DROP, drop_caps[i], 0, 0, 0)) { @@ -37,6 +53,9 @@ int capabilities(void) return 1; } } + + /* 第二步:从 inheritable set 中清除 + * inheritable set 控制 execve 后哪些 capability 可被继承 */ fprintf(stderr, "inheritable..."); cap_t caps = NULL; if (!(caps = cap_get_proc()) diff --git a/src/cgroup.c b/src/cgroup.c index a1ca0d6..2840408 100644 --- a/src/cgroup.c +++ b/src/cgroup.c @@ -1,25 +1,40 @@ +/* cgroup.c — Cgroup 资源限制模块 + * + * 通过 cgroup v1 限制容器能使用的资源: + * - memory: 内存限制 1GB + * - cpu: CPU 时间片权重 256 + * - pids: 最大进程数 64 + * + * 容器退出后清理创建的 cgroup 目录。 + */ + #include "container.h" #include "cgroup.h" -#define MEMORY "1073741824" -#define SHARES "256" -#define PIDS "64" -#define WEIGHT "10" +/* 资源限制值 */ +#define MEMORY "1073741824" /* 1GB 内存限制 */ +#define SHARES "256" /* CPU 时间片权重 */ +#define PIDS "64" /* 最大进程数 */ +#define WEIGHT "10" /* I/O 权重(备用) */ +/* cgroup 控制器结构体:描述一个 cgroup 子系统及其配置项 */ struct cgrp_control { - char control[256]; + char control[256]; /* cgroup 子系统名,如 "memory"、"cpu" */ struct cgrp_setting { - char name[256]; - char value[256]; + char name[256]; /* 配置文件名,如 "memory.limit_in_bytes" */ + char value[256]; /* 写入的值 */ } **settings; }; +/* 通用配置:将当前进程加入 cgroup 的 tasks 文件 */ static struct cgrp_setting add_to_tasks = { .name = "tasks", .value = "0" }; +/* cgroup 配置数组:定义要创建的 cgroup 及其限制 */ static struct cgrp_control *cgrps[] = { + /* 内存限制:1GB */ & (struct cgrp_control) { .control = "memory", .settings = (struct cgrp_setting *[]) { @@ -35,6 +50,7 @@ static struct cgrp_control *cgrps[] = { NULL } }, + /* CPU 权重:256(默认 1024,值越小分到的 CPU 越少) */ & (struct cgrp_control) { .control = "cpu", .settings = (struct cgrp_setting *[]) { @@ -46,6 +62,7 @@ static struct cgrp_control *cgrps[] = { NULL } }, + /* 进程数限制:最多 64 个进程 */ & (struct cgrp_control) { .control = "pids", .settings = (struct cgrp_setting *[]) { @@ -60,12 +77,14 @@ static struct cgrp_control *cgrps[] = { NULL }; +/* 创建 cgroup 目录并写入资源限制值 */ int resources(struct child_config *config) { fprintf(stderr, "=> setting cgroups..."); for (struct cgrp_control **cgrp = cgrps; *cgrp; cgrp++) { char dir[PATH_MAX] = {0}; fprintf(stderr, "%s...", (*cgrp)->control); + /* 创建 /sys/fs/cgroup/<子系统>/<主机名>/ 目录 */ if (snprintf(dir, sizeof(dir), "/sys/fs/cgroup/%s/%s", (*cgrp)->control, config->hostname) == -1) { return -1; @@ -74,6 +93,7 @@ int resources(struct child_config *config) fprintf(stderr, "mkdir %s failed: %m\n", dir); return -1; } + /* 遍历该子系统的所有配置项,逐个写入 */ for (struct cgrp_setting **setting = (*cgrp)->settings; *setting; setting++) { char path[PATH_MAX] = {0}; int fd = 0; @@ -95,6 +115,8 @@ int resources(struct child_config *config) } } fprintf(stderr, "done.\n"); + + /* 设置文件描述符限制(rlimit) */ fprintf(stderr, "=> setting rlimit..."); if (setrlimit(RLIMIT_NOFILE, & (struct rlimit) { @@ -108,6 +130,7 @@ int resources(struct child_config *config) return 0; } +/* 清理 cgroup 目录:先把进程移回根 cgroup,再删除目录 */ int free_resources(struct child_config *config) { fprintf(stderr, "=> cleaning cgroups..."); @@ -122,6 +145,7 @@ int free_resources(struct child_config *config) fprintf(stderr, "snprintf failed: %m\n"); return -1; } + /* 将当前进程写入根 cgroup 的 tasks,从容器 cgroup 中移出 */ if ((task_fd = open(task, O_WRONLY)) == -1) { fprintf(stderr, "opening %s failed: %m\n", task); return -1; @@ -132,6 +156,7 @@ int free_resources(struct child_config *config) return -1; } close(task_fd); + /* 删除容器的 cgroup 目录 */ if (rmdir(dir)) { fprintf(stderr, "rmdir %s failed: %m", dir); return -1; diff --git a/src/hostname.c b/src/hostname.c index 31a8d4a..8e70b1d 100644 --- a/src/hostname.c +++ b/src/hostname.c @@ -1,26 +1,43 @@ +/* hostname.c — 随机主机名生成模块 + * + * 基于当前时间戳的纳秒部分,从 78 张塔罗牌中随机选取一张作为主机名。 + * 格式:<5位时间戳十六进制>-<塔罗牌名> + * 例如:1204f86c-king-of-cups + * + * 78 张牌 = 22 张大阿卡纳 + 56 张小阿卡纳(4花色×14数字) + */ + #include #include #include "hostname.h" int choose_hostname(char *buff, size_t len) { + /* 小阿卡纳四花色 */ static const char *suits[] = { "swords", "wands", "pentacles", "cups" }; + /* 小阿卡纳 14 张数字牌 */ static const char *minor[] = { "ace", "two", "three", "four", "five", "six", "seven", "eight", "nine", "ten", "page", "knight", "queen", "king" }; + /* 大阿卡纳 22 张 */ static const char *major[] = { "fool", "magician", "high-priestess", "empress", "emperor", "hierophant", "lovers", "chariot", "strength", "hermit", "wheel", "justice", "hanged-man", "death", "temperance", "devil", "tower", "star", "moon", "sun", "judgment", "world" }; + + /* 用单调时钟的纳秒部分做随机种子 */ struct timespec now = {0}; clock_gettime(CLOCK_MONOTONIC, &now); - size_t ix = now.tv_nsec % 78; + size_t ix = now.tv_nsec % 78; /* 78 张牌 */ + if (ix < sizeof(major) / sizeof(*major)) { + /* 0~21:大阿卡纳 */ snprintf(buff, len, "%05lx-%s", now.tv_sec, major[ix]); } else { + /* 22~77:小阿卡纳(数字 + 花色) */ ix -= sizeof(major) / sizeof(*major); snprintf(buff, len, "%05lxc-%s-of-%s", diff --git a/src/main.c b/src/main.c index 9017b78..c981c77 100644 --- a/src/main.c +++ b/src/main.c @@ -1,3 +1,14 @@ +/* main.c — 容器主控流程 + * + * 负责: + * 1. 解析命令行参数(-m、-u、-c) + * 2. 校验内核版本 + * 3. 创建 cgroup 资源限制 + * 4. clone() 创建带 namespace 的子进程 + * 5. 父进程写入 UID 映射并等待子进程退出 + * 6. 子进程依次执行隔离初始化后 execve 目标命令 + */ + #include "container.h" #include "mount.h" #include "cgroup.h" @@ -6,9 +17,11 @@ #include "userns.h" #include "hostname.h" +/* 子进程入口函数,由 clone() 调用 */ int child(void *arg) { struct child_config *config = arg; + /* 按顺序执行容器初始化:设置主机名 → 文件系统隔离 → 用户命名空间 → 权限收窄 → syscall 过滤 */ if (sethostname(config->hostname, strlen(config->hostname)) || mounts(config) || userns(config) @@ -17,10 +30,12 @@ int child(void *arg) close(config->fd); return -1; } + /* 初始化完成,关闭与父进程的通信 socket */ if (close(config->fd)) { fprintf(stderr, "close failed: %m\n"); return -1; } + /* 替换为用户指定的进程 */ if (execve(config->argv[0], config->argv, NULL)) { fprintf(stderr, "execve failed! %m.\n"); return -1; @@ -33,12 +48,15 @@ int main(int argc, char **argv) struct child_config config = {0}; int err = 0; int option = 0; - int sockets[2] = {0}; + int sockets[2] = {0}; /* 父子进程通信用的 socket 对 */ pid_t child_pid = 0; int last_optind = 0; + + /* 解析命令行参数:-m rootfs路径, -u UID, -c 要执行的命令 */ while ((option = getopt(argc, argv, "c:m:u:"))) { switch (option) { case 'c': + /* -c 必须放最后,它后面的都当作命令参数 */ config.argc = argc - last_optind - 1; config.argv = &argv[argc - config.argc]; goto finish_options; @@ -60,6 +78,7 @@ finish_options: if (!config.argc) goto usage; if (!config.mount_dir) goto usage; + /* 校验内核版本 >= 4.7(支持 user namespace 的 unprivileged 模式) */ fprintf(stderr, "=> validating Linux version..."); struct utsname host = {0}; if (uname(&host)) { @@ -82,45 +101,57 @@ finish_options: } fprintf(stderr, "%s on %s.\n", host.release, host.machine); + /* 生成随机容器主机名(塔罗牌名) */ char hostname[256] = {0}; if (choose_hostname(hostname, sizeof(hostname))) goto error; config.hostname = hostname; + /* 创建 socket 对用于父子进程同步(通知 UID 映射完成) */ if (socketpair(AF_LOCAL, SOCK_SEQPACKET, 0, sockets)) { fprintf(stderr, "socketpair failed: %m\n"); goto error; } + /* 父进程端设置 close-on-exec,子进程继承另一端 */ if (fcntl(sockets[0], F_SETFD, FD_CLOEXEC)) { fprintf(stderr, "fcntl failed: %m\n"); goto error; } config.fd = sockets[1]; + /* 分配子进程栈空间(clone 要求调用者提供栈) */ char *stack = 0; if (!(stack = malloc(STACK_SIZE))) { fprintf(stderr, "=> malloc failed, out of memory?\n"); goto error; } + + /* 创建 cgroup 资源限制(内存、CPU、进程数) */ if (resources(&config)) { err = 1; goto clear_resources; } - int flags = CLONE_NEWNS - | CLONE_NEWCGROUP - | CLONE_NEWPID - | CLONE_NEWIPC - | CLONE_NEWNET - | CLONE_NEWUTS; + + /* clone 子进程,创建 6 种 namespace 隔离 */ + int flags = CLONE_NEWNS /* 挂载隔离 */ + | CLONE_NEWCGROUP /* cgroup 隔离 */ + | CLONE_NEWPID /* 进程号隔离 */ + | CLONE_NEWIPC /* 进程间通信隔离 */ + | CLONE_NEWNET /* 网络隔离 */ + | CLONE_NEWUTS; /* 主机名隔离 */ if ((child_pid = clone(child, stack + STACK_SIZE, flags | SIGCHLD, &config)) == -1) { fprintf(stderr, "=> clone failed! %m\n"); err = 1; goto clear_resources; } + + /* 父进程:关闭子进程端的 socket */ close(sockets[1]); sockets[1] = 0; close(sockets[1]); sockets[1] = 0; + + /* 父进程:写入 uid_map/gid_map,完成容器内 root → 宿主机普通用户的映射 */ if (handle_child_uid_map(child_pid, sockets[0])) { err = 1; goto kill_and_finish_child; @@ -128,12 +159,15 @@ finish_options: goto finish_child; kill_and_finish_child: + /* 出错时杀死子进程 */ if (child_pid) kill(child_pid, SIGKILL); finish_child:; + /* 等待子进程退出,获取退出码 */ int child_status = 0; waitpid(child_pid, &child_status, 0); err |= WEXITSTATUS(child_status); clear_resources: + /* 清理 cgroup 目录和栈空间 */ free_resources(&config); free(stack); diff --git a/src/mount.c b/src/mount.c index d3e10e7..33d7f24 100644 --- a/src/mount.c +++ b/src/mount.c @@ -1,6 +1,17 @@ +/* mount.c — 文件系统隔离模块 + * + * 负责容器的文件系统隔离: + * 1. MS_PRIVATE 重挂载,防止挂载事件泄漏到宿主机 + * 2. bind mount 将 rootfs 挂载到临时目录 + * 3. pivot_root 切换根文件系统 + * 4. 卸载旧的根文件系统 + * 5. 挂载 /proc(进程信息)和 /dev(设备节点) + */ + #include "container.h" #include "mount.h" +/* pivot_root 系统调用封装(glibc 没有提供封装函数) */ static int pivot_root(const char *new_root, const char *put_old) { return syscall(SYS_pivot_root, new_root, put_old); @@ -8,6 +19,8 @@ static int pivot_root(const char *new_root, const char *put_old) int mounts(struct child_config *config) { + /* 第一步:将当前所有挂载点标记为 MS_PRIVATE + * 防止 mount namespace 内的挂载事件传播到宿主机 */ fprintf(stderr, "=> remounting everything with MS_PRIVATE..."); if (mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL)) { fprintf(stderr, "failed! %m\n"); @@ -15,6 +28,7 @@ int mounts(struct child_config *config) } fprintf(stderr, "remounted.\n"); + /* 第二步:创建临时目录,将 rootfs bind mount 到这里 */ fprintf(stderr, "=> making a temp directory and a bind mount there..."); char mount_dir[] = "/tmp/tmp.XXXXXX"; if (!mkdtemp(mount_dir)) { @@ -22,11 +36,13 @@ int mounts(struct child_config *config) return -1; } + /* bind mount:将 -m 指定的 rootfs 挂载到临时目录 */ if (mount(config->mount_dir, mount_dir, NULL, MS_BIND | MS_PRIVATE, NULL)) { fprintf(stderr, "bind mount failed!\n"); return -1; } + /* 在挂载点内创建 oldroot 子目录,用于存放旧根 */ char inner_mount_dir[] = "/tmp/tmp.XXXXXX/oldroot.XXXXXX"; memcpy(inner_mount_dir, mount_dir, sizeof(mount_dir) - 1); if (!mkdtemp(inner_mount_dir)) { @@ -35,6 +51,7 @@ int mounts(struct child_config *config) } fprintf(stderr, "done.\n"); + /* 第三步:pivot_root,将根文件系统切换到 rootfs */ fprintf(stderr, "=> pivoting root..."); if (pivot_root(mount_dir, inner_mount_dir)) { fprintf(stderr, "failed!\n"); @@ -42,6 +59,7 @@ int mounts(struct child_config *config) } fprintf(stderr, "done.\n"); + /* 第四步:卸载旧根文件系统,彻底与宿主机文件系统断开 */ char *old_root_dir = basename(inner_mount_dir); char old_root[sizeof(inner_mount_dir) + 1] = { "/" }; strcpy(&old_root[1], old_root_dir); @@ -51,6 +69,7 @@ int mounts(struct child_config *config) fprintf(stderr, "chdir failed! %m\n"); return -1; } + /* MNT_DETACH:lazy unmount,先断开挂载点,等引用释放后再清理 */ if (umount2(old_root, MNT_DETACH)) { fprintf(stderr, "umount failed! %m\n"); return -1; @@ -61,6 +80,7 @@ int mounts(struct child_config *config) } fprintf(stderr, "done.\n"); + /* 第五步:挂载 /proc,让 ps、top 等工具能读取容器内进程信息 */ fprintf(stderr, "=> mounting /proc..."); if (mount("proc", "/proc", "proc", 0, NULL)) { fprintf(stderr, "failed: %m\n"); @@ -68,12 +88,15 @@ int mounts(struct child_config *config) } fprintf(stderr, "done.\n"); + /* 第六步:挂载 /dev(tmpfs),创建必要的设备节点 */ fprintf(stderr, "=> mounting /dev..."); if (mount("tmpfs", "/dev", "tmpfs", MS_NOSUID | MS_STRICTATIME, "mode=755,size=65536k")) { fprintf(stderr, "failed: %m\n"); return -1; } + /* 清除 umask,确保设备节点权限正确(0666) */ umask(0); + /* 创建基本设备节点:null、zero、urandom */ if (mknod("/dev/null", S_IFCHR | 0666, makedev(1, 3))) { fprintf(stderr, "mknod null failed: %m\n"); } diff --git a/src/seccomp.c b/src/seccomp.c index 9015cb2..c318c54 100644 --- a/src/seccomp.c +++ b/src/seccomp.c @@ -1,44 +1,77 @@ +/* seccomp.c — Seccomp 系统调用过滤模块 + * + * 通过 seccomp-bpf 过滤器限制容器内进程能使用的系统调用。 + * 默认策略:SCMP_ACT_ALLOW(允许所有),然后逐条禁止危险操作。 + * 违规的 syscall 返回 EPERM(Permission denied)。 + * + * 禁止的操作包括: + * - 设置 setuid/setgid 位(chmod/fchmod/fchmodat) + * - 创建新 user namespace(防止容器逃逸) + * - 终端注入(ioctl TIOCSTI) + * - 内核密钥操作(keyctl/add_key/request_key) + * - 进程跟踪(ptrace) + * - 内存策略操作(mbind/migrate_pages/move_pages/set_mempolicy/userfaultfd) + * - 性能监控(perf_event_open) + */ + #include "container.h" #include "container_seccomp.h" #include #include +/* 违规行为返回 EPERM(Operation not permitted) */ #define SCMP_FAIL SCMP_ACT_ERRNO(EPERM) int syscalls(void) { scmp_filter_ctx ctx = NULL; fprintf(stderr, "=> filtering syscalls..."); + if (!(ctx = seccomp_init(SCMP_ACT_ALLOW)) + /* 禁止 chmod 设置 SUID 位(防止提权) */ || seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(chmod), 1, SCMP_A1(SCMP_CMP_MASKED_EQ, S_ISUID, S_ISUID)) + /* 禁止 chmod 设置 SGID 位 */ || seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(chmod), 1, SCMP_A1(SCMP_CMP_MASKED_EQ, S_ISGID, S_ISGID)) + /* 禁止 fchmod 设置 SUID 位 */ || seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(fchmod), 1, SCMP_A1(SCMP_CMP_MASKED_EQ, S_ISUID, S_ISUID)) + /* 禁止 fchmod 设置 SGID 位 */ || seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(fchmod), 1, SCMP_A1(SCMP_CMP_MASKED_EQ, S_ISGID, S_ISGID)) + /* 禁止 fchmodat 设置 SUID 位 */ || seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(fchmodat), 1, SCMP_A2(SCMP_CMP_MASKED_EQ, S_ISUID, S_ISUID)) + /* 禁止 fchmodat 设置 SGID 位 */ || seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(fchmodat), 1, SCMP_A2(SCMP_CMP_MASKED_EQ, S_ISGID, S_ISGID)) + /* 禁止 unshare 创建新 user namespace(防止逃逸) */ || seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(unshare), 1, SCMP_A0(SCMP_CMP_MASKED_EQ, CLONE_NEWUSER, CLONE_NEWUSER)) + /* 禁止 clone 创建新 user namespace */ || seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(clone), 1, SCMP_A0(SCMP_CMP_MASKED_EQ, CLONE_NEWUSER, CLONE_NEWUSER)) + /* 禁止 ioctl TIOCSTI(防止终端输入注入) */ || seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(ioctl), 1, SCMP_A1(SCMP_CMP_MASKED_EQ, TIOCSTI, TIOCSTI)) + /* 禁止内核密钥管理操作 */ || seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(keyctl), 0) || seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(add_key), 0) || seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(request_key), 0) + /* 禁止 ptrace(防止进程跟踪和注入) */ || seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(ptrace), 0) + /* 禁止 NUMA 内存策略操作 */ || seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(mbind), 0) || seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(migrate_pages), 0) || seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(move_pages), 0) || seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(set_mempolicy), 0) || seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(userfaultfd), 0) + /* 禁止 perf 性能监控(防止信息泄露) */ || seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(perf_event_open), 0) + /* 关闭 NO_NEW_PRIVS 位,允许 seccomp 过滤器生效 */ || seccomp_attr_set(ctx, SCMP_FLTATR_CTL_NNP, 0) + /* 加载过滤器到内核 */ || seccomp_load(ctx)) { if (ctx) seccomp_release(ctx); fprintf(stderr, "failed: %m\n"); diff --git a/src/userns.c b/src/userns.c index b5b2e68..7be46b6 100644 --- a/src/userns.c +++ b/src/userns.c @@ -1,15 +1,35 @@ +/* userns.c — User Namespace 模块 + * + * 负责: + * 1. 子进程中 unshare(CLONE_NEWUSER) 创建用户命名空间 + * 2. 父进程通过 /proc//uid_map 和 gid_map 写入映射 + * 容器内 UID 0 → 宿主机 UID 10000(共 2000 个) + * 3. 子进程切换到指定的 UID/GID + * + * 父子进程通过 socketpair 同步: + * 子进程通知父进程是否成功创建 user namespace, + * 父进程写完映射后通知子进程继续执行。 + */ + #include "container.h" #include "userns.h" +/* 父进程:写入 UID/GID 映射 */ int handle_child_uid_map(pid_t child_pid, int fd) { int uid_map = 0; int has_userns = -1; + + /* 读取子进程是否成功创建了 user namespace */ if (read(fd, &has_userns, sizeof(has_userns)) != sizeof(has_userns)) { fprintf(stderr, "couldn't read from child!\n"); return -1; } + if (has_userns) { + /* 写入 uid_map 和 gid_map + * 格式:"0 <宿主机起始UID> <数量>" + * 表示容器内 UID 0 映射到宿主机 UID 10000,共 2000 个 */ char path[PATH_MAX] = {0}; for (char **file = (char *[]) { "uid_map", "gid_map", 0 }; *file; file++) { if (snprintf(path, sizeof(path), "/proc/%d/%s", child_pid, *file) @@ -30,6 +50,8 @@ int handle_child_uid_map(pid_t child_pid, int fd) close(uid_map); } } + + /* 通知子进程:映射已完成,可以继续执行 */ if (write(fd, & (int) { 0 }, sizeof(int)) != sizeof(int)) { fprintf(stderr, "couldn't write: %m\n"); return -1; @@ -37,25 +59,37 @@ int handle_child_uid_map(pid_t child_pid, int fd) return 0; } +/* 子进程:创建 user namespace 并切换 UID/GID */ int userns(struct child_config *config) { + /* 尝试创建 user namespace + * 需要内核 >= 4.8 或 ubuntu 特定补丁才支持 unprivileged userns */ fprintf(stderr, "=> trying a user namespace..."); int has_userns = !unshare(CLONE_NEWUSER); + + /* 通知父进程是否成功创建了 user namespace */ if (write(config->fd, &has_userns, sizeof(has_userns)) != sizeof(has_userns)) { fprintf(stderr, "couldn't write: %m\n"); return -1; } + + /* 等待父进程完成 uid_map/gid_map 写入 */ int result = 0; if (read(config->fd, &result, sizeof(result)) != sizeof(result)) { fprintf(stderr, "couldn't read: %m\n"); return -1; } if (result) return -1; + if (has_userns) { fprintf(stderr, "done.\n"); } else { fprintf(stderr, "unsupported? continuing.\n"); } + + /* 切换到指定的 UID/GID + * setgroups: 清除 supplementary groups(需要 user namespace) + * setresgid/setresgid: 设置 real/effective/saved GID/UID */ fprintf(stderr, "=> switching to uid %d / gid %d...", config->uid, config->uid); if (setgroups(1, & (gid_t) { config->uid }) || setresgid(config->uid, config->uid, config->uid) ||