Add comments to all source and header files
- include/container.h: document struct fields and constants - src/main.c: explain argument parsing, clone flow, parent/child roles - src/mount.c: describe 6-step mount isolation process - src/cgroup.c: explain cgroup data structures and resource setup/cleanup - src/capabilities.c: document capability drop (bounding + inheritable) - src/seccomp.c: explain each seccomp rule's security purpose - src/userns.c: describe parent-child sync and uid_map/gid_map format - src/hostname.c: document tarot card random naming scheme
This commit is contained in:
+18
-6
@@ -1,3 +1,9 @@
|
|||||||
|
/* container.h — 容器核心共享定义
|
||||||
|
*
|
||||||
|
* 包含所有模块共用的结构体、常量和系统头文件。
|
||||||
|
* 每个模块的 .c 文件都 include 此头文件。
|
||||||
|
*/
|
||||||
|
|
||||||
#ifndef CONTAINER_H
|
#ifndef CONTAINER_H
|
||||||
#define CONTAINER_H
|
#define CONTAINER_H
|
||||||
|
|
||||||
@@ -24,18 +30,24 @@
|
|||||||
#include <linux/capability.h>
|
#include <linux/capability.h>
|
||||||
#include <linux/limits.h>
|
#include <linux/limits.h>
|
||||||
|
|
||||||
|
/* clone() 子进程栈大小 */
|
||||||
#define STACK_SIZE (1024 * 1024)
|
#define STACK_SIZE (1024 * 1024)
|
||||||
|
|
||||||
|
/* User Namespace UID/GID 映射:容器内 0 映射到宿主机 10000~11999 */
|
||||||
#define USERNS_OFFSET 10000
|
#define USERNS_OFFSET 10000
|
||||||
#define USERNS_COUNT 2000
|
#define USERNS_COUNT 2000
|
||||||
|
|
||||||
|
/* 容器内最大打开文件描述符数 */
|
||||||
#define FD_COUNT 64
|
#define FD_COUNT 64
|
||||||
|
|
||||||
|
/* 子进程配置,由 main() 填充,传递给 child() */
|
||||||
struct child_config {
|
struct child_config {
|
||||||
int argc;
|
int argc; /* -c 后面命令的参数个数 */
|
||||||
uid_t uid;
|
uid_t uid; /* -u 指定的容器内 UID */
|
||||||
int fd;
|
int fd; /* 父子进程通信用的 socket fd */
|
||||||
char *hostname;
|
char *hostname; /* 容器主机名 */
|
||||||
char **argv;
|
char **argv; /* -c 后面的命令及参数 */
|
||||||
char *mount_dir;
|
char *mount_dir; /* -m 指定的 rootfs 路径 */
|
||||||
};
|
};
|
||||||
|
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
+39
-20
@@ -1,3 +1,14 @@
|
|||||||
|
/* capabilities.c — Linux Capabilities 权限收窄模块
|
||||||
|
*
|
||||||
|
* 将 root 的"万能权限"拆解为 30+ 个细粒度的 capability,
|
||||||
|
* 本模块 drop 掉容器不需要的 20 个危险 capability,
|
||||||
|
* 只保留最小权限集。
|
||||||
|
*
|
||||||
|
* 两步操作:
|
||||||
|
* 1. prctl(PR_CAPBSET_DROP) — 从 bounding set 移除,子进程也无法获取
|
||||||
|
* 2. cap_set_flag(CAP_INHERITABLE, CAP_CLEAR) — 清除 inheritable 集
|
||||||
|
*/
|
||||||
|
|
||||||
#include <stdio.h>
|
#include <stdio.h>
|
||||||
#include <sys/capability.h>
|
#include <sys/capability.h>
|
||||||
#include <sys/prctl.h>
|
#include <sys/prctl.h>
|
||||||
@@ -7,29 +18,34 @@
|
|||||||
int capabilities(void)
|
int capabilities(void)
|
||||||
{
|
{
|
||||||
fprintf(stderr, "=> dropping capabilities...");
|
fprintf(stderr, "=> dropping capabilities...");
|
||||||
|
/* 需要 drop 的危险 capability 列表 */
|
||||||
int drop_caps[] = {
|
int drop_caps[] = {
|
||||||
CAP_AUDIT_CONTROL,
|
CAP_AUDIT_CONTROL, /* 修改审计日志 */
|
||||||
CAP_AUDIT_READ,
|
CAP_AUDIT_READ, /* 读取审计日志 */
|
||||||
CAP_AUDIT_WRITE,
|
CAP_AUDIT_WRITE, /* 写入审计日志 */
|
||||||
CAP_BLOCK_SUSPEND,
|
CAP_BLOCK_SUSPEND, /* 阻止系统挂起 */
|
||||||
CAP_DAC_READ_SEARCH,
|
CAP_DAC_READ_SEARCH, /* 绕过文件读权限检查 */
|
||||||
CAP_FSETID,
|
CAP_FSETID, /* 设置 setuid/setgid 位 */
|
||||||
CAP_IPC_LOCK,
|
CAP_IPC_LOCK, /* 锁定共享内存 */
|
||||||
CAP_MAC_ADMIN,
|
CAP_MAC_ADMIN, /* 修改 MAC 策略(如 SELinux) */
|
||||||
CAP_MAC_OVERRIDE,
|
CAP_MAC_OVERRIDE, /* 覆盖 MAC 策略 */
|
||||||
CAP_MKNOD,
|
CAP_MKNOD, /* 创建设备文件 */
|
||||||
CAP_SETFCAP,
|
CAP_SETFCAP, /* 设置文件 capability */
|
||||||
CAP_SYSLOG,
|
CAP_SYSLOG, /* 操作内核日志 */
|
||||||
CAP_SYS_ADMIN,
|
CAP_SYS_ADMIN, /* 挂载、namespace 等大量特权操作 */
|
||||||
CAP_SYS_BOOT,
|
CAP_SYS_BOOT, /* 重启系统 */
|
||||||
CAP_SYS_MODULE,
|
CAP_SYS_MODULE, /* 加载/卸载内核模块 */
|
||||||
CAP_SYS_NICE,
|
CAP_SYS_NICE, /* 调整进程优先级 */
|
||||||
CAP_SYS_RAWIO,
|
CAP_SYS_RAWIO, /* 直接 I/O 端口操作 */
|
||||||
CAP_SYS_RESOURCE,
|
CAP_SYS_RESOURCE, /* 修改资源限制 */
|
||||||
CAP_SYS_TIME,
|
CAP_SYS_TIME, /* 修改系统时间 */
|
||||||
CAP_WAKE_ALARM
|
CAP_WAKE_ALARM /* 设置唤醒闹钟 */
|
||||||
};
|
};
|
||||||
size_t num_caps = sizeof(drop_caps) / sizeof(*drop_caps);
|
size_t num_caps = sizeof(drop_caps) / sizeof(*drop_caps);
|
||||||
|
|
||||||
|
/* 第一步:从 bounding set 中移除
|
||||||
|
* bounding set 限制了进程能获取的最大能力集,
|
||||||
|
* 一旦 drop,即使后续 execve setuid 程序也无法获取 */
|
||||||
fprintf(stderr, "bounding...");
|
fprintf(stderr, "bounding...");
|
||||||
for (size_t i = 0; i < num_caps; i++) {
|
for (size_t i = 0; i < num_caps; i++) {
|
||||||
if (prctl(PR_CAPBSET_DROP, drop_caps[i], 0, 0, 0)) {
|
if (prctl(PR_CAPBSET_DROP, drop_caps[i], 0, 0, 0)) {
|
||||||
@@ -37,6 +53,9 @@ int capabilities(void)
|
|||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* 第二步:从 inheritable set 中清除
|
||||||
|
* inheritable set 控制 execve 后哪些 capability 可被继承 */
|
||||||
fprintf(stderr, "inheritable...");
|
fprintf(stderr, "inheritable...");
|
||||||
cap_t caps = NULL;
|
cap_t caps = NULL;
|
||||||
if (!(caps = cap_get_proc())
|
if (!(caps = cap_get_proc())
|
||||||
|
|||||||
+32
-7
@@ -1,25 +1,40 @@
|
|||||||
|
/* cgroup.c — Cgroup 资源限制模块
|
||||||
|
*
|
||||||
|
* 通过 cgroup v1 限制容器能使用的资源:
|
||||||
|
* - memory: 内存限制 1GB
|
||||||
|
* - cpu: CPU 时间片权重 256
|
||||||
|
* - pids: 最大进程数 64
|
||||||
|
*
|
||||||
|
* 容器退出后清理创建的 cgroup 目录。
|
||||||
|
*/
|
||||||
|
|
||||||
#include "container.h"
|
#include "container.h"
|
||||||
#include "cgroup.h"
|
#include "cgroup.h"
|
||||||
|
|
||||||
#define MEMORY "1073741824"
|
/* 资源限制值 */
|
||||||
#define SHARES "256"
|
#define MEMORY "1073741824" /* 1GB 内存限制 */
|
||||||
#define PIDS "64"
|
#define SHARES "256" /* CPU 时间片权重 */
|
||||||
#define WEIGHT "10"
|
#define PIDS "64" /* 最大进程数 */
|
||||||
|
#define WEIGHT "10" /* I/O 权重(备用) */
|
||||||
|
|
||||||
|
/* cgroup 控制器结构体:描述一个 cgroup 子系统及其配置项 */
|
||||||
struct cgrp_control {
|
struct cgrp_control {
|
||||||
char control[256];
|
char control[256]; /* cgroup 子系统名,如 "memory"、"cpu" */
|
||||||
struct cgrp_setting {
|
struct cgrp_setting {
|
||||||
char name[256];
|
char name[256]; /* 配置文件名,如 "memory.limit_in_bytes" */
|
||||||
char value[256];
|
char value[256]; /* 写入的值 */
|
||||||
} **settings;
|
} **settings;
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/* 通用配置:将当前进程加入 cgroup 的 tasks 文件 */
|
||||||
static struct cgrp_setting add_to_tasks = {
|
static struct cgrp_setting add_to_tasks = {
|
||||||
.name = "tasks",
|
.name = "tasks",
|
||||||
.value = "0"
|
.value = "0"
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/* cgroup 配置数组:定义要创建的 cgroup 及其限制 */
|
||||||
static struct cgrp_control *cgrps[] = {
|
static struct cgrp_control *cgrps[] = {
|
||||||
|
/* 内存限制:1GB */
|
||||||
& (struct cgrp_control) {
|
& (struct cgrp_control) {
|
||||||
.control = "memory",
|
.control = "memory",
|
||||||
.settings = (struct cgrp_setting *[]) {
|
.settings = (struct cgrp_setting *[]) {
|
||||||
@@ -35,6 +50,7 @@ static struct cgrp_control *cgrps[] = {
|
|||||||
NULL
|
NULL
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
|
/* CPU 权重:256(默认 1024,值越小分到的 CPU 越少) */
|
||||||
& (struct cgrp_control) {
|
& (struct cgrp_control) {
|
||||||
.control = "cpu",
|
.control = "cpu",
|
||||||
.settings = (struct cgrp_setting *[]) {
|
.settings = (struct cgrp_setting *[]) {
|
||||||
@@ -46,6 +62,7 @@ static struct cgrp_control *cgrps[] = {
|
|||||||
NULL
|
NULL
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
|
/* 进程数限制:最多 64 个进程 */
|
||||||
& (struct cgrp_control) {
|
& (struct cgrp_control) {
|
||||||
.control = "pids",
|
.control = "pids",
|
||||||
.settings = (struct cgrp_setting *[]) {
|
.settings = (struct cgrp_setting *[]) {
|
||||||
@@ -60,12 +77,14 @@ static struct cgrp_control *cgrps[] = {
|
|||||||
NULL
|
NULL
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/* 创建 cgroup 目录并写入资源限制值 */
|
||||||
int resources(struct child_config *config)
|
int resources(struct child_config *config)
|
||||||
{
|
{
|
||||||
fprintf(stderr, "=> setting cgroups...");
|
fprintf(stderr, "=> setting cgroups...");
|
||||||
for (struct cgrp_control **cgrp = cgrps; *cgrp; cgrp++) {
|
for (struct cgrp_control **cgrp = cgrps; *cgrp; cgrp++) {
|
||||||
char dir[PATH_MAX] = {0};
|
char dir[PATH_MAX] = {0};
|
||||||
fprintf(stderr, "%s...", (*cgrp)->control);
|
fprintf(stderr, "%s...", (*cgrp)->control);
|
||||||
|
/* 创建 /sys/fs/cgroup/<子系统>/<主机名>/ 目录 */
|
||||||
if (snprintf(dir, sizeof(dir), "/sys/fs/cgroup/%s/%s",
|
if (snprintf(dir, sizeof(dir), "/sys/fs/cgroup/%s/%s",
|
||||||
(*cgrp)->control, config->hostname) == -1) {
|
(*cgrp)->control, config->hostname) == -1) {
|
||||||
return -1;
|
return -1;
|
||||||
@@ -74,6 +93,7 @@ int resources(struct child_config *config)
|
|||||||
fprintf(stderr, "mkdir %s failed: %m\n", dir);
|
fprintf(stderr, "mkdir %s failed: %m\n", dir);
|
||||||
return -1;
|
return -1;
|
||||||
}
|
}
|
||||||
|
/* 遍历该子系统的所有配置项,逐个写入 */
|
||||||
for (struct cgrp_setting **setting = (*cgrp)->settings; *setting; setting++) {
|
for (struct cgrp_setting **setting = (*cgrp)->settings; *setting; setting++) {
|
||||||
char path[PATH_MAX] = {0};
|
char path[PATH_MAX] = {0};
|
||||||
int fd = 0;
|
int fd = 0;
|
||||||
@@ -95,6 +115,8 @@ int resources(struct child_config *config)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
fprintf(stderr, "done.\n");
|
fprintf(stderr, "done.\n");
|
||||||
|
|
||||||
|
/* 设置文件描述符限制(rlimit) */
|
||||||
fprintf(stderr, "=> setting rlimit...");
|
fprintf(stderr, "=> setting rlimit...");
|
||||||
if (setrlimit(RLIMIT_NOFILE,
|
if (setrlimit(RLIMIT_NOFILE,
|
||||||
& (struct rlimit) {
|
& (struct rlimit) {
|
||||||
@@ -108,6 +130,7 @@ int resources(struct child_config *config)
|
|||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* 清理 cgroup 目录:先把进程移回根 cgroup,再删除目录 */
|
||||||
int free_resources(struct child_config *config)
|
int free_resources(struct child_config *config)
|
||||||
{
|
{
|
||||||
fprintf(stderr, "=> cleaning cgroups...");
|
fprintf(stderr, "=> cleaning cgroups...");
|
||||||
@@ -122,6 +145,7 @@ int free_resources(struct child_config *config)
|
|||||||
fprintf(stderr, "snprintf failed: %m\n");
|
fprintf(stderr, "snprintf failed: %m\n");
|
||||||
return -1;
|
return -1;
|
||||||
}
|
}
|
||||||
|
/* 将当前进程写入根 cgroup 的 tasks,从容器 cgroup 中移出 */
|
||||||
if ((task_fd = open(task, O_WRONLY)) == -1) {
|
if ((task_fd = open(task, O_WRONLY)) == -1) {
|
||||||
fprintf(stderr, "opening %s failed: %m\n", task);
|
fprintf(stderr, "opening %s failed: %m\n", task);
|
||||||
return -1;
|
return -1;
|
||||||
@@ -132,6 +156,7 @@ int free_resources(struct child_config *config)
|
|||||||
return -1;
|
return -1;
|
||||||
}
|
}
|
||||||
close(task_fd);
|
close(task_fd);
|
||||||
|
/* 删除容器的 cgroup 目录 */
|
||||||
if (rmdir(dir)) {
|
if (rmdir(dir)) {
|
||||||
fprintf(stderr, "rmdir %s failed: %m", dir);
|
fprintf(stderr, "rmdir %s failed: %m", dir);
|
||||||
return -1;
|
return -1;
|
||||||
|
|||||||
+18
-1
@@ -1,26 +1,43 @@
|
|||||||
|
/* hostname.c — 随机主机名生成模块
|
||||||
|
*
|
||||||
|
* 基于当前时间戳的纳秒部分,从 78 张塔罗牌中随机选取一张作为主机名。
|
||||||
|
* 格式:<5位时间戳十六进制>-<塔罗牌名>
|
||||||
|
* 例如:1204f86c-king-of-cups
|
||||||
|
*
|
||||||
|
* 78 张牌 = 22 张大阿卡纳 + 56 张小阿卡纳(4花色×14数字)
|
||||||
|
*/
|
||||||
|
|
||||||
#include <stdio.h>
|
#include <stdio.h>
|
||||||
#include <time.h>
|
#include <time.h>
|
||||||
#include "hostname.h"
|
#include "hostname.h"
|
||||||
|
|
||||||
int choose_hostname(char *buff, size_t len)
|
int choose_hostname(char *buff, size_t len)
|
||||||
{
|
{
|
||||||
|
/* 小阿卡纳四花色 */
|
||||||
static const char *suits[] = { "swords", "wands", "pentacles", "cups" };
|
static const char *suits[] = { "swords", "wands", "pentacles", "cups" };
|
||||||
|
/* 小阿卡纳 14 张数字牌 */
|
||||||
static const char *minor[] = {
|
static const char *minor[] = {
|
||||||
"ace", "two", "three", "four", "five", "six", "seven", "eight",
|
"ace", "two", "three", "four", "five", "six", "seven", "eight",
|
||||||
"nine", "ten", "page", "knight", "queen", "king"
|
"nine", "ten", "page", "knight", "queen", "king"
|
||||||
};
|
};
|
||||||
|
/* 大阿卡纳 22 张 */
|
||||||
static const char *major[] = {
|
static const char *major[] = {
|
||||||
"fool", "magician", "high-priestess", "empress", "emperor",
|
"fool", "magician", "high-priestess", "empress", "emperor",
|
||||||
"hierophant", "lovers", "chariot", "strength", "hermit",
|
"hierophant", "lovers", "chariot", "strength", "hermit",
|
||||||
"wheel", "justice", "hanged-man", "death", "temperance",
|
"wheel", "justice", "hanged-man", "death", "temperance",
|
||||||
"devil", "tower", "star", "moon", "sun", "judgment", "world"
|
"devil", "tower", "star", "moon", "sun", "judgment", "world"
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/* 用单调时钟的纳秒部分做随机种子 */
|
||||||
struct timespec now = {0};
|
struct timespec now = {0};
|
||||||
clock_gettime(CLOCK_MONOTONIC, &now);
|
clock_gettime(CLOCK_MONOTONIC, &now);
|
||||||
size_t ix = now.tv_nsec % 78;
|
size_t ix = now.tv_nsec % 78; /* 78 张牌 */
|
||||||
|
|
||||||
if (ix < sizeof(major) / sizeof(*major)) {
|
if (ix < sizeof(major) / sizeof(*major)) {
|
||||||
|
/* 0~21:大阿卡纳 */
|
||||||
snprintf(buff, len, "%05lx-%s", now.tv_sec, major[ix]);
|
snprintf(buff, len, "%05lx-%s", now.tv_sec, major[ix]);
|
||||||
} else {
|
} else {
|
||||||
|
/* 22~77:小阿卡纳(数字 + 花色) */
|
||||||
ix -= sizeof(major) / sizeof(*major);
|
ix -= sizeof(major) / sizeof(*major);
|
||||||
snprintf(buff, len,
|
snprintf(buff, len,
|
||||||
"%05lxc-%s-of-%s",
|
"%05lxc-%s-of-%s",
|
||||||
|
|||||||
+41
-7
@@ -1,3 +1,14 @@
|
|||||||
|
/* main.c — 容器主控流程
|
||||||
|
*
|
||||||
|
* 负责:
|
||||||
|
* 1. 解析命令行参数(-m、-u、-c)
|
||||||
|
* 2. 校验内核版本
|
||||||
|
* 3. 创建 cgroup 资源限制
|
||||||
|
* 4. clone() 创建带 namespace 的子进程
|
||||||
|
* 5. 父进程写入 UID 映射并等待子进程退出
|
||||||
|
* 6. 子进程依次执行隔离初始化后 execve 目标命令
|
||||||
|
*/
|
||||||
|
|
||||||
#include "container.h"
|
#include "container.h"
|
||||||
#include "mount.h"
|
#include "mount.h"
|
||||||
#include "cgroup.h"
|
#include "cgroup.h"
|
||||||
@@ -6,9 +17,11 @@
|
|||||||
#include "userns.h"
|
#include "userns.h"
|
||||||
#include "hostname.h"
|
#include "hostname.h"
|
||||||
|
|
||||||
|
/* 子进程入口函数,由 clone() 调用 */
|
||||||
int child(void *arg)
|
int child(void *arg)
|
||||||
{
|
{
|
||||||
struct child_config *config = arg;
|
struct child_config *config = arg;
|
||||||
|
/* 按顺序执行容器初始化:设置主机名 → 文件系统隔离 → 用户命名空间 → 权限收窄 → syscall 过滤 */
|
||||||
if (sethostname(config->hostname, strlen(config->hostname))
|
if (sethostname(config->hostname, strlen(config->hostname))
|
||||||
|| mounts(config)
|
|| mounts(config)
|
||||||
|| userns(config)
|
|| userns(config)
|
||||||
@@ -17,10 +30,12 @@ int child(void *arg)
|
|||||||
close(config->fd);
|
close(config->fd);
|
||||||
return -1;
|
return -1;
|
||||||
}
|
}
|
||||||
|
/* 初始化完成,关闭与父进程的通信 socket */
|
||||||
if (close(config->fd)) {
|
if (close(config->fd)) {
|
||||||
fprintf(stderr, "close failed: %m\n");
|
fprintf(stderr, "close failed: %m\n");
|
||||||
return -1;
|
return -1;
|
||||||
}
|
}
|
||||||
|
/* 替换为用户指定的进程 */
|
||||||
if (execve(config->argv[0], config->argv, NULL)) {
|
if (execve(config->argv[0], config->argv, NULL)) {
|
||||||
fprintf(stderr, "execve failed! %m.\n");
|
fprintf(stderr, "execve failed! %m.\n");
|
||||||
return -1;
|
return -1;
|
||||||
@@ -33,12 +48,15 @@ int main(int argc, char **argv)
|
|||||||
struct child_config config = {0};
|
struct child_config config = {0};
|
||||||
int err = 0;
|
int err = 0;
|
||||||
int option = 0;
|
int option = 0;
|
||||||
int sockets[2] = {0};
|
int sockets[2] = {0}; /* 父子进程通信用的 socket 对 */
|
||||||
pid_t child_pid = 0;
|
pid_t child_pid = 0;
|
||||||
int last_optind = 0;
|
int last_optind = 0;
|
||||||
|
|
||||||
|
/* 解析命令行参数:-m rootfs路径, -u UID, -c 要执行的命令 */
|
||||||
while ((option = getopt(argc, argv, "c:m:u:"))) {
|
while ((option = getopt(argc, argv, "c:m:u:"))) {
|
||||||
switch (option) {
|
switch (option) {
|
||||||
case 'c':
|
case 'c':
|
||||||
|
/* -c 必须放最后,它后面的都当作命令参数 */
|
||||||
config.argc = argc - last_optind - 1;
|
config.argc = argc - last_optind - 1;
|
||||||
config.argv = &argv[argc - config.argc];
|
config.argv = &argv[argc - config.argc];
|
||||||
goto finish_options;
|
goto finish_options;
|
||||||
@@ -60,6 +78,7 @@ finish_options:
|
|||||||
if (!config.argc) goto usage;
|
if (!config.argc) goto usage;
|
||||||
if (!config.mount_dir) goto usage;
|
if (!config.mount_dir) goto usage;
|
||||||
|
|
||||||
|
/* 校验内核版本 >= 4.7(支持 user namespace 的 unprivileged 模式) */
|
||||||
fprintf(stderr, "=> validating Linux version...");
|
fprintf(stderr, "=> validating Linux version...");
|
||||||
struct utsname host = {0};
|
struct utsname host = {0};
|
||||||
if (uname(&host)) {
|
if (uname(&host)) {
|
||||||
@@ -82,45 +101,57 @@ finish_options:
|
|||||||
}
|
}
|
||||||
fprintf(stderr, "%s on %s.\n", host.release, host.machine);
|
fprintf(stderr, "%s on %s.\n", host.release, host.machine);
|
||||||
|
|
||||||
|
/* 生成随机容器主机名(塔罗牌名) */
|
||||||
char hostname[256] = {0};
|
char hostname[256] = {0};
|
||||||
if (choose_hostname(hostname, sizeof(hostname)))
|
if (choose_hostname(hostname, sizeof(hostname)))
|
||||||
goto error;
|
goto error;
|
||||||
config.hostname = hostname;
|
config.hostname = hostname;
|
||||||
|
|
||||||
|
/* 创建 socket 对用于父子进程同步(通知 UID 映射完成) */
|
||||||
if (socketpair(AF_LOCAL, SOCK_SEQPACKET, 0, sockets)) {
|
if (socketpair(AF_LOCAL, SOCK_SEQPACKET, 0, sockets)) {
|
||||||
fprintf(stderr, "socketpair failed: %m\n");
|
fprintf(stderr, "socketpair failed: %m\n");
|
||||||
goto error;
|
goto error;
|
||||||
}
|
}
|
||||||
|
/* 父进程端设置 close-on-exec,子进程继承另一端 */
|
||||||
if (fcntl(sockets[0], F_SETFD, FD_CLOEXEC)) {
|
if (fcntl(sockets[0], F_SETFD, FD_CLOEXEC)) {
|
||||||
fprintf(stderr, "fcntl failed: %m\n");
|
fprintf(stderr, "fcntl failed: %m\n");
|
||||||
goto error;
|
goto error;
|
||||||
}
|
}
|
||||||
config.fd = sockets[1];
|
config.fd = sockets[1];
|
||||||
|
|
||||||
|
/* 分配子进程栈空间(clone 要求调用者提供栈) */
|
||||||
char *stack = 0;
|
char *stack = 0;
|
||||||
if (!(stack = malloc(STACK_SIZE))) {
|
if (!(stack = malloc(STACK_SIZE))) {
|
||||||
fprintf(stderr, "=> malloc failed, out of memory?\n");
|
fprintf(stderr, "=> malloc failed, out of memory?\n");
|
||||||
goto error;
|
goto error;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* 创建 cgroup 资源限制(内存、CPU、进程数) */
|
||||||
if (resources(&config)) {
|
if (resources(&config)) {
|
||||||
err = 1;
|
err = 1;
|
||||||
goto clear_resources;
|
goto clear_resources;
|
||||||
}
|
}
|
||||||
int flags = CLONE_NEWNS
|
|
||||||
| CLONE_NEWCGROUP
|
/* clone 子进程,创建 6 种 namespace 隔离 */
|
||||||
| CLONE_NEWPID
|
int flags = CLONE_NEWNS /* 挂载隔离 */
|
||||||
| CLONE_NEWIPC
|
| CLONE_NEWCGROUP /* cgroup 隔离 */
|
||||||
| CLONE_NEWNET
|
| CLONE_NEWPID /* 进程号隔离 */
|
||||||
| CLONE_NEWUTS;
|
| CLONE_NEWIPC /* 进程间通信隔离 */
|
||||||
|
| CLONE_NEWNET /* 网络隔离 */
|
||||||
|
| CLONE_NEWUTS; /* 主机名隔离 */
|
||||||
if ((child_pid = clone(child, stack + STACK_SIZE, flags | SIGCHLD, &config)) == -1) {
|
if ((child_pid = clone(child, stack + STACK_SIZE, flags | SIGCHLD, &config)) == -1) {
|
||||||
fprintf(stderr, "=> clone failed! %m\n");
|
fprintf(stderr, "=> clone failed! %m\n");
|
||||||
err = 1;
|
err = 1;
|
||||||
goto clear_resources;
|
goto clear_resources;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* 父进程:关闭子进程端的 socket */
|
||||||
close(sockets[1]);
|
close(sockets[1]);
|
||||||
sockets[1] = 0;
|
sockets[1] = 0;
|
||||||
close(sockets[1]);
|
close(sockets[1]);
|
||||||
sockets[1] = 0;
|
sockets[1] = 0;
|
||||||
|
|
||||||
|
/* 父进程:写入 uid_map/gid_map,完成容器内 root → 宿主机普通用户的映射 */
|
||||||
if (handle_child_uid_map(child_pid, sockets[0])) {
|
if (handle_child_uid_map(child_pid, sockets[0])) {
|
||||||
err = 1;
|
err = 1;
|
||||||
goto kill_and_finish_child;
|
goto kill_and_finish_child;
|
||||||
@@ -128,12 +159,15 @@ finish_options:
|
|||||||
|
|
||||||
goto finish_child;
|
goto finish_child;
|
||||||
kill_and_finish_child:
|
kill_and_finish_child:
|
||||||
|
/* 出错时杀死子进程 */
|
||||||
if (child_pid) kill(child_pid, SIGKILL);
|
if (child_pid) kill(child_pid, SIGKILL);
|
||||||
finish_child:;
|
finish_child:;
|
||||||
|
/* 等待子进程退出,获取退出码 */
|
||||||
int child_status = 0;
|
int child_status = 0;
|
||||||
waitpid(child_pid, &child_status, 0);
|
waitpid(child_pid, &child_status, 0);
|
||||||
err |= WEXITSTATUS(child_status);
|
err |= WEXITSTATUS(child_status);
|
||||||
clear_resources:
|
clear_resources:
|
||||||
|
/* 清理 cgroup 目录和栈空间 */
|
||||||
free_resources(&config);
|
free_resources(&config);
|
||||||
free(stack);
|
free(stack);
|
||||||
|
|
||||||
|
|||||||
+23
@@ -1,6 +1,17 @@
|
|||||||
|
/* mount.c — 文件系统隔离模块
|
||||||
|
*
|
||||||
|
* 负责容器的文件系统隔离:
|
||||||
|
* 1. MS_PRIVATE 重挂载,防止挂载事件泄漏到宿主机
|
||||||
|
* 2. bind mount 将 rootfs 挂载到临时目录
|
||||||
|
* 3. pivot_root 切换根文件系统
|
||||||
|
* 4. 卸载旧的根文件系统
|
||||||
|
* 5. 挂载 /proc(进程信息)和 /dev(设备节点)
|
||||||
|
*/
|
||||||
|
|
||||||
#include "container.h"
|
#include "container.h"
|
||||||
#include "mount.h"
|
#include "mount.h"
|
||||||
|
|
||||||
|
/* pivot_root 系统调用封装(glibc 没有提供封装函数) */
|
||||||
static int pivot_root(const char *new_root, const char *put_old)
|
static int pivot_root(const char *new_root, const char *put_old)
|
||||||
{
|
{
|
||||||
return syscall(SYS_pivot_root, new_root, put_old);
|
return syscall(SYS_pivot_root, new_root, put_old);
|
||||||
@@ -8,6 +19,8 @@ static int pivot_root(const char *new_root, const char *put_old)
|
|||||||
|
|
||||||
int mounts(struct child_config *config)
|
int mounts(struct child_config *config)
|
||||||
{
|
{
|
||||||
|
/* 第一步:将当前所有挂载点标记为 MS_PRIVATE
|
||||||
|
* 防止 mount namespace 内的挂载事件传播到宿主机 */
|
||||||
fprintf(stderr, "=> remounting everything with MS_PRIVATE...");
|
fprintf(stderr, "=> remounting everything with MS_PRIVATE...");
|
||||||
if (mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL)) {
|
if (mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL)) {
|
||||||
fprintf(stderr, "failed! %m\n");
|
fprintf(stderr, "failed! %m\n");
|
||||||
@@ -15,6 +28,7 @@ int mounts(struct child_config *config)
|
|||||||
}
|
}
|
||||||
fprintf(stderr, "remounted.\n");
|
fprintf(stderr, "remounted.\n");
|
||||||
|
|
||||||
|
/* 第二步:创建临时目录,将 rootfs bind mount 到这里 */
|
||||||
fprintf(stderr, "=> making a temp directory and a bind mount there...");
|
fprintf(stderr, "=> making a temp directory and a bind mount there...");
|
||||||
char mount_dir[] = "/tmp/tmp.XXXXXX";
|
char mount_dir[] = "/tmp/tmp.XXXXXX";
|
||||||
if (!mkdtemp(mount_dir)) {
|
if (!mkdtemp(mount_dir)) {
|
||||||
@@ -22,11 +36,13 @@ int mounts(struct child_config *config)
|
|||||||
return -1;
|
return -1;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* bind mount:将 -m 指定的 rootfs 挂载到临时目录 */
|
||||||
if (mount(config->mount_dir, mount_dir, NULL, MS_BIND | MS_PRIVATE, NULL)) {
|
if (mount(config->mount_dir, mount_dir, NULL, MS_BIND | MS_PRIVATE, NULL)) {
|
||||||
fprintf(stderr, "bind mount failed!\n");
|
fprintf(stderr, "bind mount failed!\n");
|
||||||
return -1;
|
return -1;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* 在挂载点内创建 oldroot 子目录,用于存放旧根 */
|
||||||
char inner_mount_dir[] = "/tmp/tmp.XXXXXX/oldroot.XXXXXX";
|
char inner_mount_dir[] = "/tmp/tmp.XXXXXX/oldroot.XXXXXX";
|
||||||
memcpy(inner_mount_dir, mount_dir, sizeof(mount_dir) - 1);
|
memcpy(inner_mount_dir, mount_dir, sizeof(mount_dir) - 1);
|
||||||
if (!mkdtemp(inner_mount_dir)) {
|
if (!mkdtemp(inner_mount_dir)) {
|
||||||
@@ -35,6 +51,7 @@ int mounts(struct child_config *config)
|
|||||||
}
|
}
|
||||||
fprintf(stderr, "done.\n");
|
fprintf(stderr, "done.\n");
|
||||||
|
|
||||||
|
/* 第三步:pivot_root,将根文件系统切换到 rootfs */
|
||||||
fprintf(stderr, "=> pivoting root...");
|
fprintf(stderr, "=> pivoting root...");
|
||||||
if (pivot_root(mount_dir, inner_mount_dir)) {
|
if (pivot_root(mount_dir, inner_mount_dir)) {
|
||||||
fprintf(stderr, "failed!\n");
|
fprintf(stderr, "failed!\n");
|
||||||
@@ -42,6 +59,7 @@ int mounts(struct child_config *config)
|
|||||||
}
|
}
|
||||||
fprintf(stderr, "done.\n");
|
fprintf(stderr, "done.\n");
|
||||||
|
|
||||||
|
/* 第四步:卸载旧根文件系统,彻底与宿主机文件系统断开 */
|
||||||
char *old_root_dir = basename(inner_mount_dir);
|
char *old_root_dir = basename(inner_mount_dir);
|
||||||
char old_root[sizeof(inner_mount_dir) + 1] = { "/" };
|
char old_root[sizeof(inner_mount_dir) + 1] = { "/" };
|
||||||
strcpy(&old_root[1], old_root_dir);
|
strcpy(&old_root[1], old_root_dir);
|
||||||
@@ -51,6 +69,7 @@ int mounts(struct child_config *config)
|
|||||||
fprintf(stderr, "chdir failed! %m\n");
|
fprintf(stderr, "chdir failed! %m\n");
|
||||||
return -1;
|
return -1;
|
||||||
}
|
}
|
||||||
|
/* MNT_DETACH:lazy unmount,先断开挂载点,等引用释放后再清理 */
|
||||||
if (umount2(old_root, MNT_DETACH)) {
|
if (umount2(old_root, MNT_DETACH)) {
|
||||||
fprintf(stderr, "umount failed! %m\n");
|
fprintf(stderr, "umount failed! %m\n");
|
||||||
return -1;
|
return -1;
|
||||||
@@ -61,6 +80,7 @@ int mounts(struct child_config *config)
|
|||||||
}
|
}
|
||||||
fprintf(stderr, "done.\n");
|
fprintf(stderr, "done.\n");
|
||||||
|
|
||||||
|
/* 第五步:挂载 /proc,让 ps、top 等工具能读取容器内进程信息 */
|
||||||
fprintf(stderr, "=> mounting /proc...");
|
fprintf(stderr, "=> mounting /proc...");
|
||||||
if (mount("proc", "/proc", "proc", 0, NULL)) {
|
if (mount("proc", "/proc", "proc", 0, NULL)) {
|
||||||
fprintf(stderr, "failed: %m\n");
|
fprintf(stderr, "failed: %m\n");
|
||||||
@@ -68,12 +88,15 @@ int mounts(struct child_config *config)
|
|||||||
}
|
}
|
||||||
fprintf(stderr, "done.\n");
|
fprintf(stderr, "done.\n");
|
||||||
|
|
||||||
|
/* 第六步:挂载 /dev(tmpfs),创建必要的设备节点 */
|
||||||
fprintf(stderr, "=> mounting /dev...");
|
fprintf(stderr, "=> mounting /dev...");
|
||||||
if (mount("tmpfs", "/dev", "tmpfs", MS_NOSUID | MS_STRICTATIME, "mode=755,size=65536k")) {
|
if (mount("tmpfs", "/dev", "tmpfs", MS_NOSUID | MS_STRICTATIME, "mode=755,size=65536k")) {
|
||||||
fprintf(stderr, "failed: %m\n");
|
fprintf(stderr, "failed: %m\n");
|
||||||
return -1;
|
return -1;
|
||||||
}
|
}
|
||||||
|
/* 清除 umask,确保设备节点权限正确(0666) */
|
||||||
umask(0);
|
umask(0);
|
||||||
|
/* 创建基本设备节点:null、zero、urandom */
|
||||||
if (mknod("/dev/null", S_IFCHR | 0666, makedev(1, 3))) {
|
if (mknod("/dev/null", S_IFCHR | 0666, makedev(1, 3))) {
|
||||||
fprintf(stderr, "mknod null failed: %m\n");
|
fprintf(stderr, "mknod null failed: %m\n");
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,44 +1,77 @@
|
|||||||
|
/* seccomp.c — Seccomp 系统调用过滤模块
|
||||||
|
*
|
||||||
|
* 通过 seccomp-bpf 过滤器限制容器内进程能使用的系统调用。
|
||||||
|
* 默认策略:SCMP_ACT_ALLOW(允许所有),然后逐条禁止危险操作。
|
||||||
|
* 违规的 syscall 返回 EPERM(Permission denied)。
|
||||||
|
*
|
||||||
|
* 禁止的操作包括:
|
||||||
|
* - 设置 setuid/setgid 位(chmod/fchmod/fchmodat)
|
||||||
|
* - 创建新 user namespace(防止容器逃逸)
|
||||||
|
* - 终端注入(ioctl TIOCSTI)
|
||||||
|
* - 内核密钥操作(keyctl/add_key/request_key)
|
||||||
|
* - 进程跟踪(ptrace)
|
||||||
|
* - 内存策略操作(mbind/migrate_pages/move_pages/set_mempolicy/userfaultfd)
|
||||||
|
* - 性能监控(perf_event_open)
|
||||||
|
*/
|
||||||
|
|
||||||
#include "container.h"
|
#include "container.h"
|
||||||
#include "container_seccomp.h"
|
#include "container_seccomp.h"
|
||||||
#include <seccomp.h>
|
#include <seccomp.h>
|
||||||
#include <termios.h>
|
#include <termios.h>
|
||||||
|
|
||||||
|
/* 违规行为返回 EPERM(Operation not permitted) */
|
||||||
#define SCMP_FAIL SCMP_ACT_ERRNO(EPERM)
|
#define SCMP_FAIL SCMP_ACT_ERRNO(EPERM)
|
||||||
|
|
||||||
int syscalls(void)
|
int syscalls(void)
|
||||||
{
|
{
|
||||||
scmp_filter_ctx ctx = NULL;
|
scmp_filter_ctx ctx = NULL;
|
||||||
fprintf(stderr, "=> filtering syscalls...");
|
fprintf(stderr, "=> filtering syscalls...");
|
||||||
|
|
||||||
if (!(ctx = seccomp_init(SCMP_ACT_ALLOW))
|
if (!(ctx = seccomp_init(SCMP_ACT_ALLOW))
|
||||||
|
/* 禁止 chmod 设置 SUID 位(防止提权) */
|
||||||
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(chmod), 1,
|
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(chmod), 1,
|
||||||
SCMP_A1(SCMP_CMP_MASKED_EQ, S_ISUID, S_ISUID))
|
SCMP_A1(SCMP_CMP_MASKED_EQ, S_ISUID, S_ISUID))
|
||||||
|
/* 禁止 chmod 设置 SGID 位 */
|
||||||
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(chmod), 1,
|
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(chmod), 1,
|
||||||
SCMP_A1(SCMP_CMP_MASKED_EQ, S_ISGID, S_ISGID))
|
SCMP_A1(SCMP_CMP_MASKED_EQ, S_ISGID, S_ISGID))
|
||||||
|
/* 禁止 fchmod 设置 SUID 位 */
|
||||||
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(fchmod), 1,
|
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(fchmod), 1,
|
||||||
SCMP_A1(SCMP_CMP_MASKED_EQ, S_ISUID, S_ISUID))
|
SCMP_A1(SCMP_CMP_MASKED_EQ, S_ISUID, S_ISUID))
|
||||||
|
/* 禁止 fchmod 设置 SGID 位 */
|
||||||
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(fchmod), 1,
|
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(fchmod), 1,
|
||||||
SCMP_A1(SCMP_CMP_MASKED_EQ, S_ISGID, S_ISGID))
|
SCMP_A1(SCMP_CMP_MASKED_EQ, S_ISGID, S_ISGID))
|
||||||
|
/* 禁止 fchmodat 设置 SUID 位 */
|
||||||
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(fchmodat), 1,
|
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(fchmodat), 1,
|
||||||
SCMP_A2(SCMP_CMP_MASKED_EQ, S_ISUID, S_ISUID))
|
SCMP_A2(SCMP_CMP_MASKED_EQ, S_ISUID, S_ISUID))
|
||||||
|
/* 禁止 fchmodat 设置 SGID 位 */
|
||||||
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(fchmodat), 1,
|
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(fchmodat), 1,
|
||||||
SCMP_A2(SCMP_CMP_MASKED_EQ, S_ISGID, S_ISGID))
|
SCMP_A2(SCMP_CMP_MASKED_EQ, S_ISGID, S_ISGID))
|
||||||
|
/* 禁止 unshare 创建新 user namespace(防止逃逸) */
|
||||||
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(unshare), 1,
|
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(unshare), 1,
|
||||||
SCMP_A0(SCMP_CMP_MASKED_EQ, CLONE_NEWUSER, CLONE_NEWUSER))
|
SCMP_A0(SCMP_CMP_MASKED_EQ, CLONE_NEWUSER, CLONE_NEWUSER))
|
||||||
|
/* 禁止 clone 创建新 user namespace */
|
||||||
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(clone), 1,
|
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(clone), 1,
|
||||||
SCMP_A0(SCMP_CMP_MASKED_EQ, CLONE_NEWUSER, CLONE_NEWUSER))
|
SCMP_A0(SCMP_CMP_MASKED_EQ, CLONE_NEWUSER, CLONE_NEWUSER))
|
||||||
|
/* 禁止 ioctl TIOCSTI(防止终端输入注入) */
|
||||||
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(ioctl), 1,
|
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(ioctl), 1,
|
||||||
SCMP_A1(SCMP_CMP_MASKED_EQ, TIOCSTI, TIOCSTI))
|
SCMP_A1(SCMP_CMP_MASKED_EQ, TIOCSTI, TIOCSTI))
|
||||||
|
/* 禁止内核密钥管理操作 */
|
||||||
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(keyctl), 0)
|
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(keyctl), 0)
|
||||||
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(add_key), 0)
|
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(add_key), 0)
|
||||||
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(request_key), 0)
|
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(request_key), 0)
|
||||||
|
/* 禁止 ptrace(防止进程跟踪和注入) */
|
||||||
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(ptrace), 0)
|
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(ptrace), 0)
|
||||||
|
/* 禁止 NUMA 内存策略操作 */
|
||||||
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(mbind), 0)
|
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(mbind), 0)
|
||||||
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(migrate_pages), 0)
|
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(migrate_pages), 0)
|
||||||
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(move_pages), 0)
|
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(move_pages), 0)
|
||||||
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(set_mempolicy), 0)
|
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(set_mempolicy), 0)
|
||||||
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(userfaultfd), 0)
|
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(userfaultfd), 0)
|
||||||
|
/* 禁止 perf 性能监控(防止信息泄露) */
|
||||||
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(perf_event_open), 0)
|
|| seccomp_rule_add(ctx, SCMP_FAIL, SCMP_SYS(perf_event_open), 0)
|
||||||
|
/* 关闭 NO_NEW_PRIVS 位,允许 seccomp 过滤器生效 */
|
||||||
|| seccomp_attr_set(ctx, SCMP_FLTATR_CTL_NNP, 0)
|
|| seccomp_attr_set(ctx, SCMP_FLTATR_CTL_NNP, 0)
|
||||||
|
/* 加载过滤器到内核 */
|
||||||
|| seccomp_load(ctx)) {
|
|| seccomp_load(ctx)) {
|
||||||
if (ctx) seccomp_release(ctx);
|
if (ctx) seccomp_release(ctx);
|
||||||
fprintf(stderr, "failed: %m\n");
|
fprintf(stderr, "failed: %m\n");
|
||||||
|
|||||||
@@ -1,15 +1,35 @@
|
|||||||
|
/* userns.c — User Namespace 模块
|
||||||
|
*
|
||||||
|
* 负责:
|
||||||
|
* 1. 子进程中 unshare(CLONE_NEWUSER) 创建用户命名空间
|
||||||
|
* 2. 父进程通过 /proc/<pid>/uid_map 和 gid_map 写入映射
|
||||||
|
* 容器内 UID 0 → 宿主机 UID 10000(共 2000 个)
|
||||||
|
* 3. 子进程切换到指定的 UID/GID
|
||||||
|
*
|
||||||
|
* 父子进程通过 socketpair 同步:
|
||||||
|
* 子进程通知父进程是否成功创建 user namespace,
|
||||||
|
* 父进程写完映射后通知子进程继续执行。
|
||||||
|
*/
|
||||||
|
|
||||||
#include "container.h"
|
#include "container.h"
|
||||||
#include "userns.h"
|
#include "userns.h"
|
||||||
|
|
||||||
|
/* 父进程:写入 UID/GID 映射 */
|
||||||
int handle_child_uid_map(pid_t child_pid, int fd)
|
int handle_child_uid_map(pid_t child_pid, int fd)
|
||||||
{
|
{
|
||||||
int uid_map = 0;
|
int uid_map = 0;
|
||||||
int has_userns = -1;
|
int has_userns = -1;
|
||||||
|
|
||||||
|
/* 读取子进程是否成功创建了 user namespace */
|
||||||
if (read(fd, &has_userns, sizeof(has_userns)) != sizeof(has_userns)) {
|
if (read(fd, &has_userns, sizeof(has_userns)) != sizeof(has_userns)) {
|
||||||
fprintf(stderr, "couldn't read from child!\n");
|
fprintf(stderr, "couldn't read from child!\n");
|
||||||
return -1;
|
return -1;
|
||||||
}
|
}
|
||||||
|
|
||||||
if (has_userns) {
|
if (has_userns) {
|
||||||
|
/* 写入 uid_map 和 gid_map
|
||||||
|
* 格式:"0 <宿主机起始UID> <数量>"
|
||||||
|
* 表示容器内 UID 0 映射到宿主机 UID 10000,共 2000 个 */
|
||||||
char path[PATH_MAX] = {0};
|
char path[PATH_MAX] = {0};
|
||||||
for (char **file = (char *[]) { "uid_map", "gid_map", 0 }; *file; file++) {
|
for (char **file = (char *[]) { "uid_map", "gid_map", 0 }; *file; file++) {
|
||||||
if (snprintf(path, sizeof(path), "/proc/%d/%s", child_pid, *file)
|
if (snprintf(path, sizeof(path), "/proc/%d/%s", child_pid, *file)
|
||||||
@@ -30,6 +50,8 @@ int handle_child_uid_map(pid_t child_pid, int fd)
|
|||||||
close(uid_map);
|
close(uid_map);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* 通知子进程:映射已完成,可以继续执行 */
|
||||||
if (write(fd, & (int) { 0 }, sizeof(int)) != sizeof(int)) {
|
if (write(fd, & (int) { 0 }, sizeof(int)) != sizeof(int)) {
|
||||||
fprintf(stderr, "couldn't write: %m\n");
|
fprintf(stderr, "couldn't write: %m\n");
|
||||||
return -1;
|
return -1;
|
||||||
@@ -37,25 +59,37 @@ int handle_child_uid_map(pid_t child_pid, int fd)
|
|||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* 子进程:创建 user namespace 并切换 UID/GID */
|
||||||
int userns(struct child_config *config)
|
int userns(struct child_config *config)
|
||||||
{
|
{
|
||||||
|
/* 尝试创建 user namespace
|
||||||
|
* 需要内核 >= 4.8 或 ubuntu 特定补丁才支持 unprivileged userns */
|
||||||
fprintf(stderr, "=> trying a user namespace...");
|
fprintf(stderr, "=> trying a user namespace...");
|
||||||
int has_userns = !unshare(CLONE_NEWUSER);
|
int has_userns = !unshare(CLONE_NEWUSER);
|
||||||
|
|
||||||
|
/* 通知父进程是否成功创建了 user namespace */
|
||||||
if (write(config->fd, &has_userns, sizeof(has_userns)) != sizeof(has_userns)) {
|
if (write(config->fd, &has_userns, sizeof(has_userns)) != sizeof(has_userns)) {
|
||||||
fprintf(stderr, "couldn't write: %m\n");
|
fprintf(stderr, "couldn't write: %m\n");
|
||||||
return -1;
|
return -1;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* 等待父进程完成 uid_map/gid_map 写入 */
|
||||||
int result = 0;
|
int result = 0;
|
||||||
if (read(config->fd, &result, sizeof(result)) != sizeof(result)) {
|
if (read(config->fd, &result, sizeof(result)) != sizeof(result)) {
|
||||||
fprintf(stderr, "couldn't read: %m\n");
|
fprintf(stderr, "couldn't read: %m\n");
|
||||||
return -1;
|
return -1;
|
||||||
}
|
}
|
||||||
if (result) return -1;
|
if (result) return -1;
|
||||||
|
|
||||||
if (has_userns) {
|
if (has_userns) {
|
||||||
fprintf(stderr, "done.\n");
|
fprintf(stderr, "done.\n");
|
||||||
} else {
|
} else {
|
||||||
fprintf(stderr, "unsupported? continuing.\n");
|
fprintf(stderr, "unsupported? continuing.\n");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* 切换到指定的 UID/GID
|
||||||
|
* setgroups: 清除 supplementary groups(需要 user namespace)
|
||||||
|
* setresgid/setresgid: 设置 real/effective/saved GID/UID */
|
||||||
fprintf(stderr, "=> switching to uid %d / gid %d...", config->uid, config->uid);
|
fprintf(stderr, "=> switching to uid %d / gid %d...", config->uid, config->uid);
|
||||||
if (setgroups(1, & (gid_t) { config->uid }) ||
|
if (setgroups(1, & (gid_t) { config->uid }) ||
|
||||||
setresgid(config->uid, config->uid, config->uid) ||
|
setresgid(config->uid, config->uid, config->uid) ||
|
||||||
|
|||||||
Reference in New Issue
Block a user