Introduction
本文旨在记录作者完成课程任务 HITSZ os-lab2 的过程,可供参考。
本实验需要完成的内容包括
- 在
exit()时打印父子进程信息void exit(int status)
- 给
wait()添加非阻塞选项int wait(int *status, int flags)
- 添加
yield()系统调用,使进程放弃 CPU 资源void yield()
- 附加题,实现系统调用白名单,并且实现系统调用阻拦日志
int seccomp_ctl(int op, uint64 arg);int seccomp_getlog(uint64 *buf, int *len);
实验仓库链接如下
指导书链接如下(需校内网)
仓库内有多个 lab 分支,本文所提的 lab1 对应的分支切换方法如下
git clone git@gitee.com:ftutorials/xv6-oslabs-hitsz.git
git checkout syscall
任务1: exit() 时打印父子进程信息
原本的 exit() 函数的内容大致如下
- 关闭进程打开的所有文件
- 尝试唤醒
init进程 - 获取原本的父进程的锁,并且将所有的子进程重新绑定到
init中(调用reparent()) - 唤醒原本的父进程
- 将当前进程的状态更新为 Zombie,并且设置退出状态
- 跳去
sched()不会再回来
本题要求在 exit 执行流程的合适位置,打印:
当前进程退出时的父进程信息;
当前进程退出时的各个子进程信息。
请注意:
请使用课程提供的 exit_info 输出,不要直接用 printf;
进程状态统一按题目要求输出为小写。
也就是说,exit 要输出父进程信息和各个子进程信息,格式如下:
proc PID exit, parent pid PID, name NAME, state STATE
proc PID exit, child CHILD_NUM, pid PID, name NAME, state STATE
其中各字段含义如下:
PID :进程号;
CHILD_NUM :这是第几个子进程,从 0 开始编号;
NAME :进程名;
STATE :进程状态,例如 sleep、run、runble;其中 runble 是课程输出函数对 RUNNABLE 状态的固定拼写。
考虑父进程信息可以直接在 exit() 内获取,而子进程信息可以在 reparent() 中涉及,所以打印信息位置可以确定下来。
void exit(int status) {
struct proc *p = myproc();
...
acquire(&p->lock);
struct proc *original_parent = p->parent;
int pid = p->pid;
release(&p->lock);
...
acquire(&original_parent->lock);
int ppid = original_parent->pid;
char *name = original_parent->name;
enum procstate state = original_parent->state;
exit_info("proc %d exit, parent pid %d, name %s, state %s\n", pid, ppid, name, procstate_str(state));
...
}
void reparent(struct proc *p) {
struct proc *pp;
int childnum = 0;
int pid = p->pid;
for (pp = proc; pp < &proc[NPROC]; pp++) {
if (pp->parent == p) {
acquire(&pp->lock);
pp->parent = initproc;
int ppid = pp->pid;
char *name = pp->name;
enum procstate state = pp->state;
exit_info("proc %d exit, child %d, pid %d, name %s, state %s\n", pid, childnum, ppid, name, procstate_str(state));
childnum++;
release(&pp->lock);
p->child_count--;
}
}
}
注意使用 exit_info() 而不是 printf()
运行结果如下,执行 make qemu 进入 xv6,然后输入 exittest
init: starting sh
$ exittest
exit test
proc 3 exit, parent pid 2, name sh, state sleep
proc 3 exit, child 0, pid 4, name child0, state runble
proc 3 exit, child 1, pid 5, name child1, state runble
proc 3 exit, child 2, pid 6, name child2, state runble
$ proc 4 exit, parent pid 1, name init, state runble
proc 5 exit, parent pid 1, name init, state runble
proc 6 exit, parent pid 1, name init, state run
如果再按一下回车则是下面的现象
init: starting sh
$ exittest
exit test
proc 3 exit, parent pid 2, name sh, state sleep
proc 3 exit, child 0, pid 4, name child0, state runble
proc 3 exit, child 1, pid 5, name child1, state runble
proc 3 exit, child 2, pid 6, name child2, state runble
$ proc 4 exit, parent pid 1, name init, state runble
proc 5 exit, parent pid 1, name init, state runble
proc 6 exit, parent pid 1, name init, state run
proc 7 exit, parent pid 2, name sh, state sleep
$
如果反复按回车则是下面的现象
$
proc 8 exit, parent pid 2, name sh, state sleep
$
proc 9 exit, parent pid 2, name sh, state sleep
$
proc 10 exit, parent pid 2, name sh, state sleep
$
proc 11 exit, parent pid 2, name sh, state sleep
$
proc 12 exit, parent pid 2, name sh, state sleep
$
这里有一些值得探讨的现象:
- 为什么 proc 4 的前面有个
$符号? - 为什么按一下回车才会再次出现
$符号? - 为什么每按一次回车都会出现一次信息
proc X exit, parent pid 2, name sh, state sleep?
为什么每按一次回车都会出现一次信息?
从最简单的问题开始,为什么每按一次回车都会出现一次信息?这需要看看 shell 的实现,不难发现 sh.c 中,读取命令行的指令通过 getcmd() 函数实现
int getcmd(char *buf, int nbuf) {
fprintf(2, "$ ");
memset(buf, 0, nbuf);
gets(buf, nbuf);
if (buf[0] == 0) // EOF
return -1;
return 0;
}
char *gets(char *buf, int max) {
int i, cc;
char c;
for (i = 0; i + 1 < max;) {
cc = read(0, &c, 1);
if (cc < 1) break;
buf[i++] = c;
if (c == '\n' || c == '\r') break;
}
buf[i] = '\0';
return buf;
}
shell 先打印 $ ,然后清空 buf,读取终端输入。如果直接输入回车,得到的 buf 是什么样的呢?从源代码查看应该是 "\n",用 GDB 检验一下
(gdb) b sh.c:130
❌️ No source file named sh.c.
(gdb) b ../user/sh.c:130
❌️ No source file named ../user/sh.c.
注意到在实验项目中直接 b sh.c:130 是行不通的,xv6 的用户程序 ELF 没有把 sh.c 的调试信息加载进当前 GDB 的符号表。所以可以额外加载用户程序:
(gdb) add-symbol-file user/_sh
add symbol table from file "user/_sh"
Reading symbols from user/_sh...
(gdb) b sh.c:130
Breakpoint 1 at 0x3c: file user/sh.c, line 130.
(gdb) c
Continuing.
Thread 2 hit Breakpoint 1, getcmd (buf=buf@entry=0x1528 <buf> "\n", nbuf=nbuf@entry=100) at user/sh.c:130
130 if (buf[0] == 0) // EOF
(gdb) c
得到输入内容之后,再回到 sh.c 的 main
int main(void) {
...
while (getcmd(buf, sizeof(buf)) >= 0) {
if (buf[0] == 'c' && buf[1] == 'd' && buf[2] == ' ') {
// Chdir must be called by the parent, not the child.
buf[strlen(buf) - 1] = 0; // chop \n
if (chdir(buf + 3) < 0) fprintf(2, "cannot cd %s\n", buf + 3);
continue;
}
if (fork1() == 0) runcmd(parsecmd(buf));
wait(0, 0);
}
exit(0);
}
可以看到,正常情况下就会 fork 成功并且进入 runcmd() 和 parsecmd() 了,其中 parsecmd 会根据 buf 返回 cmd 类型变量
// Parsed command representation
#define EXEC 1
#define REDIR 2
#define PIPE 3
#define LIST 4
#define BACK 5
#define MAXARGS 10
struct cmd {
int type;
};
struct execcmd {
int type;
char *argv[MAXARGS];
char *eargv[MAXARGS];
};
..
gdb 中可以看到回车时返回的是 EXEC 类型,所以 runcmd 会按照这个类型来处理
Thread 1 hit Temporary breakpoint 3, runcmd (cmd=0x13f50) at user/sh.c:65
65 if (cmd == 0) exit(1);
(gdb) p cmd->type
$1 = 1
void runcmd(struct cmd *cmd) {
...
switch (cmd->type) {
default:
panic("runcmd");
case EXEC:
ecmd = (struct execcmd *)cmd;
if (ecmd->argv[0] == 0) exit(1);
exec(ecmd->argv[0], ecmd->argv);
fprintf(2, "exec %s failed\n", ecmd->argv[0]);
break;
...
}
exit(0);
}
通过 gdb 可以看到 ecmd 的情况
(gdb) tb sh.c:72
Temporary breakpoint 1 at 0xf0: file user/sh.c, line 72.
(gdb) c
Continuing.
[Switching to Thread 1.2]
Thread 2 hit Temporary breakpoint 1, runcmd (cmd=0x13f50) at user/sh.c:73
73 if (ecmd->argv[0] == 0) exit(1);
(gdb) p ecmd->type
$1 = 1
(gdb) p ecmd->argv[0]
$4 = 0x0 <getcmd>
所以实际上 fork 出来的子进程直接调用 exit(1) 了,GDB同样证明了这一点
(gdb) tb proc.c:387
Temporary breakpoint 2 at 0x80002598: file kernel/proc.c, line 387.
(gdb) c
Continuing.
Thread 1 hit Temporary breakpoint 2, exit (status=1) at kernel/proc.c:387
387 p->state = ZOMBIE;
所以就有了看到的打印信息
$
proc 3 exit, parent pid 2, name sh, state sleep
为什么 proc 4 的前面有个 $ 符号?
回忆之前的现象
init: starting sh
$ exittest
exit test
proc 3 exit, parent pid 2, name sh, state sleep
proc 3 exit, child 0, pid 4, name child0, state runble
proc 3 exit, child 1, pid 5, name child1, state runble
proc 3 exit, child 2, pid 6, name child2, state runble
$ proc 4 exit, parent pid 1, name init, state runble
proc 5 exit, parent pid 1, name init, state runble
proc 6 exit, parent pid 1, name init, state run
检查一下 existtest.c 可以发现其 fork 了三个子进程并起了不同的名字, sleep 一段时间然后 exit(0)。那么什么时候 sh 会打印美元符号呢?前面提到了是在 getcmd() 的时候,说明 sh 被唤醒了,所以可以推导出 proc 3 在退出时唤醒了 sh ,从 exit() 代码中可以找到对应。
void exit(int status) {
...
// Parent might be sleeping in wait().
wakeup1(original_parent);
...
}
而三个子进程的退出跟 shell 没有关系,但是还是会打印退出信息,于是就会出现 $ 符号后接着 proc 4 exit… 的信息。这个时候再按下回车,相当于在美元符号后面正常输入指令,也就有了新的进程退出的信息。
我们可以尝试在 getcmd 这里添加断点来查看调试信息
int getcmd(char *buf, int nbuf) {
fprintf(2, "$ ");
memset(buf, 0, nbuf);
gets(buf, nbuf);
if (buf[0] == 0) // Breakpoint here
return -1;
return 0;
}
调试结果如下,注意到最后一个 "\n" 是 exittest 运行完成后手动输入的,第一个 "exittest\n" 也是同样也是手动输入的。
(gdb) b sh.c:130
Breakpoint 1 at 0x3c: file user/sh.c, line 130.
(gdb) c
Continuing.
Thread 1 hit Breakpoint 1, getcmd (buf=buf@entry=0x1528 <buf> "exittest\n", nbuf=nbuf@entry=100) at user/sh.c:130
130 if (buf[0] == 0) // EOF
(gdb) c
Continuing.
[Switching to Thread 1.2]
Thread 2 hit Breakpoint 1, getcmd (buf=0x3 <getcmd+3> "\374\"\370&\364", <incomplete sequence \360\200>, nbuf=3) at user/sh.c:130
130 if (buf[0] == 0) // EOF
(gdb) c
Continuing.
Thread 2 hit Breakpoint 1, getcmd (buf=buf@entry=0x1528 <buf> "\n", nbuf=nbuf@entry=100) at user/sh.c:130
130 if (buf[0] == 0) // EOF
(gdb) c
Continuing.
但是中间那个 "\374\"\370&\364" 是什么?GDB查看到的信息如下
(gdb) p/x buf[0]
$7 = 0xfc
(gdb) p/x buf[1]
$8 = 0x22
(gdb) p/x buf[2]
$9 = 0xf8
(gdb) p/x buf[3]
$10 = 0x26
(gdb) p/x buf[4]
$11 = 0xf4
(gdb) p buf[0]
$16 = 252 '\374'
(gdb) p buf[1]
$17 = 34 '"'
(gdb) p buf[2]
$18 = 248 '\370'
(gdb) p buf[3]
$19 = 38 '&'
(gdb) p buf[4]
$20 = 244 '\364'
(gdb) p nbuf
$13 = 3
以 p buf[0] 为例,252 是十进制数,'\374' 本来应该是 ASCII 字符,但是打印不出来,只能显示八进制数。
所以这串神秘数字是什么?由于篇幅比较长,所以把这段探索过程放到附录中了!
任务2: 为 wait() 添加非阻塞选项
这一题围绕 wait 系统调用展开。原来的接口是:
int wait(int *status)
它的作用是:让父进程等待子进程退出,并在成功回收某个子进程时返回该子进程的 pid,同时把退出状态写到 status 指向的位置。
本题要求你把它改成:
int wait(int *status, int flags)
其中:
status :用于接收子进程退出状态;
flags :用于控制这次 wait 是阻塞还是非阻塞。
语义如下:
flags == 0 :保持原来的阻塞等待;
flags == 1 :如果当前没有可回收的僵尸子进程,就不要睡眠,而是直接返回 -1。
系统调用实际上涉及的几个地方如下
user/user.h中提供了用户程序使用的接口声明user/usys.pl生成的usys.S提供了系统调用的汇编代码kernel/syscall.h提供了ecall需要的系统调用号,系统调用号会在usys.S中加载到a7寄存器中kernel/syscall.c根据a7中的系统调用号,调用对应的系统调用函数kernel/sysproc.c提供了系统调用函数的实现(sys_XXX())
在本次任务中,需要修改的包括用户态接口声明(课程组提前改好了)以及内核态系统调用函数的声明和实现。由于 sys_wait() 实际上调用的是 kernel/proc.c 中的 wait() 函数,所以要改的还多了 wait() 的声明(defs.h 中)以及 wait() 函数本身。
// defs.h
int wait(uint64, int);
// user.h
int wait(int*, int);
这里使用了 argint 来传递多出来的 int 类型变量,本质是读取 a1 寄存器
uint64 sys_wait(void) {
uint64 p;
int x;
if (argaddr(0, &p) < 0) return -1;
if (argint(1, &x) < 0) return -1;
return wait(p, x);
}
修改完这部分之后就可以专注于 wait() 的修改了,实际上也没什么好改的,在循环末尾添加一个检测 flag 的标志即可
int wait(uint64 addr, int flags) {
struct proc *np;
int havekids, pid;
struct proc *p = myproc();
acquire(&p->lock);
for (;;) {
havekids = 0;
for (np = proc; np < &proc[NPROC]; np++) {
... // 检测有没有ZOMBIE状态的子进程,有的话把子进程退出状态复制到 addr 中
}
if (!havekids || p->killed) {
release(&p->lock);
return -1;
}
if (flags == 1) {
release(&p->lock);
return -1;
}
sleep(p, &p->lock); // DOC: wait-sleep
}
}
修改完之后运行 waittest 测试,结果类似于下面的样子
$ waittest
wait test
no child exited yet, round 0
no child exited yet, round 1
no child exited yet, round 2
proc 4 exit, parent pid 3, name waittest, state sleep
child exited, pid 4
wait test OK
proc 3 exit, parent pid 2, name sh, state sleep
任务3: 添加 yield() 系统调用并打印调度相关信息
建议先把“系统调用外壳”补齐,再写具体逻辑。通常需要改这些地方:
user/user.h:增加 yield 的用户态声明;
user/usys.pl:增加一个新的 entry;
kernel/syscall.h:增加新的系统调用号;
kernel/syscall.c:补 extern 声明,并把它加入 syscalls[] 分发表;
kernel/sysproc.c:增加 sys_yield();
Makefile:把 _yieldtest 加进 UPROGS。
如果这几步没补齐,即使你写好了 sys_yield(),用户态程序也调不到它。
和任务2中提及的一样,需要先修改上面的文件,新增 sys_yield() 如下,由于已经有现成的 yield() 了,直接调用即可
uint64 sys_yield(void) {
yield();
return 0;
}
至此我们已经添加完成了新的系统调用,但是我们还需要打印调度信息。
Save the context of the process to the memory region from address 0x??? to 0x???
Current running process pid is ??? and user pc is 0x???
Next runnable process pid is ??? and user pc is 0x???
实际上,上下文保存发生在 uservec 中,并且还切换了页表,恢复了内核相关的寄存器,当进入到 sys_yield() 的时候,上下文早已经保存到内存中了,所以我们可以直接在这里进行打印,趁着 CPU 还没被让出去。
关于系统调度的链路,详见另一篇文章 OSTEP Note with xv6: CPU Virtualization
uint64 sys_yield(void) {
struct proc *p = myproc();
print_next_runnable = 1;
printf("Save the context of the process to the memory region from address %p to %p\n", &p->context,
(char *)&p->context + sizeof(p->context));
printf("Current running process pid is %d and user pc is %p\n", p->pid, p->trapframe->epc);
yield();
return 0;
}
接下来调用 yield() 之后,经过一系列调用,最终程序会回到 scheduler() 函数中,选择下一个进程来执行。我们还差下一个进程的信息没有打印出来,这个进程只有进入到 scheduler() 中选择完成下一个进程之后才能知道,所以我们的打印信息只能放在这里。
void scheduler(void) {
struct proc *p;
struct cpu *c = mycpu();
c->proc = 0;
for (;;) {
intr_on();
int found = 0;
for (p = proc; p < &proc[NPROC]; p++) {
acquire(&p->lock);
if (p->state == RUNNABLE) {
p->state = RUNNING;
c->proc = p;
printf("Next runnable process pid is %d and user pc is %p\n", p->pid, p->trapframe->epc);
swtch(&c->context, &p->context);
c->proc = 0;
found = 1;
}
release(&p->lock);
}
if (found == 0) {
intr_on();
asm volatile("wfi");
}
}
}
这样一来,就可以打印全部信息了,运行 yieldtest 结果如下
$ yieldtest
Next runnable process pid is 2 and user pc is 0x0000000000000e36
Next runnable process pid is 3 and user pc is 0x0000000000000e16
Next runnable process pid is 3 and user pc is 0x0000000000000e56
Next runnable process pid is 3 and user pc is 0x0000000000000e56
Next runnable process pid is 3 and user pc is 0x0000000000000e56
yield test
parent yield
Save the context of the process to the memory region from address 0x0000000080012098 to 0x0000000080012108
Current running process pid is 3 and user pc is 0x00000000000003e8
Next runnable process pid is 4 and user pc is 0x0000000000000338
Child with PID 4 begins to run
[INFO] proc 4 exit, parent pid 3, name yieldtest, state runble
Next runnable process pid is 5 and user pc is 0x0000000000000338
Child with PID 5 begins to run
[INFO] proc 5 exit, parent pid 3, name yieldtest, state runble
Next runnable process pid is 6 and user pc is 0x0000000000000338
Child with PID 6 begins to run
[INFO] proc 6 exit, parent pid 3, name yieldtest, state runble
Next runnable process pid is 1 and user pc is 0x0000000000000396
Next runnable process pid is 3 and user pc is 0x00000000000003e8
parent yield finished
[INFO] proc 3 exit, parent pid 2, name sh, state sleep
Next runnable process pid is 1 and user pc is 0x0000000000000396
Next runnable process pid is 2 and user pc is 0x0000000000000e26
虽然我们确实把信息都打印了出来,但是普通的 RR 调度导致的进程切换也给打印出来了,这并不符合实验要求,应该只打印一次才对。
所以我添加了一个全局变量来让 scheduler 只打印一次
if (print_next_runnable == 1) {
printf("Next runnable process pid is %d and user pc is %p\n", p->pid, p->trapframe->epc);
print_next_runnable = 0;
}
int print_next_runnable = 0;
uint64 sys_yield(void) {
struct proc *p = myproc();
print_next_runnable = 1;
printf("Save the context of the process to the memory region from address %p to %p\n", &p->context,
(char *)&p->context + sizeof(p->context));
printf("Current running process pid is %d and user pc is %p\n", p->pid, p->trapframe->epc);
yield();
return 0;
}
这里没有考虑并发的情况,因为测试只使用一个CPU,这样一来打印的结果就正常了。
sodium@nas-MacBook-Air-13 xv6-oslabs-hitsz % make qemu CPUS=1
...
init: starting sh
$ yieldtest
yield test
parent yield
Save the context of the process to the memory region from address 0x00000000800122c0 to 0x0000000080012330
Current running process pid is 3 and user pc is 0x00000000000003e8
Next runnable process pid is 4 and user pc is 0x0000000000000338
Child with PID 4 begins to run
proc 4 exit, parent pid 3, name yieldtest, state runble
Child with PID 5 begins to run
proc 5 exit, parent pid 3, name yieldtest, state runble
Child with PID 6 begins to run
proc 6 exit, parent pid 3, name yieldtest, state runble
parent yield finished
proc 3 exit, parent pid 2, name sh, state sleep
20260923更新:指导书修改增加了对这部分的说明
sys_yield() 中建议依次完成下面几件事:
1. 通过 myproc() 拿到当前进程;
2. 打印当前进程 context 的地址范围;
3. 通过 trapframe 打印当前进程的用户态 pc;
4. 参考 scheduler() 的工作方式 ,从当前进程开始环形遍历全局进程表,寻找下一个 RUNNABLE 进程;
5. 打印这个“下一个可运行进程”的 pid 和用户态 pc;
6. 最后调用内核已经提供的 yield() ,让当前进程真正让出 CPU。
这里有两个细节尤其值得注意:
如果要在 sysproc.c 中遍历全局进程表 proc[],你需要能访问它的声明;
遍历进程表时要注意锁的获取和释放,避免把锁带出循环。
所以原本的打印方法需要按照这里的说明可以修改
extern struct proc proc[NPROC];
void predict_next_process(){
int found = 0;
struct proc *p;
for (p = proc; p < &proc[NPROC]; p++) {
acquire(&p->lock);
if (p->state == RUNNABLE) {
printf("Next runnable process pid is %d and user pc is %p\n", p->pid, p->trapframe->epc);
found = 1;
release(&p->lock);
break;
}
release(&p->lock);
}
if (found == 0) {
printf("Failed to find next runnable process\n");
}
}
但是询问实验老师得知具体打印方法没有强制要求,所以也可以不使用上面的方法。
执行 make grade 总测试
$ make qemu-gdb LAB_SYSCALL_TEST=on
exit test: OK (4.6s)
== Test wait test ==
$ make qemu-gdb LAB_SYSCALL_TEST=on
wait test: OK (1.3s)
== Test yield test ==
$ make qemu-gdb CPUS=1 LAB_SYSCALL_TEST=on
yield test: OK (0.3s)
(Old xv6.out.yield_test failure log removed)
== Test seccomp test ==
$ make qemu-gdb CPUS=1 LAB_SYSCALL_TEST=on
seccomp test: FAIL (1.3s)
...
--- seccomp test FAIL ---
seccomp test FAIL
proc 4 exit, parent pid 1, name init, state runble
proc 4 exit, child 0, pid 11, name seccomptest, state zombie
$ qemu-system-riscv64: terminating on signal 15 from pid 56210 (<unknown process>)
MISSING 'seccomp test OK'
QEMU output saved to xv6.out.seccomp_test
Score: 100/120
make: *** [grade] Error 1
实际上关于打印调度信息,有一个不可避免的问题就是,打印有一定概率会被打断。如果反复测试,有概率会遇到类似于下面的情况
$ yieldtest
yield test
pareChild with PID 45 begins to run
proc 45 exit, parent pid 44, name yieldtest, state runble
Child with PID 46 begins to run
proc 46 exit, parent pid 44, name yieldtest, state runble
Child with PID 47 begins to run
proc 47 exit, parent pid 44, name yieldtest, state runble
nt yield
Save the context of the process to the memory region from address 0x00000000800122c0 to 0x0000000080012330
Current running process pid is 44 and user pc is 0x00000000000003e8
Next runnable process pid is 44 and user pc is 0x00000000000003e8
parent yield finished
proc 44 exit, parent pid 2, name sh, state sleep
"parent yield" 打印到一半,子进程全部都执行完了,同样的,还有可能出现子进程还没运行完,父进程就被唤醒并且打断的情况
$ yieldtest
yield test
parent yield
Save the context of the process to the memory region from address 0x00000000800122c0 to 0x0000000080012330
Current running process pid is 3 and user pc is 0x00000000000003e8
Next runnable process pid is 4 and user pc is 0x0000000000000338
Child with PID 4 begins to run
proc 4 exit, parent pid 3, name yieldtest, state runble
Child with PID 5 begins to run
proc 5 exit, parent pid 3, name yieldtest, state runble
Child wparent yield finished
ith PID 6 begins to run
proc 6 exit, parent pid 3, name yieldtest, state sleep
proc 3 exit, parent pid 2, name sh, state sleep
$ qemu-system-riscv64: terminating on signal 15 from pid 2484 (<unknown process>)
这个问题与并发无关,单纯是调度器导致的。
附加题:系统调用白名单和审计日志
附加题需要实现的接口如下
// op: 操作码
// 0 — 设置系统调用白名单位掩码(arg 为位掩码)
// 1 — 设置最大子进程数(arg 为最大数量)
int seccomp_ctl(int op, uint64 arg);
// buf: 存放审计日志条目的缓冲区(至少 32 * sizeof(uint64) 字节)
// len: 传入缓冲区容量,返回实际写入的条目数
int seccomp_getlog(uint64 *buf, int *len);
虽然实验指导书要求的三个任务顺序是
- 任务一:系统调用白名单(Syscall Allowlist)
- 任务二:审计日志(Audit Log)
- 任务三:进程资源限制(Process Resource Limits)
但是考虑到 seccomp_ctl 同时牵扯到任务一和三,所以这里不区分顺序了。
直接添加系统调用白名单位图、日志数组、日志数量、子进程上限、子进程计数到 proc.h 中
// Per-process state
struct proc {
struct spinlock lock;
// p->lock must be held when using these:
enum procstate state; // Process state
struct proc *parent; // Parent process
void *chan; // If non-zero, sleeping on chan
int killed; // If non-zero, have been killed
int xstate; // Exit status to be returned to parent's wait
int pid; // Process ID
// these are private to the process, so p->lock need not be held.
uint64 kstack; // Virtual address of kernel stack
uint64 sz; // Size of process memory (bytes)
pagetable_t pagetable; // User page table
struct trapframe *trapframe; // data page for trampoline.S
struct context context; // swtch() here to run process
struct file *ofile[NOFILE]; // Open files
struct inode *cwd; // Current directory
char name[16]; // Process name (debugging)
// new
uint64 seccomp_mask; // Syscall Allowlist
uint64 seccomp_log[SECCOMP_LOG_SIZE]; // Recording Intercepted Sys Call
int seccomp_log_count;
int child_count; // Number of Children
int child_limit;
};
比较需要注意的是并发问题,我们需要考虑这些新增的变量是否需要用锁机制
seccomp_mask- 由于这个变量不会被其他进程使用,所以不用考虑并发
child_count- 这个变量同样也不会被其他进程使用
seccomp_log[]- 这个变量也不会被其他进程使用
seccomp_log_count- 这个变量不会被其他进程使用
child_count- 这个变量在按要求进程
fork时有检查的操作和添加的操作,在wait时有减少的操作,不会被其他进程使用
- 这个变量在按要求进程
child_limit- 这个变量在
fork时应该继承,但是也不会有别的进程会访问这个
- 这个变量在
白名单机制添加和子进程限制
和指导书一样,我采用了位掩码的方式来记录白名单,系统调用号为 x ,那么对应的掩码的第 x 位就是系统调用的允许情况
需要注意的一点是指导书要求 seccomp_ctl 和 seccomp_getlog 应该始终可用,即使设置了白名单也能被调用,否则进程将无法恢复或查询日志。(虽然没有这个机制也能通过测试)
void syscall(void) {
int num;
struct proc *p = myproc();
num = p->trapframe->a7;
if (num > 0 && num < NELEM(syscalls) && syscalls[num]) {
// 检查对应位是否为 1,是则允许系统调用
if (p->seccomp_mask && !(p->seccomp_mask & (1ULL << num))
&& num != SYS_seccomp_ctl && num != SYS_seccomp_getlog) { // 定义 mask=0 时不启用白名单,初始化时就是 0
printf("sys call %d is NOT allowed\n", num);
if (p->seccomp_log_count < SECCOMP_LOG_SIZE) { // 记录拦截的系统调用号
p->seccomp_log[p->seccomp_log_count] = num;
p->seccomp_log_count++;
}
p->trapframe->a0 = -1;
return;
}
p->trapframe->a0 = syscalls[num]();
} else {
printf("%d %s: unknown sys call %d\n", p->pid, p->name, num);
p->trapframe->a0 = -1;
}
}
int seccomp_ctl(int op, uint64 arg) {
struct proc *p = myproc();
if (op == 0) {
p->seccomp_mask = arg;
} else {
p->child_limit = arg;
}
return 0;
}
关于子进程限制,没有什么特别值得说的,按照指导书说的修改即可
你需要:
1. 在 struct proc 中增加子进程计数和最大子进程数限制字段
2. 扩展 seccomp_ctl 系统调用的功能,使其支持设置子进程数上限
3. 修改 fork() 函数,在创建子进程前检查上限
4. 当一个子进程被 wait() 回收时,相应减少计数
需要考虑白名单以及资源限制在 fork 时的继承关系——子进程应当继承父进程的沙箱配置。
// check child limit
if (p->child_limit != -1 && p->child_count >= p->child_limit) {
printf("Error: Number of child processes exceeds the limit current/limit: %d/%d.", p->child_count, p->child_limit);
return -1;
}
审计日志实现
前面使用了一个数组来存储被拦截的系统调用号,并且用一个 int 变量来记录拦截的数量,超过最大限制之后不再改变数组,那么拦截记录完毕之后还差一个读取的系统调用。
具体代码如下,没有什么特别值得说明的
uint64 sys_seccomp_getlog(void) {
uint64 buf;
uint64 len_addr;
int len;
struct proc *p = myproc();
// 获取用户态 buf 和 len 的地址
if (argaddr(0, &buf) < 0 || argaddr(1, &len_addr) < 0) return -1;
// 读取用户传进来的 *len
if (copyin(p->pagetable, (char *)&len, len_addr, sizeof(int)) < 0) return -1;
if (len < 0) return -1;
// len限制
if (len > p->seccomp_log_count) len = p->seccomp_log_count;
if (len >= SECCOMP_LOG_SIZE) len = SECCOMP_LOG_SIZE;
// 把日志复制到用户空间
if (copyout(p->pagetable, buf, (char *)p->seccomp_log, len * sizeof(uint64)) < 0) return -1;
// 把实际写入的条目数写回用户空间
if (copyout(p->pagetable, len_addr, (char *)&len, sizeof(int)) < 0) return -1;
return 0;
}
全部任务完成后,进行测试
$ seccomptest
seccomp test
-- Task 1: Syscall Allowlist --
sys call 1 is NOT allowed
PASS: fork() blocked by seccomp
sys call 15 is NOT allowed
PASS: open() blocked by seccomp
write() still works - good
-- Task 2: Audit Log --
Audit log has 2 blocked syscall(s):
[0] syscall 1 blocked
[1] syscall 15 blocked
PASS: audit log contains blocked syscalls
-- Task 3: Process Resource Limits --
PASS: first fork() succeeds (within limit)
proc 4 exit, parent pid 3, name seccomptest, state sleep
PASS: second fork() succeeds
PASS: third fork() succeeds (child_count = 2, limit = 2)
Error: Number of child processes exceeds the limit current/limit: 2/2.PASS: fourth fork() blocked by child limit
proc 5 exit, parent pid 3, name seccomptest, state sleep
proc 6 exit, parent pid 3, name seccomptest, state runble
--- seccomp test OK ---
seccomp test OK
proc 3 exit, parent pid 2, name sh, state sleep
附录: 任务1 GDB 中出现的神秘数字 FC 22 F8 26... 是什么?
任务一种观察到了一个有趣的现象是观察到 GDB 中出现 buf={FC,22,F8,26...} ,但是 xv6 终端并没有打印对应的美元符号,所以我打算尝试追踪这一点。
(gdb) add-symbol-file user/_sh
add symbol table from file "user/_sh"
Reading symbols from user/_sh...
(gdb) b sh.c:126
Breakpoint 1 at 0x0: file user/sh.c, line 126.
(gdb) b sh.c:128
Breakpoint 2 at 0x22: file user/sh.c, line 128.
(gdb) b sh.c:130
Breakpoint 3 at 0x3c: file user/sh.c, line 130.
(gdb) c
Continuing.
Breakpoint 1, getcmd (buf=0x505050505050505 <error: Cannot access memory at address 0x505050505050505>, nbuf=84215045) at user/sh.c:126
126 int getcmd(char *buf, int nbuf) {
(gdb) c
Continuing.
Breakpoint 1, getcmd (buf=0x1 <getcmd+1> "\021\006\354\"\350&\344", <incomplete sequence \340>, nbuf=12256) at user/sh.c:126
126 int getcmd(char *buf, int nbuf) {
(gdb) c
Continuing.
Breakpoint 2, getcmd (buf=0x505050505050505 <error: Cannot access memory at address 0x505050505050505>, nbuf=84215045) at user/sh.c:128
128 memset(buf, 0, nbuf);
(gdb) c
Continuing.
Breakpoint 1, getcmd (buf=buf@entry=0x1528 <buf> "", nbuf=nbuf@entry=100) at user/sh.c:126
126 int getcmd(char *buf, int nbuf) {
(gdb) c
Continuing.
Breakpoint 2, getcmd (buf=buf@entry=0x1528 <buf> "", nbuf=nbuf@entry=100) at user/sh.c:128
128 memset(buf, 0, nbuf);
(gdb) c
Continuing.
Breakpoint 3, getcmd (buf=buf@entry=0x1528 <buf> "exittest\n", nbuf=nbuf@entry=100) at user/sh.c:130
130 if (buf[0] == 0) // EOF
(gdb) c
Continuing.
Breakpoint 1, getcmd (buf=0x1 <getcmd+1> "q\006\374\"\370&\364", <incomplete sequence \360\200>, nbuf=12256) at user/sh.c:126
126 int getcmd(char *buf, int nbuf) {
(gdb) c
Continuing.
Breakpoint 3, getcmd (buf=0x3 <getcmd+3> "\374\"\370&\364", <incomplete sequence \360\200>, nbuf=3) at user/sh.c:130
130 if (buf[0] == 0) // EOF
(gdb) c
Continuing.
Breakpoint 1, getcmd (buf=buf@entry=0x1528 <buf> "exittest\n", nbuf=nbuf@entry=100) at user/sh.c:126
126 int getcmd(char *buf, int nbuf) {
(gdb) c
Continuing.
Breakpoint 2, getcmd (buf=buf@entry=0x1528 <buf> "exittest\n", nbuf=nbuf@entry=100) at user/sh.c:128
128 memset(buf, 0, nbuf);
(gdb) c
Continuing.
其实到这里已经可以发现了,只有完整出现 sh.c:126,128,130 以及 nbuf=100 的才是真实的 getcmd() 调用,GDB 出现的断点未必是真实的。下面的结果同样可以佐证这一点:
(gdb) add-symbol-file user/_exittest
add symbol table from file "user/_exittest"
Reading symbols from user/_exittest...
(gdb) b exittest.c:8
Breakpoint 1 at 0xc: file user/exittest.c, line 8.
(gdb) c
Continuing.
Breakpoint 1, exittest () at user/exittest.c:8
8 printf("exit test\n");
(gdb)
虽然出现了 breakpoint 1,但是 xv6 连shell 都没启动
xv6 kernel is booting
实际上 GDB 的断点是根据 VA 来确定的,如果切换了页表,这个断点很可能就不准确了。因此,我们可以接着分析下去,这个奇怪的数字应该不是来自于 sh 进程,断点实际上停在了别的地方!所以下一步我决定获取不同进程的 satp 。
先确定这个神秘数字的 satp
Breakpoint 3, getcmd (buf=0x3 <getcmd+3> "\374\"\370&\364", <incomplete sequence \360\200>, nbuf=3) at user/sh.c:130
130 if (buf[0] == 0) // EOF
(gdb) p/x $satp
$9 = 0x8000000000087f48
对比 shell 本身的 satp
Breakpoint 2, getcmd (buf=buf@entry=0x1528 <buf> "exittest\n", nbuf=nbuf@entry=100) at user/sh.c:128
128 memset(buf, 0, nbuf);
(gdb) p/x $satp
$12 = 0x8000000000087f63
果然不一样!这证明了这个断点在不是 sh.c 内!
然后是内核的 satp
Breakpoint 4, exit (status=1) at kernel/proc.c:331
331 struct proc *p = myproc();
(gdb) p/x $satp
$13 = 0x8000000000087fff
可以看得出来,神秘数字应该不是内核导致的。
接下来是 exittest 的 satp
Breakpoint 1, exittest () at user/exittest.c:8
8 printf("exit test\n");
(gdb) p/x $satp
$1 = 0x8000000000087f48
证实了神秘数字就是进程 exittest 导致的。那么我们接下来就可以进一步确定这个神秘数字到底来自哪里。
Breakpoint 1, getcmd (buf=0x3 <getcmd+3> "\374\"\370&\364", <incomplete sequence \360\200>, nbuf=3) at user/sh.c:130
130 if (buf[0] == 0) // EOF
(gdb) p/x $pc
$2 = 0x3c
(gdb) p/x $ra
$3 = 0xd4
(gdb) p/x $sp
$4 = 0x2f90
(gdb) p/x $a0
$5 = 0x0
(gdb) p/x $a1
$6 = 0x2ecf
(gdb) p &buf[0]
$7 = 0x3 <getcmd+3> "\374\"\370&\364", <incomplete sequence \360\200>
(gdb) p &nbuf
❌️ Address requested for identifier "nbuf" which is in register $s2
(gdb) p $s2
$8 = 3
发现这里 PC=0x3c,检查一下 GDB 的指令:
(gdb) x/16i $pc
=> 0x3c <getcmd+60>: ld s0,48(sp)
0x3e <getcmd+62>: ld s1,40(sp)
0x40 <getcmd+64>: ld s2,32(sp)
0x42 <getcmd+66>: addi sp,sp,64
0x44 <getcmd+68>: ret
0x46 <getcmd+70>: auipc a5,0x1
0x4a <getcmd+74>: addi a5,a5,-1918
0x4e <getcmd+78>: lw a4,0(a5)
0x50 <getcmd+80>: sw a4,-48(s0)
0x54 <panic>: lhu a5,4(a5)
0x58 <panic+4>: sh a5,-44(s0)
0x5c <panic+8>: sh zero,-42(s0)
0x60 <panic+12>: sh zero,-40(s0)
0x64 <panic+16>: sh zero,-38(s0)
0x68 <panic+20>: sh zero,-36(s0)
0x6c <panic+24>: sh zero,-34(s0)
对比 exittest 的汇编代码
3a: 70e2 ld ra,56(sp)
3c: 7442 ld s0,48(sp)
3e: 74a2 ld s1,40(sp)
40: 7902 ld s2,32(sp)
42: 6121 addi sp,sp,64
44: 8082 ret
char name[16] = "child";
46: 00001797 auipc a5,0x1
4a: 88278793 addi a5,a5,-1918 # 8c8 <malloc+0x118>
4e: 4398 lw a4,0(a5)
50: fce42823 sw a4,-48(s0)
54: 0047d783 lhu a5,4(a5)
58: fcf41a23 sh a5,-44(s0)
5c: fc041b23 sh zero,-42(s0)
60: fc041c23 sh zero,-40(s0)
64: fc041d23 sh zero,-38(s0)
68: fc041e23 sh zero,-36(s0)
6c: fc041f23 sh zero,-34(s0)
char idx = '0' + i;
完全一致!CPU 这个时候就是在执行 exittest 的内容!所以目前为止我们可以猜测神秘数字应该就来自于 exittest 进程中 VA=0x3,0x4,0x5,0x6 的数据。
那么这些数据是怎么来的呢?我们查看 ELF 文件的内容就知道了。
sodium@nas-MacBook-Air-13 xv6-oslabs-hitsz % riscv64-unknown-elf-objdump -s --section=.text user/_exittest
user/_exittest: file format elf64-littleriscv
Contents of section .text:
0000 397106fc 22f826f4 4af08000 17150000 9q..".&.J.......
0010 1305458a 97000000 e780c06d 81440d49 ..E........m.D.I
0020 97000000 e7804034 19cd8524 e39a24ff ......@4...$..$.
0030 05459700 0000e780 a03ce270 4274a274 .E.......<.pBt.t
0040 02792161 82809717 00009387 27889843 .y!a........'..C
转换成 VA 和数据如下
VA bytes
0x00 39 71 06 fc
0x04 22 f8 26 f4
0x08 4a f0 80 00
0x0c 17 15 00 00
0x10 13 05 45 8a
...
0x3 是 FC,0x4 是 22… 这实际上是指令数据
void exittest(void) {
0: 7139 addi sp,sp,-64
2: fc06 sd ra,56(sp)
4: f822 sd s0,48(sp)
6: f426 sd s1,40(sp)
8: f04a sd s2,32(sp)
a: 0080 addi s0,sp,64
....
}
还有一个问题,为什么 buf=0x3, nbuf=0x3?
这实际上应该是由调用决定的,GDB 展示了正常情况下 getcmd() 怎么传递参数
Breakpoint 1, getcmd (buf=buf@entry=0x1528 <buf> "exittest\n", nbuf=nbuf@entry=100) at user/sh.c:130
130 if (buf[0] == 0) // EOF
(gdb) p &buf
❌️ Address requested for identifier "buf" which is in register $s1
(gdb) p &nbuf
❌️ Address requested for identifier "nbuf" which is in register $s2
(gdb)
看到函数内访问 buf,nbuf 时 GDB 返回的是 s1,s2 的值
int getcmd(char *buf, int nbuf) {
0:1101 addi sp,sp,-32
2:ec06 sd ra,24(sp)
4:e822 sd s0,16(sp)
6:e426 sd s1,8(sp)
8:e04a sd s2,0(sp)
a:1000 addi s0,sp,32
c:84aa mv s1,a0
e:892e mv s2,a1
...
}
汇编代码这里同样表明了 a0,a1 传递的参数被保存到了 s1, s2 中,GDB 是通过读取 s1,s2 寄存器的值来告诉用户 buf=0x3, nbuf=0x3 的。检查一下神秘数字出现时的寄存器信息:
Breakpoint 1, getcmd (buf=0x3 <getcmd+3> "\374\"\370&\364", <incomplete sequence \360\200>, nbuf=3) at user/sh.c:130
130 if (buf[0] == 0) // EOF
(gdb) p $s1
$2 = 3
(gdb) p $s2
$3 = 3
所以这就得到了 buf=0x3, nbuf=0x3 出现的原因:因为 s1,s2 就是这个值。
那么为什么这两个寄存器的数值都是 3 呢?这就得看 exittest 的代码了
void exittest(void) {
...
for (int i = 0; i < 3; i++) {
1c:4481 li s1,0
1e:490d li s2,3
int pid = fork();
20:00000097 auipc ra,0x0
24:344080e7 jalr 836(ra) # 364 <fork>
if (pid == 0) {
28:cd19 beqz a0,46 <exittest+0x46>
for (int i = 0; i < 3; i++) {
2a:2485 addiw s1,s1,1
2c:ff249ae3 bne s1,s2,20 <exittest+0x20>
sleep(10);
exit(0);
}
}
sleep(1);
30:4505 li a0,1
32:00000097 auipc ra,0x0
36:3ca080e7 jalr 970(ra) # 3fc <sleep>
}
重点关注在断点 $pc=0x3c 之前, s1,s2 的变化。不难发现,这实际上 s1=s2=3 因为要 fork 三个子进程导致的, s1 相当于 for() 中的 i,s2 则是循环终止条件,执行完成之后两个寄存器的数值都是3。如果修改原本 exittest 的代码,让它循环次数变多,也许可以得到不一样的数字。
至此,一切神秘数据都有了清晰的解释。