结合中断上下文切换和进程上下文切换分析Linux内核的一般执行过程
一、实验目标
-
以fork和execve系统调用为例分析中断上下文的切换
-
分析execve系统调用中断上下文的特殊之处
-
分析fork子进程启动执行时进程上下文的特殊之处
-
以系统调用作为特殊的中断,结合中断上下文切换和进程上下文切换分析Linux系统的一般执行过程
二、fork系统调用
在Linux内核中与C标准库中的差别:
Since version 2.3.3, rather than invoking the kernel's fork() system call, the glibc fork() wrapper that is provided as part of the NPTL threading implementation invokes clone(2) with flags that provide the same effect as the traditional system call. (A call to fork() is equivalent to a call to clone(2) specifying flags as just SIGCHLD.) The glibc wrapper invokes any fork handlers that have been established using pthread_atfork(3).
通过调用C标准库的fork函数并不会调用fork系统调用,通过实验可以验证如上结果:
编写如下代码调用C标准库:
1 /* 2 * fork_test 3 * Created on: 2020-6-10 4 */ 5 #include <unistd.h> 6 #include <stdio.h> 7 int main () 8 { r 9 pid_t fpid; //fpid表示fork函数返回的值 10 int count=0; 11 fpid=fork(); 12 if (fpid < 0) 13 printf("error"); 14 else if (fpid == 0) { 15 printf("i am the child process, my process id is %d\n",getpid()); 16 } 17 else { 18 printf("i am the parent process, my process id is %d\n",getpid()); 19 } 20 return 0; 21 }
在syscall_64.tbl中查找fork系统调用符号:

DEBUG:

如下结果可以看出未触发断点:

汇编调用:
1 /* 2 * fork_test 3 * Created on: 2020-6-10 4 */ 5 6 int main () 7 { 8 asm volatile( 9 "movl $0x39,%eax\n\t" 10 "syscall\n\t" 11 12 ); 13 return 0; 14 15 }
DEBUG:

fork调用过程:

_do_fork源码:
1 /* 2 * Ok, this is the main fork-routine. 3 * 4 * It copies the process, and if successful kick-starts 5 * it and waits for it to finish using the VM if required. 6 * 7 * args->exit_signal is expected to be checked for sanity by the caller. 8 */ 9 long _do_fork(struct kernel_clone_args *args) 10 { 11 u64 clone_flags = args->flags; 12 struct completion vfork; 13 struct pid *pid; 14 struct task_struct *p; 15 int trace = 0; 16 long nr; 17 18 /* 19 * Determine whether and which event to report to ptracer. When 20 * called from kernel_thread or CLONE_UNTRACED is explicitly 21 * requested, no event is reported; otherwise, report if the event 22 * for the type of forking is enabled. 23 */ 24 if (!(clone_flags & CLONE_UNTRACED)) { 25 if (clone_flags & CLONE_VFORK) 26 trace = PTRACE_EVENT_VFORK; 27 else if (args->exit_signal != SIGCHLD) 28 trace = PTRACE_EVENT_CLONE; 29 else 30 trace = PTRACE_EVENT_FORK; 31 32 if (likely(!ptrace_event_enabled(current, trace))) 33 trace = 0; 34 } 35 36 p = copy_process(NULL, trace, NUMA_NO_NODE, args); 37 add_latent_entropy(); 38 39 if (IS_ERR(p)) 40 return PTR_ERR(p); 41 42 /* 43 * Do this prior waking up the new thread - the thread pointer 44 * might get invalid after that point, if the thread exits quickly. 45 */ 46 trace_sched_process_fork(current, p); 47 48 pid = get_task_pid(p, PIDTYPE_PID); 49 nr = pid_vnr(pid); 50 51 if (clone_flags & CLONE_PARENT_SETTID) 52 put_user(nr, args->parent_tid); 53 54 if (clone_flags & CLONE_VFORK) { 55 p->vfork_done = &vfork; 56 init_completion(&vfork); 57 get_task_struct(p); 58 } 59 60 wake_up_new_task(p); 61 62 /* forking complete and child started to run, tell ptracer */ 63 if (unlikely(trace)) 64 ptrace_event_pid(trace, pid); 65 66 if (clone_flags & CLONE_VFORK) { 67 if (!wait_for_vfork_done(p, &vfork)) 68 ptrace_event_pid(PTRACE_EVENT_VFORK_DONE, pid); 69 } 70 71 put_pid(pid); 72 return nr; 73 }
copy_process 复制进程描述符和执⾏时所需的其他数据结构
dup_task_struct 复制进程描述符task_struct、创建内核堆栈等
copy_thread_tls 初始化⼦进程内核栈和thread
wake_up_new_task 将⼦进程添加到就绪队列
二、execve系统调用
在syscall_64.tbl中查找execve系统调用符号:

汇编调用:
1 /*
2 * fork_test
3 * Created on: 2020-6-10
4 */
5
6 int main ()
7 {
8 asm volatile(
9 "movl $0x40,%eax\n\t"
10 "syscall\n\t"
11
12 );
13 return 0;
14
15 }
DEBUG:

execve调用过程:



__do_execve_file源码:
1 /* 2 * sys_execve() executes a new program. 3 */ 4 static int __do_execve_file(int fd, struct filename *filename, 5 struct user_arg_ptr argv, 6 struct user_arg_ptr envp, 7 int flags, struct file *file) 8 { 9 char *pathbuf = NULL; 10 struct linux_binprm *bprm; 11 struct files_struct *displaced; 12 int retval; 13 14 if (IS_ERR(filename)) 15 return PTR_ERR(filename); 16 17 /* 18 * We move the actual failure in case of RLIMIT_NPROC excess from 19 * set*uid() to execve() because too many poorly written programs 20 * don't check setuid() return code. Here we additionally recheck 21 * whether NPROC limit is still exceeded. 22 */ 23 if ((current->flags & PF_NPROC_EXCEEDED) && 24 atomic_read(¤t_user()->processes) > rlimit(RLIMIT_NPROC)) { 25 retval = -EAGAIN; 26 goto out_ret; 27 } 28 29 /* We're below the limit (still or again), so we don't want to make 30 * further execve() calls fail. */ 31 current->flags &= ~PF_NPROC_EXCEEDED; 32 33 retval = unshare_files(&displaced); 34 if (retval) 35 goto out_ret; 36 37 retval = -ENOMEM; 38 bprm = kzalloc(sizeof(*bprm), GFP_KERNEL); 39 if (!bprm) 40 goto out_files; 41 42 retval = prepare_bprm_creds(bprm); 43 if (retval) 44 goto out_free; 45 46 check_unsafe_exec(bprm); 47 current->in_execve = 1; 48 49 if (!file) 50 file = do_open_execat(fd, filename, flags); 51 retval = PTR_ERR(file); 52 if (IS_ERR(file)) 53 goto out_unmark; 54 55 sched_exec(); 56 57 bprm->file = file; 58 if (!filename) { 59 bprm->filename = "none"; 60 } else if (fd == AT_FDCWD || filename->name[0] == '/') { 61 bprm->filename = filename->name; 62 } else { 63 if (filename->name[0] == '\0') 64 pathbuf = kasprintf(GFP_KERNEL, "/dev/fd/%d", fd); 65 else 66 pathbuf = kasprintf(GFP_KERNEL, "/dev/fd/%d/%s", 67 fd, filename->name); 68 if (!pathbuf) { 69 retval = -ENOMEM; 70 goto out_unmark; 71 } 72 /* 73 * Record that a name derived from an O_CLOEXEC fd will be 74 * inaccessible after exec. Relies on having exclusive access to 75 * current->files (due to unshare_files above). 76 */ 77 if (close_on_exec(fd, rcu_dereference_raw(current->files->fdt))) 78 bprm->interp_flags |= BINPRM_FLAGS_PATH_INACCESSIBLE; 79 bprm->filename = pathbuf; 80 } 81 bprm->interp = bprm->filename; 82 83 retval = bprm_mm_init(bprm); 84 if (retval) 85 goto out_unmark; 86 87 retval = prepare_arg_pages(bprm, argv, envp); 88 if (retval < 0) 89 goto out; 90 91 retval = prepare_binprm(bprm); 92 if (retval < 0) 93 goto out; 94 95 retval = copy_strings_kernel(1, &bprm->filename, bprm); 96 if (retval < 0) 97 goto out; 98 99 bprm->exec = bprm->p; 100 retval = copy_strings(bprm->envc, envp, bprm); 101 if (retval < 0) 102 goto out; 103 104 retval = copy_strings(bprm->argc, argv, bprm); 105 if (retval < 0) 106 goto out; 107 108 would_dump(bprm, bprm->file); 109 110 retval = exec_binprm(bprm); 111 if (retval < 0) 112 goto out; 113 114 /* execve succeeded */ 115 current->fs->in_exec = 0; 116 current->in_execve = 0; 117 rseq_execve(current); 118 acct_update_integrals(current); 119 task_numa_free(current, false); 120 free_bprm(bprm); 121 kfree(pathbuf); 122 if (filename) 123 putname(filename); 124 if (displaced) 125 put_files_struct(displaced); 126 return retval; 127 128 out: 129 if (bprm->mm) { 130 acct_arg_size(bprm, 0); 131 mmput(bprm->mm); 132 } 133 134 out_unmark: 135 current->fs->in_exec = 0; 136 current->in_execve = 0; 137 138 out_free: 139 free_bprm(bprm); 140 kfree(pathbuf); 141 142 out_files: 143 if (displaced) 144 reset_files_struct(displaced); 145 out_ret: 146 if (filename) 147 putname(filename); 148 return retval; 149 }
Linux中断上下文:

Linux中断分为两个半部:上半部(tophalf)和下半部(bottom half)。上半部的功能是"登记中断",当一个中断发生时,它进行相应地硬件读写后就把中断例程的下半部挂到该设备的下半部执行队列中去。因此,上半部 执行的速度就会很快,可以服务更多的中断请求。但是,仅有"登记中断"是远远不够的,因为中断的事件可能很复杂。因此,Linux引入了一个下半部,来完 成中断事件的绝大多数使命。下半部和上半部最大的不同是下半部是可中断的,而上半部是不可中断的,下半部几乎做了中断处理程序所有的事情,而且可以被新的中断打断,下半部则相对来说并不是非常紧急的,通常还是比较耗时的,因此由系统自行安排运行时机,不在中断服务上下文中执行。在Linux驱动程序中,为设备实现一个中断包含两个步骤:1)向内核注册中断、2)实现中断处理函数。其中request_irq用于实现中断的注册功能:
int request_irq(unsigned int irq, void (*handler)(int, void*, struct pt_regs *), unsigned long flags, const char *devname, void *dev_id)
中断处理程序:
中断处理程序是在中断上下文中运行的,它的行为受到某些限制:
1) 不能向用户空间发送或接受数据
2) 不能使用可能引起阻塞的函数
3) 不能使用可能引起调度的函数
下半部:
不可中断部分的共同部分放在函数do_IRQ中,需要添加中断处理函数时,通过request_irq实现。下半部放在do_softirq中,也就是软中断,通过open_softirq添加对应的处理函数。


浙公网安备 33010602011771号