xv6, line by line
kernel/sysfile.c

kernel/sysfile.c

C · 530 lines · annotated 100% · kernel · upstream

About this file

The kernel half of every system call that deals with files: open, read, write, close, dup, fstat, link, unlink, mkdir, mknod, chdir, pipe and exec. Each sys_ function is reached through the table in kernel/syscall.c after a user program executes ecall.

The functions here mostly check and translate. They fetch the arguments the user passed in registers (system call arguments), turn a small integer file descriptor into a pointer to an open file (struct file), turn a path name into an inode, and then hand the real work to the layers below: kernel/file.c (open files), kernel/fs.c (inodes and directories), kernel/pipe.c and kernel/exec.c.

Two disciplines run through the file and are worth watching for in every function:

  • Transactions. Every call that may change the disk, or may drop the last reference to an inode (which can free it), runs between begin_op and end_op, so that all its disk writes form one transaction in the write-ahead log and survive a crash together or not at all.
  • Inode locking order. When two inodes must be locked at once, the directory is locked before the entry inside it. Following one order everywhere is what prevents deadlock.

Read before: kernel/syscall.c, kernel/file.c, kernel/fs.c. Read next: kernel/exec.c and kernel/pipe.c.

1//
2// File-system system calls.
3// Mostly argument checking, since we don't trust
4// user code, and calls into file.c and fs.c.
5//
7#include "types.h"
8#include "riscv.h"
9#include "defs.h"
10#include "param.h"
11#include "stat.h"
13#include "proc.h"
14#include "fs.h"
16#include "file.h"
17#include "fcntl.h"
19// Fetch the nth word-sized system call argument as a file descriptor
20// and return both the descriptor and the corresponding struct file.
21static int
22argfd(int n, int *pfd, struct file **pf)
24 int fd;
25 struct file *f;
28 if (fd < 0 || fd >= NOFILE || (f = myproc()->ofile[fd]) == 0)
29 return -1;
30 if (pfd)
31 *pfd = fd;
32 if (pf)
33 *pf = f;
34 return 0;
37// Allocate a file descriptor for the given file.
38// Takes over file reference from caller on success.
39static int
40fdalloc(struct file *f)
42 int fd;
43 struct proc *p = myproc();
45 for (fd = 0; fd < NOFILE; fd++) {
46 if (p->ofile[fd] == 0) {
47 p->ofile[fd] = f;
48 return fd;
49 }
50 }
51 return -1;
55sys_dup(void)
57 struct file *f;
58 int fd;
60 if (argfd(0, 0, &f) < 0)
61 return -1;
62 if ((fd = fdalloc(f)) < 0)
63 return -1;
65 return fd;
71 struct file *f;
72 int n;
75 argaddr(1, &p);
76 argint(2, &n);
77 if (argfd(0, 0, &f) < 0)
78 return -1;
79 return fileread(f, p, n);
85 struct file *f;
86 int n;
89 argaddr(1, &p);
90 argint(2, &n);
91 if (argfd(0, 0, &f) < 0)
92 return -1;
94 return filewrite(f, p, n);
100 int fd;
101 struct file *f;
103 if (argfd(0, &fd, &f) < 0)
104 return -1;
105 myproc()->ofile[fd] = 0;
107 return 0;
113 struct file *f;
114 uint64 st; // user pointer to struct stat
116 argaddr(1, &st);
117 if (argfd(0, 0, &f) < 0)
118 return -1;
119 return filestat(f, st);
122// Create the path new as a link to the same inode as old.
127 struct inode *dp, *ip;
129 if (argstr(0, old, MAXPATH) < 0 || argstr(1, new, MAXPATH) < 0)
130 return -1;
133 if ((ip = namei(old)) == 0) {
135 return -1;
136 }
139 if (ip->type == T_DIR) {
142 return -1;
143 }
145 if (ip->nlink >= NLINK_MAX) {
148 return -1;
149 }
155 if ((dp = nameiparent(new, name)) == 0)
156 goto bad;
158 // dp may have been unlinked while we resolved it; linking into an
159 // orphaned directory leaks ip (itrunc discards the record without
160 // dropping ip->nlink). create() has the same guard.
161 if (dp->nlink == 0) {
163 goto bad;
164 }
165 if (dp->dev != ip->dev || dirlink(dp, name, ip->inum) < 0) {
167 goto bad;
168 }
174 return 0;
182 return -1;
185// Is the directory dp empty except for "." and ".." ?
186static int
189 int off;
190 struct dirent de;
192 for (off = 2 * sizeof(de); off < dp->size; off += sizeof(de)) {
193 if (readi(dp, 0, (uint64)&de, off, sizeof(de)) != sizeof(de))
194 panic("isdirempty: readi");
195 if (de.inum != 0)
196 return 0;
197 }
198 return 1;
204 struct inode *ip, *dp;
205 struct dirent de;
209 if (argstr(0, path, MAXPATH) < 0)
210 return -1;
213 if ((dp = nameiparent(path, name)) == 0) {
215 return -1;
216 }
220 // Cannot unlink "." or "..".
221 if (namecmp(name, ".") == 0 || namecmp(name, "..") == 0)
222 goto bad;
224 if ((ip = dirlookup(dp, name, &off)) == 0)
225 goto bad;
228 if (ip->nlink < 1)
229 panic("unlink: nlink < 1");
230 if (ip->type == T_DIR && !isdirempty(ip)) {
232 goto bad;
233 }
235 memset(&de, 0, sizeof(de));
236 if (writei(dp, 0, (uint64)&de, off, sizeof(de)) != sizeof(de))
237 panic("unlink: writei");
238 if (ip->type == T_DIR) {
241 }
250 return 0;
255 return -1;
258static struct inode *
259create(char *path, short type, short major, short minor)
261 struct inode *ip, *dp;
262 char name[DIRSIZ];
264 if ((dp = nameiparent(path, name)) == 0)
265 return 0;
269 if (dp->nlink == 0) {
271 return 0;
272 }
274 // a new directory's ".." would push dp->nlink past its maximum
275 if (type == T_DIR && dp->nlink >= NLINK_MAX) {
277 return 0;
278 }
280 if ((ip = dirlookup(dp, name, 0)) != 0) {
283 if (type == T_FILE && (ip->type == T_FILE || ip->type == T_DEVICE))
284 return ip;
286 return 0;
287 }
289 if ((ip = ialloc(dp->dev, type)) == 0) {
291 return 0;
292 }
297 ip->nlink = 1;
300 if (type == T_DIR) { // Create . and .. entries.
301 // No ip->nlink++ for ".": avoid cyclic ref count.
302 if (dirlink(ip, ".", ip->inum) < 0 || dirlink(ip, "..", dp->inum) < 0)
303 goto fail;
304 }
306 if (dirlink(dp, name, ip->inum) < 0)
307 goto fail;
309 if (type == T_DIR) {
310 // now that success is guaranteed:
311 dp->nlink++; // for ".."
313 }
317 return ip;
320 // something went wrong. de-allocate ip.
321 ip->nlink = 0;
325 return 0;
332 int fd, omode;
333 struct file *f;
334 struct inode *ip;
335 int n;
338 if ((n = argstr(0, path, MAXPATH)) < 0)
339 return -1;
343 if (omode & O_CREATE) {
344 ip = create(path, T_FILE, 0, 0);
345 if (ip == 0) {
347 return -1;
348 }
349 } else {
350 if ((ip = namei(path)) == 0) {
352 return -1;
353 }
355 if (ip->type == T_DIR && omode != O_RDONLY) {
358 return -1;
359 }
360 }
362 if (ip->type == T_DEVICE && (ip->major < 0 || ip->major >= NDEV)) {
365 return -1;
366 }
368 if ((f = filealloc()) == 0 || (fd = fdalloc(f)) < 0) {
369 if (f)
373 return -1;
374 }
376 if (ip->type == T_DEVICE) {
379 } else {
381 f->off = 0;
382 }
383 f->ip = ip;
387 if ((omode & O_TRUNC) && ip->type == T_FILE) {
389 }
394 return fd;
401 struct inode *ip;
404 if (argstr(0, path, MAXPATH) < 0 || (ip = create(path, T_DIR, 0, 0)) == 0) {
406 return -1;
407 }
410 return 0;
416 struct inode *ip;
423 if ((argstr(0, path, MAXPATH)) < 0 ||
424 (ip = create(path, T_DEVICE, major, minor)) == 0) {
426 return -1;
427 }
430 return 0;
437 struct inode *ip;
438 struct proc *p = myproc();
441 if (argstr(0, path, MAXPATH) < 0 || (ip = namei(path)) == 0) {
443 return -1;
444 }
446 if (ip->type != T_DIR) {
449 return -1;
450 }
454 p->cwd = ip;
455 return 0;
462 int i;
466 if (argstr(0, path, MAXPATH) < 0) {
467 return -1;
468 }
469 memset(argv, 0, sizeof(argv));
470 for (i = 0;; i++) {
471 if (i >= NELEM(argv)) {
472 goto bad;
473 }
474 if (fetchaddr(uargv + sizeof(uint64) * i, (uint64 *)&uarg) < 0) {
475 goto bad;
476 }
477 if (uarg == 0) {
478 argv[i] = 0;
479 break;
480 }
482 if (argv[i] == 0)
483 goto bad;
484 if (fetchstr(uarg, argv[i], PGSIZE) < 0)
485 goto bad;
486 }
488 int ret = kexec(path, argv);
490 for (i = 0; i < NELEM(argv) && argv[i] != 0; i++)
493 return ret;
496 for (i = 0; i < NELEM(argv) && argv[i] != 0; i++)
498 return -1;
504 uint64 fdarray; // user pointer to array of two integers
505 struct file *rf, *wf;
506 int fd0, fd1;
507 struct proc *p = myproc();
510 if (pipealloc(&rf, &wf) < 0)
511 return -1;
512 fd0 = -1;
513 if ((fd0 = fdalloc(rf)) < 0 || (fd1 = fdalloc(wf)) < 0) {
514 if (fd0 >= 0)
515 p->ofile[fd0] = 0;
518 return -1;
519 }
520 if (copyout(p->pagetable, p->sz, fdarray, (char *)&fd0, sizeof(fd0)) < 0 ||
521 copyout(p->pagetable, p->sz, fdarray + sizeof(fd0), (char *)&fd1,
522 sizeof(fd1)) < 0) {
523 p->ofile[fd0] = 0;
524 p->ofile[fd1] = 0;
527 return -1;
528 }
529 return 0;