diff --git a/lib/libc/sys-minix/mmap.c b/lib/libc/sys-minix/mmap.c index f5d5915a7..f213f8fbf 100644 --- a/lib/libc/sys-minix/mmap.c +++ b/lib/libc/sys-minix/mmap.c @@ -14,8 +14,6 @@ __weak_alias(vm_getphys, _vm_getphys) __weak_alias(vm_getrefcount, _vm_getrefcount) __weak_alias(minix_mmap, _minix_mmap) __weak_alias(minix_munmap, _minix_munmap) -__weak_alias(mmap, _minix_mmap) -__weak_alias(munmap, _minix_munmap) #endif diff --git a/servers/vfs/const.h b/servers/vfs/const.h index 6c2ff41cd..46dd5b03c 100644 --- a/servers/vfs/const.h +++ b/servers/vfs/const.h @@ -1,11 +1,18 @@ #ifndef __VFS_CONST_H__ #define __VFS_CONST_H__ +/* How many FD's must be reserved per process? OPEN_MAX are + * user-visible, SYSTEM_FDS_MAX are system-usable. + */ +#define SYSTEM_FDSTART OPEN_MAX /* How many user FD's */ +#define SYSTEM_FDS_MAX OPEN_MAX /* How many system FD's */ +#define FDS_PER_PROCESS (OPEN_MAX+SYSTEM_FDS_MAX) /* Total FD's */ + /* Tables sizes */ -#define NR_FILPS 512 /* # slots in filp table */ +#define NR_FILPS 1024 /* # slots in filp table */ #define NR_LOCKS 8 /* # slots in the file locking table */ #define NR_MNTS 16 /* # slots in mount table */ -#define NR_VNODES 512 /* # slots in vnode table */ +#define NR_VNODES 1024 /* # slots in vnode table */ #define NR_WTHREADS 8 /* # slots in worker thread table */ #define NR_NONEDEVS NR_MNTS /* # slots in nonedev bitmap */ diff --git a/servers/vfs/coredump.c b/servers/vfs/coredump.c index 9b5d24573..578d43df5 100644 --- a/servers/vfs/coredump.c +++ b/servers/vfs/coredump.c @@ -177,7 +177,7 @@ static void adjust_offsets(Elf_Phdr phdrs[], int phnum) *===========================================================================*/ static void write_buf(struct filp *f, char *buf, size_t size) { - read_write(WRITING, f, buf, size, VFS_PROC_NR); + read_write(fp, WRITING, f, buf, size, VFS_PROC_NR); } /*===========================================================================* diff --git a/servers/vfs/exec.c b/servers/vfs/exec.c index f21ffb18c..2ad9826dc 100644 --- a/servers/vfs/exec.c +++ b/servers/vfs/exec.c @@ -312,7 +312,7 @@ int pm_exec(endpoint_t proc_e, vir_bytes path, size_t path_len, /* The interpreter (loader) needs an fd to the main program, * which is currently in finalexec */ - if((r = execi.elf_main_fd = common_open(finalexec, O_RDONLY, 0)) < 0) { + if((r = execi.elf_main_fd = common_open(finalexec, O_RDONLY, 0, 1)) < 0) { printf("VFS: exec: dynamic: open main exec failed %s (%d)\n", fullpath, r); FAILCHECK(r); @@ -338,19 +338,16 @@ int pm_exec(endpoint_t proc_e, vir_bytes path, size_t path_len, /* We also want an FD so VM can mmap() the process in if possible. */ { int openr; - fp->fp_fdscan = OPEN_MAX/2; - if((openr=execi.procfd=common_open(firstexec, O_RDONLY, 0)) < 0) { + if((openr=execi.procfd=common_open(firstexec, O_RDONLY, 0, 0)) < 0) { printf("vfs: exec: open failed of %s (%d), can't mmap\n", firstexec, openr); execi.procfd = -1; } else { if(fp->fp_filp[openr]->filp_vno->v_vmnt->m_haspeek && major(fp->fp_filp[openr]->filp_vno->v_dev) != MEMORY_MAJOR) { - FD_SET(openr, &fp->fp_filp_system); /* systemfd */ execi.args.memmap = vfs_memmap; } } - fp->fp_fdscan = 0; } /* callback functions and data */ @@ -706,7 +703,7 @@ static void clo_exec(struct fproc *rfp) int i; /* Check the file desriptors one by one for presence of FD_CLOEXEC. */ - for (i = 0; i < OPEN_MAX; i++) + for (i = 0; i < FDS_PER_PROCESS; i++) if ( FD_ISSET(i, &rfp->fp_cloexec_set)) (void) close_fd(rfp, i, 0); } diff --git a/servers/vfs/filedes.c b/servers/vfs/filedes.c index 6cac9fabe..2e404f16e 100644 --- a/servers/vfs/filedes.c +++ b/servers/vfs/filedes.c @@ -15,12 +15,13 @@ * the receiver. */ +#include "fs.h" + #include #include #include #include #include -#include "fs.h" #include "file.h" #include "fproc.h" #include "vnode.h" @@ -147,7 +148,8 @@ void init_filps(void) /*===========================================================================* * get_fd * *===========================================================================*/ -int get_fd(int start, mode_t bits, int *k, struct filp **fpt) +int get_fd(struct fproc *rfp, int start, mode_t bits, int *k, + struct filp **fpt, int userfd) { /* Look for a free file descriptor and a free filp slot. Fill in the mode word * in the latter, but don't claim either one yet, since the open() or creat() @@ -156,10 +158,11 @@ int get_fd(int start, mode_t bits, int *k, struct filp **fpt) register struct filp *f; register int i; + int limit = userfd ? OPEN_MAX : FDS_PER_PROCESS; /* Search the fproc fp_filp table for a free file descriptor. */ - for (i = start; i < OPEN_MAX; i++) { - if (fp->fp_filp[i] == NULL && !FD_ISSET(i, &fp->fp_filp_inuse)) { + for (i = start; i < limit; i++) { + if (rfp->fp_filp[i] == NULL && !FD_ISSET(i, &rfp->fp_filp_inuse)) { /* A file descriptor has been located. */ *k = i; break; @@ -167,7 +170,7 @@ int get_fd(int start, mode_t bits, int *k, struct filp **fpt) } /* Check to see if a file descriptor has been found. */ - if (i >= OPEN_MAX) return(EMFILE); + if (i >= limit) return(EMFILE); /* If we don't care about a filp, return now */ if (fpt == NULL) return(OK); @@ -203,36 +206,33 @@ int fild; /* file descriptor */ tll_access_t locktype; { /* See if 'fild' refers to a valid file descr. If so, return its filp ptr. */ - return get_filp2(fp, fild, locktype); + return get_filp2(fp, fild, locktype, 1); } /*===========================================================================* * get_filp2 * *===========================================================================*/ -struct filp *get_filp2(rfp, fild, locktype) +struct filp *get_filp2(rfp, fild, locktype, userrequest) register struct fproc *rfp; int fild; /* file descriptor */ tll_access_t locktype; +int userrequest; { /* See if 'fild' refers to a valid file descr. If so, return its filp ptr. */ struct filp *filp; + int fdlimit = userrequest ? OPEN_MAX : FDS_PER_PROCESS; filp = NULL; - if (fild < 0 || fild >= OPEN_MAX) + if (fild < 0 || fild >= fdlimit) { err_code = EBADF; - else if (rfp->fp_filp[fild] == NULL && FD_ISSET(fild, &rfp->fp_filp_inuse)) + } else if (rfp->fp_filp[fild] == NULL && FD_ISSET(fild, &rfp->fp_filp_inuse)) err_code = EIO; /* The filedes is not there, but is not closed either. */ - else if ((filp = rfp->fp_filp[fild]) == NULL) + else if ((filp = rfp->fp_filp[fild]) == NULL) { err_code = EBADF; - else { - if (FD_ISSET(fild, &fp->fp_filp_system)) { -#if 0 - printf("VFS: warning: doing get_filp of system fd %d: %d\n", - fp->fp_endpoint, fild); -#endif - } + } else { + assert(filp->filp_count > 0); lock_filp(filp, locktype); /* All is fine */ } @@ -275,7 +275,7 @@ int invalidate_filp(struct filp *rfilp) int f, fd, n = 0; for (f = 0; f < NR_PROCS; f++) { if (fproc[f].fp_pid == PID_FREE) continue; - for (fd = 0; fd < OPEN_MAX; fd++) { + for (fd = 0; fd < FDS_PER_PROCESS; fd++) { if(fproc[f].fp_filp[fd] && fproc[f].fp_filp[fd] == rfilp) { fproc[f].fp_filp[fd] = NULL; n++; @@ -441,7 +441,7 @@ int fd; if (isokendpt(ep, &slot) != OK) return(NULL); - rfilp = get_filp2(&fproc[slot], fd, VNODE_READ); + rfilp = get_filp2(&fproc[slot], fd, VNODE_READ, 1); return(rfilp); } @@ -504,7 +504,7 @@ filp_id_t cfilp; rfp = &fproc[slot]; /* Find an open slot in fp_filp */ - for (fd = 0; fd < OPEN_MAX; fd++) { + for (fd = 0; fd < FDS_PER_PROCESS; fd++) { if (rfp->fp_filp[fd] == NULL && !FD_ISSET(fd, &rfp->fp_filp_inuse)) { diff --git a/servers/vfs/fproc.h b/servers/vfs/fproc.h index 634296186..dfab0afe4 100644 --- a/servers/vfs/fproc.h +++ b/servers/vfs/fproc.h @@ -21,10 +21,9 @@ EXTERN struct fproc { struct vnode *fp_wd; /* working directory; NULL during reboot */ struct vnode *fp_rd; /* root directory; NULL during reboot */ - struct filp *fp_filp[OPEN_MAX];/* the file descriptor table */ + struct filp *fp_filp[FDS_PER_PROCESS]; /* the file descriptor table */ fd_set fp_filp_inuse; /* which fd's are in use? */ fd_set fp_cloexec_set; /* bit map for POSIX Table 6-2 FD_CLOEXEC */ - fd_set fp_filp_system; /* system-reserved; user process can't touch it */ dev_t fp_tty; /* major/minor of controlling tty */ @@ -48,7 +47,6 @@ EXTERN struct fproc { struct job fp_job; /* pending job */ thread_t fp_wtid; /* Thread ID of worker */ char fp_name[PROC_NAME_LEN]; /* Last exec() */ - int fp_fdscan; /* Start scanning fd's here */ #if LOCK_DEBUG int fp_vp_rdlocks; /* number of read-only locks on vnodes */ int fp_vmnt_rdlocks; /* number of read-only locks on vmnts */ diff --git a/servers/vfs/fs.h b/servers/vfs/fs.h index b628f69d0..d5fbf0f5f 100644 --- a/servers/vfs/fs.h +++ b/servers/vfs/fs.h @@ -10,7 +10,8 @@ #include "fdset.h" /* The following are so basic, all the *.c files get them automatically. */ -#include /* MUST be first */ +#include + #include #include #include @@ -26,7 +27,6 @@ #include #include -#include "const.h" #include "dmap.h" #include "proto.h" #include "threads.h" diff --git a/servers/vfs/main.c b/servers/vfs/main.c index 437e794de..abef931e7 100644 --- a/servers/vfs/main.c +++ b/servers/vfs/main.c @@ -265,6 +265,7 @@ static void *do_control_msgs(void *arg) /* Check for special control messages. */ if (job_m_in.m_source == CLOCK) { /* Alarm timer expired. Used only for select(). Check it. */ + printf("handling clock message.\n"); expire_timers(job_m_in.NOTIFY_TIMESTAMP); } diff --git a/servers/vfs/misc.c b/servers/vfs/misc.c index e11e08a6f..7d800bf70 100644 --- a/servers/vfs/misc.c +++ b/servers/vfs/misc.c @@ -131,7 +131,7 @@ int do_fcntl(message *UNUSED(m_out)) case F_DUPFD: /* This replaces the old dup() system call. */ if (fcntl_argx < 0 || fcntl_argx >= OPEN_MAX) r = EINVAL; - else if ((r = get_fd(fcntl_argx, 0, &new_fd, NULL)) == OK) { + else if ((r = get_fd(fp, fcntl_argx, 0, &new_fd, NULL, 1)) == OK) { f->filp_count++; fp->fp_filp[new_fd] = f; FD_SET(new_fd, &fp->fp_filp_inuse); @@ -330,7 +330,7 @@ int do_vm_call(message *m_out) u32_t length = job_m_in.VFS_VMCALL_LENGTH; int result = OK; int slot; - struct fproc *fdsrc, *vmf; + struct fproc *rfp, *vmf; struct filp *f = NULL; if(job_m_in.m_source != VM_PROC_NR) @@ -338,16 +338,10 @@ int do_vm_call(message *m_out) okendpt(ep, &slot); - fdsrc = &fproc[slot]; + rfp = &fproc[slot]; vmf = &fproc[VM_PROC_NR]; assert(fp == vmf); - - /* We do everything on behalf of the target process. */ - fp = fdsrc; - - if(fp == vmf) { - printf("VFS: huh, performing request for VM?\n"); - } + assert(rfp != vmf); switch(req) { case VMVFSREQ_FDLOOKUP: @@ -355,7 +349,7 @@ int do_vm_call(message *m_out) int procfd; /* Lookup fd in referenced process. */ - if ((f = get_filp(req_fd, VNODE_READ)) == NULL) { + if ((f = get_filp2(rfp, req_fd, VNODE_READ, 1)) == NULL) { result = err_code; goto reqdone; } @@ -376,7 +370,8 @@ int do_vm_call(message *m_out) goto reqdone; } - if((result=get_fd(OPEN_MAX/2, 0, &procfd, NULL)) != OK) + if((result=get_fd(rfp, SYSTEM_FDSTART, 0, &procfd, + NULL, 0)) != OK) goto reqdone; if(S_ISBLK(f->filp_vno->v_mode)) { @@ -392,57 +387,47 @@ int do_vm_call(message *m_out) roundup(f->filp_vno->v_size, PAGE_SIZE)/PAGE_SIZE; } -#if 1 + m_out->VMV_FD = procfd; f->filp_count++; - fp->fp_filp[procfd] = f; + assert(f->filp_count > 0); + rfp->fp_filp[procfd] = f; - /* mmap FD's are inuse, system-owned (user can't - * close it), and close-on-exec - */ - FD_SET(procfd, &fp->fp_filp_inuse); - FD_SET(procfd, &fp->fp_filp_system); - FD_SET(procfd, &fp->fp_cloexec_set); -#else - m_out->VMV_FD = req_fd; -#endif + /* mmap FD's are inuse and close-on-exec */ + FD_SET(procfd, &rfp->fp_filp_inuse); + FD_SET(procfd, &rfp->fp_cloexec_set); -#if 0 - printf("%d: fd %d is system now\n", fp->fp_endpoint, procfd); -#endif result = OK; #if 0 - printf("VFS: FDLOOKUP: result %d dev 0x%x ino %ld; procfd %d\n", - result, m_out->VMV_DEV, m_out->VMV_INO, procfd); + printf("VFS: FDLOOKUP: result %d dev 0x%x ino %ld; procfd %d for %d\n", + result, m_out->VMV_DEV, m_out->VMV_INO, procfd, rfp->fp_endpoint); #endif break; } case VMVFSREQ_FDCLOSE: { - int procfd = req_fd; - scratch(fp).file.fd_nr = procfd; - result = close_fd(fp, scratch(fp).file.fd_nr, 0); + result = close_fd(rfp, req_fd, 0); if(result != OK) { printf("VFS: VM fd close for fd %d, %d (%d)\n", - procfd, fp->fp_endpoint, result); + req_fd, rfp->fp_endpoint, result); } break; } case VMVFSREQ_FDIO: { message dummy_out; - result = actual_llseek(&dummy_out, req_fd, SEEK_SET, - offset); + + result = actual_llseek(rfp, &dummy_out, req_fd, + SEEK_SET, offset, 0); if(result != OK) { - printf("VFS: llseek for mmap i/o %d (%llu) failed (%d)\n", - fp->fp_endpoint, offset, result); + printf("VFS: llseek for mmap i/o for %d, fd %d, (offset %llu) failed (%d)\n", + rfp->fp_endpoint, req_fd, offset, result); } else { - fp = fdsrc; - result = do_read_write_peek(PEEKING, - req_fd, NULL, length); + result = actual_read_write_peek(rfp, PEEKING, + req_fd, NULL, length, 0); } break; @@ -456,8 +441,8 @@ reqdone: if(f) unlock_filp(f); - /* We're VM now. */ - fp = vmf; + /* fp is VM still. */ + assert(fp == vmf); m_out->VMV_ENDPOINT = ep; m_out->VMV_RESULT = result; m_out->VMV_REQID = req_id; @@ -551,7 +536,7 @@ void pm_fork(endpoint_t pproc, endpoint_t cproc, pid_t cpid) cp = &fproc[childno]; pp = &fproc[parentno]; - for (i = 0; i < OPEN_MAX; i++) + for (i = 0; i < FDS_PER_PROCESS; i++) if (cp->fp_filp[i] != NULL) cp->fp_filp[i]->filp_count++; /* Fill in new process and endpoint id. */ @@ -595,7 +580,7 @@ static void free_proc(struct fproc *exiter, int flags) unpause(exiter->fp_endpoint); /* Loop on file descriptors, closing any that are open. */ - for (i = 0; i < OPEN_MAX; i++) { + for (i = 0; i < FDS_PER_PROCESS; i++) { (void) close_fd(exiter, i, 0); } @@ -632,7 +617,7 @@ static void free_proc(struct fproc *exiter, int flags) if(rfp->fp_pid == PID_FREE) continue; if (rfp->fp_tty == dev) rfp->fp_tty = 0; - for (i = 0; i < OPEN_MAX; i++) { + for (i = 0; i < FDS_PER_PROCESS; i++) { if ((rfilp = rfp->fp_filp[i]) == NULL) continue; if (rfilp->filp_mode == FILP_CLOSED) continue; vp = rfilp->filp_vno; @@ -853,7 +838,7 @@ int pm_dumpcore(endpoint_t proc_e, int csig, vir_bytes exe_name) /* open core file */ snprintf(core_path, PATH_MAX, "%s.%d.%d", CORE_NAME, fp->fp_pid, seq); - core_fd = common_open(core_path, O_WRONLY | O_CREAT | O_TRUNC, CORE_MODE); + core_fd = common_open(core_path, O_WRONLY | O_CREAT | O_TRUNC, CORE_MODE, 1); if (core_fd < 0) { r = core_fd; goto core_exit; } /* get process' name */ @@ -929,3 +914,4 @@ void panic_hook(void) printf("VFS mthread stacktraces:\n"); mthread_stacktraces(); } + diff --git a/servers/vfs/open.c b/servers/vfs/open.c index d574c4a28..3a4b1bfca 100644 --- a/servers/vfs/open.c +++ b/servers/vfs/open.c @@ -37,37 +37,6 @@ static struct vnode *new_node(struct lookup *resolve, int oflags, mode_t bits); static int pipe_open(struct vnode *vp, mode_t bits, int oflags); -/*===========================================================================* - * lock_close * - *===========================================================================*/ -static void lock_close(void) -{ - struct fproc *org_fp; - struct worker_thread *org_self; - - /* First try to get it right off the bat */ - if (mutex_trylock(&closefd_lock) == 0) - return; - - org_fp = fp; - org_self = self; - - if (mutex_lock(&closefd_lock) != 0) - panic("Could not obtain lock on close"); - - fp = org_fp; - self = org_self; -} - -/*===========================================================================* - * unlock_close * - *===========================================================================*/ -static void unlock_close(void) -{ - if (mutex_unlock(&closefd_lock) != 0) - panic("Could not release lock on closefd_lock"); -} - /*===========================================================================* * do_open * *===========================================================================*/ @@ -103,14 +72,14 @@ int do_open(message *UNUSED(m_out)) } if (r != OK) return(err_code); /* name was bad */ - return common_open(fullpath, open_mode, create_mode); + return common_open(fullpath, open_mode, create_mode, 1); } /*===========================================================================* * common_open * *===========================================================================*/ -int common_open(char path[PATH_MAX], int oflags, mode_t omode) +int common_open(char path[PATH_MAX], int oflags, mode_t omode, int userfd) { /* Common code from do_creat and do_open. */ int b, r, exist = TRUE, major_dev; @@ -121,13 +90,16 @@ int common_open(char path[PATH_MAX], int oflags, mode_t omode) struct vmnt *vmp; struct dmap *dp; struct lookup resolve; + int start = userfd ? 0 : SYSTEM_FDSTART; /* Remap the bottom two bits of oflags. */ bits = (mode_t) mode_map[oflags & O_ACCMODE]; if (!bits) return(EINVAL); /* See if file descriptor and filp slots are available. */ - if ((r = get_fd(fp->fp_fdscan, bits, &(scratch(fp).file.fd_nr), &filp)) != OK) return(r); + if ((r = get_fd(fp, start, bits, &(scratch(fp).file.fd_nr), + &filp, userfd)) != OK) + return(r); lookup_init(&resolve, path, PATH_NOFLAGS, &vmp, &vp); @@ -681,7 +653,8 @@ int do_lseek(message *m_out) /*===========================================================================* * actual_llseek * *===========================================================================*/ -int actual_llseek(message *m_out, int seekfd, int seekwhence, u64_t offset) +int actual_llseek(struct fproc *rfp, message *m_out, int seekfd, int seekwhence, + u64_t offset, int userreq) { /* Perform the llseek(ls_fd, offset, whence) system call. */ register struct filp *rfilp; @@ -690,7 +663,11 @@ int actual_llseek(message *m_out, int seekfd, int seekwhence, u64_t offset) long off_hi = ex64hi(offset); /* Check to see if the file descriptor is valid. */ - if ( (rfilp = get_filp(seekfd, VNODE_READ)) == NULL) return(err_code); + if ( (rfilp = get_filp2(rfp, seekfd, VNODE_READ, userreq)) == NULL) { + printf("actual_llseek: get_filp2 failed for fp %d fd %d userreq %d err %d\n", + rfp->fp_endpoint, seekfd, userreq, err_code); + return(err_code); + } /* No lseek on pipes. */ if (S_ISFIFO(rfilp->filp_vno->v_mode)) { @@ -733,8 +710,8 @@ int actual_llseek(message *m_out, int seekfd, int seekwhence, u64_t offset) int do_llseek(message *m_out) { - return actual_llseek(m_out, job_m_in.ls_fd, job_m_in.whence, - make64(job_m_in.offset_lo, job_m_in.offset_high)); + return actual_llseek(fp, m_out, job_m_in.ls_fd, job_m_in.whence, + make64(job_m_in.offset_lo, job_m_in.offset_high), 1); } /*===========================================================================* @@ -744,8 +721,7 @@ int do_close(message *UNUSED(m_out)) { /* Perform the close(fd) system call. */ int thefd = job_m_in.fd; - scratch(fp).file.fd_nr = thefd; - return close_fd(fp, scratch(fp).file.fd_nr, 1); + return close_fd(fp, thefd, 1); } @@ -763,28 +739,20 @@ int user_request; struct file_lock *flp; int lock_count; - lock_close(); - - if(user_request && - fd_nr >= 0 && fd_nr < OPEN_MAX && FD_ISSET(fd_nr, &fp->fp_filp_system)) { -#if 0 - printf("VFS: close_fd: closing system FD %d from %d, not doing it\n", - fd_nr, rfp->fp_endpoint); -#endif - unlock_close(); - return EBADF; - } - /* First locate the vnode that belongs to the file descriptor. */ - if ( (rfilp = get_filp2(rfp, fd_nr, VNODE_OPCL)) == NULL) { - unlock_close(); + if ( (rfilp = get_filp2(rfp, fd_nr, VNODE_WRITE, user_request)) == NULL) { return(err_code); } vp = rfilp->filp_vno; - close_filp(rfilp); + /* first, make all future get_filp2()'s fail; otherwise + * we might try to close the same fd in different threads + */ rfp->fp_filp[fd_nr] = NULL; + + close_filp(rfilp); + FD_CLR(fd_nr, &rfp->fp_cloexec_set); FD_CLR(fd_nr, &rfp->fp_filp_inuse); @@ -802,8 +770,6 @@ int user_request; lock_revive(); /* one or more locks released */ } - unlock_close(); - return(OK); } diff --git a/servers/vfs/pipe.c b/servers/vfs/pipe.c index a6cc0f9e0..02dc3d684 100644 --- a/servers/vfs/pipe.c +++ b/servers/vfs/pipe.c @@ -101,7 +101,7 @@ static int create_pipe(int fil_des[2], int flags) /* Acquire two file descriptors. */ rfp = fp; - if ((r = get_fd(0, R_BIT, &fil_des[0], &fil_ptr0)) != OK) { + if ((r = get_fd(fp, 0, R_BIT, &fil_des[0], &fil_ptr0, 1)) != OK) { unlock_vnode(vp); unlock_vmnt(vmp); return(r); @@ -109,7 +109,7 @@ static int create_pipe(int fil_des[2], int flags) rfp->fp_filp[fil_des[0]] = fil_ptr0; FD_SET(fil_des[0], &rfp->fp_filp_inuse); fil_ptr0->filp_count = 1; /* mark filp in use */ - if ((r = get_fd(0, W_BIT, &fil_des[1], &fil_ptr1)) != OK) { + if ((r = get_fd(fp, 0, W_BIT, &fil_des[1], &fil_ptr1, 1)) != OK) { rfp->fp_filp[fil_des[0]] = NULL; FD_CLR(fil_des[0], &rfp->fp_filp_inuse); fil_ptr0->filp_count = 0; /* mark filp free */ @@ -601,7 +601,7 @@ void unpause(endpoint_t proc_e) } fild = scratch(rfp).file.fd_nr; - if (fild < 0 || fild >= OPEN_MAX) + if (fild < 0 || fild >= FDS_PER_PROCESS) panic("file descriptor out-of-range"); f = rfp->fp_filp[fild]; dev = (dev_t) f->filp_vno->v_sdev; /* device hung on */ diff --git a/servers/vfs/proto.h b/servers/vfs/proto.h index 49f6ad156..83c76282b 100644 --- a/servers/vfs/proto.h +++ b/servers/vfs/proto.h @@ -86,10 +86,11 @@ void check_filp_locks(void); void check_filp_locks_by_me(void); void init_filps(void); struct filp *find_filp(struct vnode *vp, mode_t bits); -int get_fd(int start, mode_t bits, int *k, struct filp **fpt); +int get_fd(struct fproc *rfp, int start, mode_t bits, int *k, + struct filp **fpt, int user); struct filp *get_filp(int fild, tll_access_t locktype); struct filp *get_filp2(struct fproc *rfp, int fild, tll_access_t - locktype); + locktype, int userrequest); void lock_filp(struct filp *filp, tll_access_t locktype); void unlock_filp(struct filp *filp); void unlock_filps(struct filp *filp1, struct filp *filp2); @@ -162,7 +163,7 @@ void unmount_all(int force); int do_close(message *m_out); int close_fd(struct fproc *rfp, int fd_nr, int flag); void close_reply(void); -int common_open(char path[PATH_MAX], int oflags, mode_t omode); +int common_open(char path[PATH_MAX], int oflags, mode_t omode, int userfd); int do_creat(void); int do_lseek(message *m_out); int do_llseek(message *m_out); @@ -171,7 +172,8 @@ int do_mkdir(message *m_out); int do_open(message *m_out); int do_slink(message *m_out); int actual_lseek(message *m_out, int seekfd, int seekwhence, off_t offset); -int actual_llseek(message *m_out, int seekfd, int seekwhence, u64_t offset); +int actual_llseek(struct fproc *rfp, message *m_out, int seekfd, + int seekwhence, u64_t offset, int userrequest); int do_vm_open(void); int do_vm_close(void); @@ -216,8 +218,10 @@ void lock_bsf(void); void unlock_bsf(void); void check_bsf_lock(void); int do_read_write_peek(int rw_flag, int fd, char *buf, size_t bytes); -int read_write(int rw_flag, struct filp *f, char *buffer, size_t nbytes, - endpoint_t for_e); +int actual_read_write_peek(struct fproc *rfp, int rw_flag, int fd, char *buf, + size_t bytes, int userreq); +int read_write(struct fproc *rfp, int rw_flag, struct filp *f, char *buffer, + size_t nbytes, endpoint_t for_e); int rw_pipe(int rw_flag, endpoint_t usr, struct filp *f, char *buf, size_t req_size); diff --git a/servers/vfs/read.c b/servers/vfs/read.c index 66c921631..b7c0b954c 100644 --- a/servers/vfs/read.c +++ b/servers/vfs/read.c @@ -83,9 +83,10 @@ void check_bsf_lock(void) } /*===========================================================================* - * do_read_write_peek * + * actual_read_write_peek * *===========================================================================*/ -int do_read_write_peek(int rw_flag, int io_fd, char *io_buf, size_t io_nbytes) +int actual_read_write_peek(struct fproc *rfp, int rw_flag, int io_fd, + char *io_buf, size_t io_nbytes, int userreq) { /* Perform read(fd, buffer, nbytes) or write(fd, buffer, nbytes) call. */ struct filp *f; @@ -95,34 +96,45 @@ int do_read_write_peek(int rw_flag, int io_fd, char *io_buf, size_t io_nbytes) if(rw_flag == WRITING) ro = 0; - scratch(fp).file.fd_nr = io_fd; - scratch(fp).io.io_buffer = io_buf; - scratch(fp).io.io_nbytes = io_nbytes; + scratch(rfp).file.fd_nr = io_fd; + scratch(rfp).io.io_buffer = io_buf; + scratch(rfp).io.io_nbytes = io_nbytes; - locktype = ro ? VNODE_READ : VNODE_WRITE; - if ((f = get_filp(scratch(fp).file.fd_nr, locktype)) == NULL) + locktype = rw_flag == READING ? VNODE_READ : VNODE_WRITE; + if ((f = get_filp2(rfp, scratch(rfp).file.fd_nr, locktype, userreq)) == NULL) return(err_code); + + assert(f->filp_count > 0); + if (((f->filp_mode) & (ro ? R_BIT : W_BIT)) == 0) { unlock_filp(f); return(f->filp_mode == FILP_CLOSED ? EIO : EBADF); } - if (scratch(fp).io.io_nbytes == 0) { + if (scratch(rfp).io.io_nbytes == 0) { unlock_filp(f); return(0); /* so char special files need not check for 0*/ } - r = read_write(rw_flag, f, scratch(fp).io.io_buffer, scratch(fp).io.io_nbytes, - who_e); + r = read_write(rfp, rw_flag, f, scratch(rfp).io.io_buffer, + scratch(rfp).io.io_nbytes, who_e); unlock_filp(f); return(r); } +/*===========================================================================* + * do_read_write_peek * + *===========================================================================*/ +int do_read_write_peek(int rw_flag, int io_fd, char *io_buf, size_t io_nbytes) +{ + return actual_read_write_peek(fp, rw_flag, io_fd, io_buf, io_nbytes, 1); +} + /*===========================================================================* * read_write * *===========================================================================*/ -int read_write(int rw_flag, struct filp *f, char *buf, size_t size, - endpoint_t for_e) +int read_write(struct fproc *rfp, int rw_flag, struct filp *f, + char *buf, size_t size, endpoint_t for_e) { register struct vnode *vp; u64_t position, res_pos; @@ -144,7 +156,7 @@ int read_write(int rw_flag, struct filp *f, char *buf, size_t size, op = (rw_flag == READING ? VFS_DEV_READ : VFS_DEV_WRITE); if (S_ISFIFO(vp->v_mode)) { /* Pipes */ - if (fp->fp_cum_io_partial != 0) { + if (rfp->fp_cum_io_partial != 0) { panic("VFS: read_write: fp_cum_io_partial not clear"); } if(rw_flag == PEEKING) { @@ -240,7 +252,7 @@ int read_write(int rw_flag, struct filp *f, char *buf, size_t size, * generate s SIGPIPE signal. */ if (!(f->filp_flags & O_NOSIGPIPE)) { - sys_kill(fp->fp_endpoint, SIGPIPE); + sys_kill(rfp->fp_endpoint, SIGPIPE); } } diff --git a/servers/vfs/select.c b/servers/vfs/select.c index 191b01ffc..98ce6f2c7 100644 --- a/servers/vfs/select.c +++ b/servers/vfs/select.c @@ -32,8 +32,8 @@ static struct selectentry { fd_set readfds, writefds, errorfds; fd_set ready_readfds, ready_writefds, ready_errorfds; fd_set *vir_readfds, *vir_writefds, *vir_errorfds; - struct filp *filps[OPEN_MAX]; - int type[OPEN_MAX]; + struct filp *filps[FDS_PER_PROCESS]; + int type[FDS_PER_PROCESS]; int nfds, nreadyfds; int error; char block; @@ -103,7 +103,7 @@ int do_select(message *UNUSED(m_out)) vtimeout = (vir_bytes) job_m_in.SEL_TIMEOUT; /* Sane amount of file descriptors? */ - if (nfds < 0 || nfds > OPEN_MAX) return(EINVAL); + if (nfds < 0 || nfds > FDS_PER_PROCESS) return(EINVAL); /* Find a slot to store this select request */ for (s = 0; s < MAXSELECTS; s++) @@ -543,7 +543,7 @@ static int copy_fdsets(struct selectentry *se, int nfds, int direction) endpoint_t src_e, dst_e; fd_set *src_fds, *dst_fds; - if (nfds < 0 || nfds > OPEN_MAX) + if (nfds < 0 || nfds > FDS_PER_PROCESS) panic("select copy_fdsets: nfds wrong: %d", nfds); /* Only copy back as many bits as the user expects. */ diff --git a/servers/vm/Makefile b/servers/vm/Makefile index ea9a67f37..f24cf428f 100644 --- a/servers/vm/Makefile +++ b/servers/vm/Makefile @@ -6,7 +6,7 @@ SRCS= main.c alloc.c utility.c exit.c fork.c break.c \ mmap.c slaballoc.c region.c pagefaults.c \ rs.c queryexit.c pb.c regionavl.c \ mem_anon.c mem_directphys.c mem_anon_contig.c mem_shared.c \ - mem_cache.c cache.c + mem_cache.c cache.c vfs.c mem_file.c fdref.c .if ${MACHINE_ARCH} == "earm" LDFLAGS+= -T ${.CURDIR}/arch/${MACHINE_ARCH}/vm.lds diff --git a/servers/vm/arch/i386/pagetable.c b/servers/vm/arch/i386/pagetable.c index 8008f1138..5a16bd60a 100644 --- a/servers/vm/arch/i386/pagetable.c +++ b/servers/vm/arch/i386/pagetable.c @@ -1330,10 +1330,6 @@ int pt_bind(pt_t *pt, struct vmproc *who) pdeslot * ARCH_PAGEDIR_SIZE); #endif -#if 0 - printf("VM: slot %d endpoint %d has pde val 0x%lx at kernel address 0x%lx\n", - slot, who->vm_endpoint, page_directories[slot], pdes); -#endif /* Tell kernel about new page table root. */ return sys_vmctl_set_addrspace(who->vm_endpoint, pt->pt_dir_phys, pdes); } diff --git a/servers/vm/exit.c b/servers/vm/exit.c index f287b327a..686bec593 100644 --- a/servers/vm/exit.c +++ b/servers/vm/exit.c @@ -43,6 +43,7 @@ void clear_proc(struct vmproc *vmp) vmp->vm_bytecopies = 0; #endif vmp->vm_region_top = 0; + vmp->fdrefs = NULL; } /*===========================================================================* @@ -118,10 +119,12 @@ int do_procctl(message *msg) if(msg->m_source != RS_PROC_NR && msg->m_source != VFS_PROC_NR) return EPERM; + vmp->vm_flags |= VMF_EXECING; free_proc(vmp); if(pt_new(&vmp->vm_pt) != OK) panic("VMPPARAM_CLEAR: pt_new failed"); pt_bind(&vmp->vm_pt, vmp); + vmp->vm_flags &= ~VMF_EXECING; return OK; default: return EINVAL; diff --git a/servers/vm/fdref.c b/servers/vm/fdref.c new file mode 100644 index 000000000..4239ddd5d --- /dev/null +++ b/servers/vm/fdref.c @@ -0,0 +1,132 @@ + +/* File that implements the 'fdref' data structure. It keeps track + * of how many times a particular fd (per process) is referenced by + * mmapped objects. + * + * This is used to + * - have many references to the same file, without needing an FD each + * - deciding when we have to close an FD (last reference disappears) + * + * Examples: + * - if a file-mmapped region is split, the refcount increases; there are + * now two regions referencing the same FD. We can't simply close the + * FD once either region is unmapped, as the pagefaults for the other + * would stop working. So we increase the refcount to that fd. + * - if a new file-maped region is requested, we might find out it's the + * same dev/inode the same process already has referenced. we could + * decide to close the new reference and use an existing one, so + * references to the same file aren't fd-limited. + * - if a file-mapped region is copied, we have to create a new + * fdref object, as the source process might disappear; we have to + * use the new process' fd for it. + */ + +#include +#include + +#include + +#include "proto.h" +#include "vm.h" +#include "fdref.h" +#include "vmproc.h" + +struct fdref *fdref_new(struct vmproc *owner, ino_t ino, dev_t dev, int fd) +{ + struct fdref *fdref; + + if(!SLABALLOC(fdref)) return NULL; + + fdref->fd = fd; + fdref->owner = owner; + fdref->refcount = 0; + fdref->dev = dev; + fdref->ino = ino; + fdref->next = owner->fdrefs; + owner->fdrefs = fdref; + + return fdref; +} + +void fdref_ref(struct fdref *ref, struct vir_region *region) +{ + assert(ref); + assert(ref->owner == region->parent); + region->param.file.fdref = ref; + ref->refcount++; +} + +void fdref_deref(struct vir_region *region) +{ + struct fdref *ref = region->param.file.fdref; + int fd; + + assert(ref); + assert(ref->refcount > 0); + assert(ref->owner == region->parent); + + fd = ref->fd; + region->param.file.fdref = NULL; + ref->refcount--; + assert(ref->refcount >= 0); + if(ref->refcount > 0) return; + + if(region->parent->fdrefs == ref) region->parent->fdrefs = ref->next; + else { + struct fdref *r; + for(r = region->parent->fdrefs; r->next != ref; r = r->next) + ; + assert(r); + assert(r->next == ref); + r->next = ref->next; + } + + SLABFREE(ref); + ref = NULL; + + if(region->parent->vm_flags & (VMF_EXITING|VMF_EXECING)) { + return; + } + + /* If the last reference has disappeared, free the + * ref object and asynchronously close the fd in VFS. + * + * We don't need a callback as a close failing, although + * unexpected, isn't a problem and can't be handled. VFS + * will print a diagnostic. + */ +#if 1 + if(vfs_request(VMVFSREQ_FDCLOSE, fd, region->parent, + 0, 0, NULL, NULL, NULL, 0) != OK) { + panic("fdref_deref: could not send close request"); + } +#endif +} + +struct fdref *fdref_dedup_or_new(struct vmproc *owner, + ino_t ino, dev_t dev, int fd) +{ + struct fdref *fr; + + for(fr = owner->fdrefs; fr; fr = fr->next) { + assert(fr->owner == owner); + if(fr->fd == fd) { + assert(ino == fr->ino); + assert(dev == fr->dev); + return fr; + } + if(ino == fr->ino && dev == fr->dev) { + assert(fd != fr->fd); +#if 1 + if(vfs_request(VMVFSREQ_FDCLOSE, fd, owner, + 0, 0, NULL, NULL, NULL, 0) != OK) { + printf("fdref_dedup_or_new: could not close\n"); + } +#endif + return fr; + } + } + + return fdref_new(owner, ino, dev, fd); +} + diff --git a/servers/vm/fdref.h b/servers/vm/fdref.h new file mode 100644 index 000000000..ce919ffc8 --- /dev/null +++ b/servers/vm/fdref.h @@ -0,0 +1,29 @@ + +#ifndef _FDREF_H +#define _FDREF_H 1 + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +struct fdref { + int fd; + int refcount; + dev_t dev; + ino_t ino; + struct vmproc *owner; + struct fdref *next; +} *fdref; + +#endif + diff --git a/servers/vm/fork.c b/servers/vm/fork.c index 877ec7f7b..16c71be65 100644 --- a/servers/vm/fork.c +++ b/servers/vm/fork.c @@ -60,6 +60,7 @@ int do_fork(message *msg) /* The child is basically a copy of the parent. */ origpt = vmc->vm_pt; *vmc = *vmp; + vmc->fdrefs = NULL; vmc->vm_slot = childproc; region_init(&vmc->vm_regions_avl); vmc->vm_endpoint = NONE; /* In case someone tries to use it. */ diff --git a/servers/vm/main.c b/servers/vm/main.c index caf148de5..baa4ff4f0 100644 --- a/servers/vm/main.c +++ b/servers/vm/main.c @@ -414,6 +414,10 @@ void init_vm(void) CALLMAP(VM_WILLEXIT, do_willexit); CALLMAP(VM_NOTIFY_SIG, do_notify_sig); + /* Calls from VFS. */ + CALLMAP(VM_VFS_REPLY, do_vfs_reply); + CALLMAP(VM_VFS_MMAP, do_vfs_mmap); + /* Calls from RS */ CALLMAP(VM_RS_SET_PRIV, do_rs_set_priv); CALLMAP(VM_RS_UPDATE, do_rs_update); diff --git a/servers/vm/mem_cache.c b/servers/vm/mem_cache.c index b5fd758c8..7bc0bc24e 100644 --- a/servers/vm/mem_cache.c +++ b/servers/vm/mem_cache.c @@ -118,7 +118,6 @@ do_mapcache(message *msg) printf("VM: map_pf failed\n"); return ENOMEM; } - assert(!vr->param.pb_cache); } diff --git a/servers/vm/mem_file.c b/servers/vm/mem_file.c new file mode 100644 index 000000000..3b7abbb70 --- /dev/null +++ b/servers/vm/mem_file.c @@ -0,0 +1,248 @@ + +/* This file implements the methods of memory-mapped files. */ + +#include + +#include "proto.h" +#include "vm.h" +#include "region.h" +#include "glo.h" +#include "cache.h" + +/* These functions are static so as to not pollute the + * global namespace, and are accessed through their function + * pointers. + */ + +static void mappedfile_split(struct vmproc *vmp, struct vir_region *vr, + struct vir_region *r1, struct vir_region *r2); +static int mappedfile_unreference(struct phys_region *pr); +static int mappedfile_pagefault(struct vmproc *vmp, struct vir_region *region, + struct phys_region *ph, int write, vfs_callback_t callback, void *, int); +static int mappedfile_sanitycheck(struct phys_region *pr, char *file, int line); +static int mappedfile_writable(struct phys_region *pr); +static int mappedfile_copy(struct vir_region *vr, struct vir_region *newvr); +static int mappedfile_lowshrink(struct vir_region *vr, vir_bytes len); +static void mappedfile_delete(struct vir_region *region); + +struct mem_type mem_type_mappedfile = { + .name = "file-mapped memory", + .ev_unreference = mappedfile_unreference, + .ev_pagefault = mappedfile_pagefault, + .ev_sanitycheck = mappedfile_sanitycheck, + .ev_copy = mappedfile_copy, + .writable = mappedfile_writable, + .ev_split = mappedfile_split, + .ev_lowshrink = mappedfile_lowshrink, + .ev_delete = mappedfile_delete, +}; + +static int mappedfile_unreference(struct phys_region *pr) +{ + assert(pr->ph->refcount == 0); + if(pr->ph->phys != MAP_NONE) + free_mem(ABS2CLICK(pr->ph->phys), 1); + return OK; +} + +static int cow_block(struct vmproc *vmp, struct vir_region *region, + struct phys_region *ph, u16_t clearend) +{ + int r; + + if((r=mem_cow(region, ph, MAP_NONE, MAP_NONE)) != OK) { + printf("mappedfile_pagefault: COW failed\n"); + return r; + } + + /* After COW we are a normal piece of anonymous memory. */ + ph->memtype = &mem_type_anon; + + if(clearend) { + phys_bytes phaddr = ph->ph->phys, po = VM_PAGE_SIZE-clearend; + assert(clearend < VM_PAGE_SIZE); + phaddr += po; + if(sys_memset(NONE, 0, phaddr, clearend) != OK) { + panic("cow_block: clearend failed\n"); + } + } + + return OK; +} + +static int mappedfile_pagefault(struct vmproc *vmp, struct vir_region *region, + struct phys_region *ph, int write, vfs_callback_t cb, + void *state, int statelen) +{ + u32_t allocflags; + int procfd = region->param.file.fdref->fd; + + allocflags = vrallocflags(region->flags); + + assert(ph->ph->refcount > 0); + assert(region->param.file.inited); + assert(region->param.file.fdref); + assert(region->param.file.fdref->dev != NO_DEV); + + /* Totally new block? Create it. */ + if(ph->ph->phys == MAP_NONE) { + struct cached_page *cp; + u64_t referenced_offset = + region->param.file.offset + ph->offset; + if(region->param.file.fdref->ino == VMC_NO_INODE) { + cp = find_cached_page_bydev(region->param.file.fdref->dev, + referenced_offset, VMC_NO_INODE, 0, 1); + } else { + cp = find_cached_page_byino(region->param.file.fdref->dev, + region->param.file.fdref->ino, referenced_offset, 1); + } + if(cp) { + int result = OK; + pb_unreferenced(region, ph, 0); + pb_link(ph, cp->page, ph->offset, region); + + if(roundup(ph->offset+region->param.file.clearend, + VM_PAGE_SIZE) >= region->length) { + result = cow_block(vmp, region, ph, + region->param.file.clearend); + } else if(result == OK && write) { + result = cow_block(vmp, region, ph, 0); + } + + return result; + } + + if(!cb) { + printf("VM: mem_file: no callback, returning EFAULT\n"); + sys_sysctl_stacktrace(vmp->vm_endpoint); + return EFAULT; + } + + if(vfs_request(VMVFSREQ_FDIO, procfd, vmp, referenced_offset, + VM_PAGE_SIZE, cb, NULL, state, statelen) != OK) { + printf("VM: mappedfile_pagefault: vfs_request failed\n"); + return ENOMEM; + } + + return SUSPEND; + } + + if(!write) { + printf("mappedfile_pagefault: nonwrite fault?\n"); + return EFAULT; + } + + return cow_block(vmp, region, ph, 0); +} + +static int mappedfile_sanitycheck(struct phys_region *pr, char *file, int line) +{ + MYASSERT(usedpages_add(pr->ph->phys, VM_PAGE_SIZE) == OK); + return OK; +} + +static int mappedfile_writable(struct phys_region *pr) +{ + /* We are never writable. */ + return 0; +} + +int mappedfile_copy(struct vir_region *vr, struct vir_region *newvr) +{ + assert(vr->param.file.inited); + mappedfile_setfile(newvr->parent, newvr, vr->param.file.fdref->fd, + vr->param.file.offset, + vr->param.file.fdref->dev, vr->param.file.fdref->ino, + vr->param.file.clearend, 0); + assert(newvr->param.file.inited); + + return OK; +} + +int mappedfile_setfile(struct vmproc *owner, + struct vir_region *region, int fd, u64_t offset, + dev_t dev, ino_t ino, u16_t clearend, int prefill) +{ + vir_bytes vaddr; + struct fdref *newref = fdref_dedup_or_new(owner, ino, dev, fd); + + assert(newref); + assert(!region->param.file.inited); + assert(dev != NO_DEV); + fdref_ref(newref, region); + region->param.file.offset = offset; + region->param.file.clearend = clearend; + region->param.file.inited = 1; + + if(!prefill) return OK; + + for(vaddr = 0; vaddr < region->length; vaddr+=VM_PAGE_SIZE) { + struct cached_page *cp = NULL; + struct phys_region *pr; + u64_t referenced_offset = offset + vaddr; + + if(roundup(vaddr+region->param.file.clearend, + VM_PAGE_SIZE) >= region->length) { + break; + } + + if(ino == VMC_NO_INODE) { + cp = find_cached_page_bydev(dev, referenced_offset, + VMC_NO_INODE, 0, 1); + } else { + cp = find_cached_page_byino(dev, ino, + referenced_offset, 1); + } + if(!cp) continue; + if(!(pr = pb_reference(cp->page, vaddr, region, + &mem_type_mappedfile))) { + printf("mappedfile_setfile: pb_reference failed\n"); + break; + } + if(map_ph_writept(region->parent, region, pr) != OK) { + printf("mappedfile_setfile: map_ph_writept failed\n"); + break; + } + } + + return OK; +} + +static void mappedfile_split(struct vmproc *vmp, struct vir_region *vr, + struct vir_region *r1, struct vir_region *r2) +{ + assert(!r1->param.file.inited); + assert(!r2->param.file.inited); + assert(vr->param.file.inited); + assert(r1->length + r2->length == vr->length); + assert(vr->def_memtype == &mem_type_mappedfile); + assert(r1->def_memtype == &mem_type_mappedfile); + assert(r2->def_memtype == &mem_type_mappedfile); + + r1->param.file = vr->param.file; + r2->param.file = vr->param.file; + + fdref_ref(vr->param.file.fdref, r1); + fdref_ref(vr->param.file.fdref, r2); + + r1->param.file.clearend = 0; + r2->param.file.offset += r1->length; + + assert(r1->param.file.inited); + assert(r2->param.file.inited); +} + +static int mappedfile_lowshrink(struct vir_region *vr, vir_bytes len) +{ + assert(vr->param.file.inited); + vr->param.file.offset += len; + return OK; +} + +static void mappedfile_delete(struct vir_region *region) +{ + assert(region->def_memtype == &mem_type_mappedfile); + assert(region->param.file.inited); + assert(region->param.file.fdref); + fdref_deref(region); +} diff --git a/servers/vm/mmap.c b/servers/vm/mmap.c index 4340f94b5..c8b68a25a 100644 --- a/servers/vm/mmap.c +++ b/servers/vm/mmap.c @@ -81,6 +81,124 @@ static struct vir_region *mmap_region(struct vmproc *vmp, vir_bytes addr, return vr; } +static int mmap_file(struct vmproc *vmp, + int vmfd, u32_t off_lo, u32_t off_hi, int flags, + ino_t ino, dev_t dev, u64_t filesize, vir_bytes addr, vir_bytes len, + vir_bytes *retaddr, u16_t clearend, int writable) +{ +/* VFS has replied to a VMVFSREQ_FDLOOKUP request. */ + struct vir_region *vr; + u64_t file_offset, page_offset; + int result = OK; + u32_t vrflags = 0; + + if(writable) vrflags |= VR_WRITABLE; + + if(flags & MAP_THIRDPARTY) { + file_offset = off_lo; + } else { + file_offset = make64(off_lo, off_hi); + if(off_hi && !off_lo) { + /* XXX clang compatability hack */ + off_hi = file_offset = 0; + } + } + + /* Do some page alignments. */ + if((page_offset = (file_offset % VM_PAGE_SIZE))) { + file_offset -= page_offset; + len += page_offset; + } + + len = roundup(len, VM_PAGE_SIZE); + + /* All numbers should be page-aligned now. */ + assert(!(len % VM_PAGE_SIZE)); + assert(!(filesize % VM_PAGE_SIZE)); + assert(!(file_offset % VM_PAGE_SIZE)); + +#if 0 + /* XXX ld.so relies on longer-than-file mapping */ + if((u64_t) len + file_offset > filesize) { + printf("VM: truncating mmap dev 0x%x ino %d beyond file size in %d; offset %llu, len %lu, size %llu; ", + dev, ino, vmp->vm_endpoint, + file_offset, len, filesize); + len = filesize - file_offset; + return EINVAL; + } +#endif + + if(!(vr = mmap_region(vmp, addr, flags, len, + vrflags, &mem_type_mappedfile, 0))) { + result = ENOMEM; + } else { + *retaddr = vr->vaddr + page_offset; + result = OK; + + mappedfile_setfile(vmp, vr, vmfd, + file_offset, dev, ino, clearend, 1); + } + + return result; +} + +int do_vfs_mmap(message *m) +{ + vir_bytes v; + struct vmproc *vmp; + int r, n; + u16_t clearend, flags = 0; + + clearend = (m->m_u.m_vm_vfs.clearend_and_flags & MVM_LENMASK); + flags = (m->m_u.m_vm_vfs.clearend_and_flags & MVM_FLAGSMASK); + + if((r=vm_isokendpt(m->m_u.m_vm_vfs.who, &n)) != OK) + panic("bad ep %d from vfs", m->m_u.m_vm_vfs.who); + vmp = &vmproc[n]; + + return mmap_file(vmp, m->m_u.m_vm_vfs.fd, m->m_u.m_vm_vfs.offset, 0, + MAP_PRIVATE | MAP_FIXED, + m->m_u.m_vm_vfs.ino, m->m_u.m_vm_vfs.dev, + (u64_t) LONG_MAX * VM_PAGE_SIZE, + m->m_u.m_vm_vfs.vaddr, m->m_u.m_vm_vfs.len, &v, + clearend, flags); +} + +static void mmap_file_cont(struct vmproc *vmp, message *replymsg, void *cbarg, + void *origmsg_v) +{ + message *origmsg = (message *) origmsg_v; + message mmap_reply; + int result; + int writable = 0; + vir_bytes v = (vir_bytes) MAP_FAILED; + + if(origmsg->VMM_PROT & PROT_WRITE) + writable = 1; + + if(replymsg->VMV_RESULT != OK) { + printf("VM: VFS reply failed\n"); + sys_sysctl_stacktrace(vmp->vm_endpoint); + result = origmsg->VMV_RESULT; + } else { + /* Finish mmap */ + result = mmap_file(vmp, replymsg->VMV_FD, origmsg->VMM_OFFSET_LO, + origmsg->VMM_OFFSET_HI, origmsg->VMM_FLAGS, + replymsg->VMV_INO, replymsg->VMV_DEV, + (u64_t) replymsg->VMV_SIZE_PAGES*PAGE_SIZE, + origmsg->VMM_ADDR, + origmsg->VMM_LEN, &v, 0, writable); + } + + /* Unblock requesting process. */ + memset(&mmap_reply, 0, sizeof(mmap_reply)); + mmap_reply.m_type = result; + mmap_reply.VMM_ADDR = v; + + if(send(vmp->vm_endpoint, &mmap_reply) != OK) + panic("VM: mmap_file_cont: send() failed"); +} + /*===========================================================================* * do_mmap * *===========================================================================*/ @@ -111,11 +229,16 @@ int do_mmap(message *m) vmp = &vmproc[n]; + /* "SUSv3 specifies that mmap() should fail if length is 0" */ + if(len <= 0) { + return EINVAL; + } + if(m->VMM_FD == -1 || (m->VMM_FLAGS & MAP_ANON)) { /* actual memory in some form */ mem_type_t *mt = NULL; - if(m->VMM_FD != -1 || len <= 0) { + if(m->VMM_FD != -1) { printf("VM: mmap: fd %d, len 0x%x\n", m->VMM_FD, len); return EINVAL; } @@ -134,7 +257,20 @@ int do_mmap(message *m) return ENOMEM; } } else { - return ENXIO; + /* files get private copies of pages on writes. */ + if(!(m->VMM_FLAGS & MAP_PRIVATE)) { + printf("VM: mmap file must MAP_PRIVATE\n"); + return ENXIO; + } + + if(vfs_request(VMVFSREQ_FDLOOKUP, m->VMM_FD, vmp, 0, 0, + mmap_file_cont, NULL, m, sizeof(*m)) != OK) { + printf("VM: vfs_request for mmap failed\n"); + return ENXIO; + } + + /* request queued; don't reply. */ + return SUSPEND; } /* Return mapping, as seen from process. */ diff --git a/servers/vm/proto.h b/servers/vm/proto.h index 2d5018a50..9fd619293 100644 --- a/servers/vm/proto.h +++ b/servers/vm/proto.h @@ -228,5 +228,12 @@ int vfs_request(int reqno, int fd, struct vmproc *vmp, u64_t offset, int do_vfs_reply(message *m); /* mem_file.c */ -void mappedfile_setfile(struct vir_region *region, int fd, u64_t offset, +int mappedfile_setfile(struct vmproc *owner, struct vir_region *region, + int fd, u64_t offset, dev_t dev, ino_t ino, u16_t clearend, int prefill); + +/* fdref.c */ +struct fdref *fdref_new(struct vmproc *owner, ino_t ino, dev_t dev, int fd); +struct fdref *fdref_dedup_or_new(struct vmproc *owner, ino_t ino, dev_t dev, int fd); +void fdref_ref(struct fdref *ref, struct vir_region *region); +void fdref_deref(struct vir_region *region); diff --git a/servers/vm/region.c b/servers/vm/region.c index 53611931d..c4bd19453 100644 --- a/servers/vm/region.c +++ b/servers/vm/region.c @@ -833,6 +833,8 @@ struct vir_region *map_copy_region(struct vmproc *vmp, struct vir_region *vr) if(!(newvr = region_new(vr->parent, vr->vaddr, vr->length, vr->flags, vr->def_memtype))) return NULL; + USE(newvr, newvr->parent = vmp;); + if(vr->def_memtype->ev_copy && (r=vr->def_memtype->ev_copy(vr, newvr)) != OK) { map_free(newvr); printf("VM: memtype-specific copy failed (%d)\n", r); @@ -980,7 +982,6 @@ struct vir_region *start_src_vr; map_free_proc(dst); return ENOMEM; } - USE(newvr, newvr->parent = dst;); region_insert(&dst->vm_regions_avl, newvr); assert(vr->length == newvr->length); diff --git a/servers/vm/region.h b/servers/vm/region.h index 9d615ef03..070c0d467 100644 --- a/servers/vm/region.h +++ b/servers/vm/region.h @@ -19,6 +19,7 @@ #include "phys_region.h" #include "memtype.h" #include "vm.h" +#include "fdref.h" struct phys_block { #if SANITYCHECKS @@ -53,11 +54,9 @@ typedef struct vir_region { } shared; struct phys_block *pb_cache; struct { - int procfd; /* cloned fd in proc for mmap */ - dev_t dev; - ino_t ino; - u64_t offset; int inited; + struct fdref *fdref; + u64_t offset; u16_t clearend; } file; } param; diff --git a/servers/vm/vfs.c b/servers/vm/vfs.c new file mode 100644 index 000000000..274c2afb6 --- /dev/null +++ b/servers/vm/vfs.c @@ -0,0 +1,144 @@ + +/* Sending requests to VFS and handling the replies. */ + +#define _SYSTEM 1 + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "proto.h" +#include "glo.h" +#include "util.h" +#include "region.h" +#include "sanitycheck.h" + +#define STATELEN 50 + +static struct vfs_request_node { + message reqmsg; + char reqstate[STATELEN]; + void *opaque; + endpoint_t who; + u32_t req_id; + vfs_callback_t callback; + struct vfs_request_node *next; +} *first_queued, *active; + +static void activate(void) +{ + assert(!active); + assert(first_queued); + + active = first_queued; + first_queued = first_queued->next; + + if(asynsend3(VFS_PROC_NR, &active->reqmsg, AMF_NOREPLY) != OK) + panic("VM: asynsend to VFS failed"); +} + +/*===========================================================================* + * vfs_request * + *===========================================================================*/ +int vfs_request(int reqno, int fd, struct vmproc *vmp, u64_t offset, u32_t len, + vfs_callback_t reply_callback, void *cbarg, void *state, int statelen) +{ +/* Perform an asynchronous request to VFS. + * We send a message of type VFS_VMCALL to VFS. VFS will respond + * with message type VM_VFS_REPLY. We send the request asynchronously + * and then handle the reply as it if were a VM_VFS_REPLY request. + */ + message *m; + static u32_t reqid = 0; + struct vfs_request_node *reqnode; + + reqid++; + + assert(statelen <= STATELEN); + + if(!SLABALLOC(reqnode)) { + printf("vfs_request: no memory for request node\n"); + return ENOMEM; + } + + m = &reqnode->reqmsg; + m->m_type = VFS_VMCALL; + m->VFS_VMCALL_REQ = reqno; + m->VFS_VMCALL_FD = fd; + m->VFS_VMCALL_REQID = reqid; + m->VFS_VMCALL_ENDPOINT = vmp->vm_endpoint; + m->VFS_VMCALL_OFFSET_LO = ex64lo(offset); + m->VFS_VMCALL_OFFSET_HI = ex64hi(offset); + m->VFS_VMCALL_LENGTH = len; + + reqnode->who = vmp->vm_endpoint; + reqnode->req_id = reqid; + reqnode->next = first_queued; + reqnode->callback = reply_callback; + reqnode->opaque = cbarg; + if(state) memcpy(reqnode->reqstate, state, statelen); + first_queued = reqnode; + + /* Send the request message if none pending. */ + if(!active) + activate(); + + return OK; +} + +/*===========================================================================* + * do_vfs_reply * + *===========================================================================*/ +int do_vfs_reply(message *m) +{ +/* VFS has handled a VM request and VFS has replied. It must be the + * active request. + */ + struct vfs_request_node *orignode = active; + vfs_callback_t req_callback; + void *cbarg; + int n; + struct vmproc *vmp; + if(m->m_source != VFS_PROC_NR) + return ENOSYS; + + assert(active); + assert(active->req_id == m->VMV_REQID); + + /* the endpoint may have exited */ + if(vm_isokendpt(m->VMV_ENDPOINT, &n) != OK) + vmp = NULL; + else vmp = &vmproc[n]; + + req_callback = active->callback; + cbarg = active->opaque; + active = NULL; + + /* Invoke requested reply-callback within VM. */ + if(req_callback) req_callback(vmp, m, cbarg, orignode->reqstate); + + SLABFREE(orignode); + + /* Send the next request message if any. */ + if(first_queued) + activate(); + + return SUSPEND; /* don't reply to the reply */ +} + diff --git a/servers/vm/vmproc.h b/servers/vm/vmproc.h index 5e7145505..bfa671cab 100644 --- a/servers/vm/vmproc.h +++ b/servers/vm/vmproc.h @@ -22,6 +22,7 @@ struct vmproc { vir_bytes vm_region_top; /* highest vaddr last inserted */ bitchunk_t vm_call_mask[VM_CALL_MASK_SIZE]; int vm_slot; /* process table slot */ + struct fdref *fdrefs; #if VMSTATS int vm_bytecopies; #endif @@ -31,5 +32,6 @@ struct vmproc { #define VMF_INUSE 0x001 /* slot contains a process */ #define VMF_EXITING 0x002 /* PM is cleaning up this process */ #define VMF_WATCHEXIT 0x008 /* Store in queryexit table */ +#define VMF_EXECING 0x010 /* exec() in progress */ #endif