From 0ffae85825e7e0bd88e0c6b6908e192ffaa10d11 Mon Sep 17 00:00:00 2001 From: Ben Gras Date: Tue, 7 May 2013 12:41:07 +0000 Subject: [PATCH] pm, vfs, includes: mmap support Change-Id: I831551e45a6781c74313c450eb9c967a68505932 --- commands/service/parse.c | 1 + etc/system.conf | 3 +- include/minix/callnr.h | 4 +- include/minix/com.h | 9 +- include/minix/vm.h | 8 ++ lib/libc/sys-minix/mmap.c | 28 +++++++ servers/is/inc.h | 2 + servers/pm/getset.c | 2 +- servers/pm/table.c | 1 + servers/procfs/inc.h | 2 + servers/vfs/exec.c | 59 +++++++++++++- servers/vfs/fdset.h | 11 +++ servers/vfs/filedes.c | 10 ++- servers/vfs/fproc.h | 3 + servers/vfs/fs.h | 3 + servers/vfs/main.c | 5 +- servers/vfs/misc.c | 167 +++++++++++++++++++++++++++++++++++++- servers/vfs/open.c | 104 ++++++++++++++++++------ servers/vfs/proto.h | 5 +- servers/vfs/read.c | 60 +++++++++----- servers/vfs/table.c | 1 + servers/vfs/vnode.h | 1 + servers/vfs/worker.c | 4 +- sys/sys/mman.h | 1 + 24 files changed, 432 insertions(+), 62 deletions(-) create mode 100644 servers/vfs/fdset.h diff --git a/commands/service/parse.c b/commands/service/parse.c index ea805a073..2808da172 100644 --- a/commands/service/parse.c +++ b/commands/service/parse.c @@ -734,6 +734,7 @@ struct { "PROCCTL", VM_PROCCTL }, { "MAPCACHEPAGE", VM_MAPCACHEPAGE }, { "SETCACHEPAGE", VM_SETCACHEPAGE }, + { "VFS_MMAP", VM_VFS_MMAP }, { NULL, 0 }, }; diff --git a/etc/system.conf b/etc/system.conf index 496888114..6050d210d 100644 --- a/etc/system.conf +++ b/etc/system.conf @@ -94,7 +94,8 @@ service vfs VIRCOPY # 15 MEMSET ; - vm PROCCTL; + vm PROCCTL + VFS_MMAP; io NONE; # No I/O range allowed irq NONE; # No IRQ allowed sigmgr rs; # Signal manager is RS diff --git a/include/minix/callnr.h b/include/minix/callnr.h index e38d881d8..0ec74931b 100644 --- a/include/minix/callnr.h +++ b/include/minix/callnr.h @@ -1,4 +1,4 @@ -#define NCALLS 117 /* number of system calls allowed */ +#define NCALLS 118 /* number of system calls allowed */ /* In case it isn't obvious enough: this list is sorted numerically. */ #define EXIT 1 @@ -107,6 +107,8 @@ #define CLOCK_GETTIME 115 /* clock_gettime() */ #define CLOCK_SETTIME 116 /* clock_settime() */ +#define VFS_VMCALL 117 + #define TASK_REPLY 121 /* to VFS: reply code from drivers, not * really a standalone call. */ diff --git a/include/minix/com.h b/include/minix/com.h index 8af307b9f..08496d220 100644 --- a/include/minix/com.h +++ b/include/minix/com.h @@ -1005,6 +1005,7 @@ # define VMV_DEV m10_i4 # define VMV_INO m10_l1 # define VMV_FD m10_l2 +# define VMV_SIZE_PAGES m10_l3 #define VM_REMAP (VM_RQ_BASE+33) # define VMRE_D m1_i1 @@ -1075,8 +1076,10 @@ #define VMPPARAM_CLEAR 1 /* values for VMPCTL_PARAM */ +#define VM_VFS_MMAP (VM_RQ_BASE+46) + /* Total. */ -#define NR_VM_CALLS 46 +#define NR_VM_CALLS 47 #define VM_CALL_MASK_SIZE BITMAP_CHUNKS(NR_VM_CALLS) /* not handled as a normal VM call, thus at the end of the reserved rage */ @@ -1086,8 +1089,8 @@ /* Basic vm calls allowed to every process. */ #define VM_BASIC_CALLS \ - VM_MMAP, VM_MUNMAP, VM_MAP_PHYS, VM_UNMAP_PHYS, \ - VM_INFO, VM_MAPCACHEPAGE + VM_MMAP, VM_VFS_REPLY, VM_MUNMAP, VM_MAP_PHYS, VM_UNMAP_PHYS, \ + VM_INFO /*===========================================================================* * Messages for IPC server * diff --git a/include/minix/vm.h b/include/minix/vm.h index 698905954..6ed245e08 100644 --- a/include/minix/vm.h +++ b/include/minix/vm.h @@ -26,6 +26,14 @@ int vm_forgetblock(u64_t id); void vm_forgetblocks(void); int vm_yield_block_get_block(u64_t yieldid, u64_t getid, void *mem, vir_bytes len); +int minix_vfs_mmap(endpoint_t who, u32_t offset, u32_t len, + u32_t dev, u32_t ino, u16_t fd, u32_t vaddr, u16_t clearend, u16_t + flags); + +/* minix vfs mmap flags */ +#define MVM_LENMASK 0x0FFF +#define MVM_FLAGSMASK 0xF000 +#define MVM_WRITABLE 0x8000 /* Invalid ID with special meaning for the vm_yield_block_get_block * interface. diff --git a/lib/libc/sys-minix/mmap.c b/lib/libc/sys-minix/mmap.c index 4a3de773e..f5d5915a7 100644 --- a/lib/libc/sys-minix/mmap.c +++ b/lib/libc/sys-minix/mmap.c @@ -14,6 +14,8 @@ __weak_alias(vm_getphys, _vm_getphys) __weak_alias(vm_getrefcount, _vm_getrefcount) __weak_alias(minix_mmap, _minix_mmap) __weak_alias(minix_munmap, _minix_munmap) +__weak_alias(mmap, _minix_mmap) +__weak_alias(munmap, _minix_munmap) #endif @@ -51,12 +53,38 @@ void *minix_mmap_for(endpoint_t forwhom, return (void *) m.VMM_RETADDR; } +int minix_vfs_mmap(endpoint_t who, u32_t offset, u32_t len, + u32_t dev, u32_t ino, u16_t fd, u32_t vaddr, u16_t clearend, + u16_t flags) +{ + message m; + + memset(&m, 0, sizeof(message)); + + m.m_u.m_vm_vfs.who = who; + m.m_u.m_vm_vfs.offset = offset; + m.m_u.m_vm_vfs.dev = dev; + m.m_u.m_vm_vfs.ino = ino; + m.m_u.m_vm_vfs.vaddr = vaddr; + m.m_u.m_vm_vfs.len = len; + m.m_u.m_vm_vfs.fd = fd; + m.m_u.m_vm_vfs.clearend_and_flags = clearend | flags; + + return _syscall(VM_PROC_NR, VM_VFS_MMAP, &m); +} + void *minix_mmap(void *addr, size_t len, int prot, int flags, int fd, off_t offset) { return minix_mmap_for(SELF, addr, len, prot, flags, fd, offset); } +void *minix_mmap64(void *addr, size_t len, int prot, int flags, + int fd, u64_t offset) +{ + return minix_mmap_for(SELF, addr, len, prot, flags, fd, offset); +} + int minix_munmap(void *addr, size_t len) { message m; diff --git a/servers/is/inc.h b/servers/is/inc.h index fe5025e67..5186b50a3 100644 --- a/servers/is/inc.h +++ b/servers/is/inc.h @@ -6,6 +6,8 @@ #define _SYSTEM 1 /* get OK and negative error codes */ +#include "../vfs/fdset.h" + #include #include #include diff --git a/servers/pm/getset.c b/servers/pm/getset.c index 407a5a63a..5303841dd 100644 --- a/servers/pm/getset.c +++ b/servers/pm/getset.c @@ -23,7 +23,7 @@ int do_get() */ register struct mproc *rmp = mp; - int r, i; + int r; int ngroups; switch(call_nr) { diff --git a/servers/pm/table.c b/servers/pm/table.c index 258ed1501..d1c2075a1 100644 --- a/servers/pm/table.c +++ b/servers/pm/table.c @@ -128,6 +128,7 @@ int (*call_vec[])(void) = { do_getres, /* 114 = clock_getres */ do_gettime, /* 115 = clock_gettime */ do_settime, /* 116 = clock_settime */ + no_sys, /* 117 = (vmcall) */ }; /* This should not fail with "array size is negative": */ extern int dummy[sizeof(call_vec) == NCALLS * sizeof(call_vec[0]) ? 1 : -1]; diff --git a/servers/procfs/inc.h b/servers/procfs/inc.h index ae411730f..5316847c3 100644 --- a/servers/procfs/inc.h +++ b/servers/procfs/inc.h @@ -3,6 +3,8 @@ #define _SYSTEM 1 +#include "vfs/fdset.h" + #include #include #include diff --git a/servers/vfs/exec.c b/servers/vfs/exec.c index ce00e56c8..f21ffb18c 100644 --- a/servers/vfs/exec.c +++ b/servers/vfs/exec.c @@ -16,6 +16,7 @@ #include "fs.h" #include +#include #include #include #include @@ -30,6 +31,7 @@ #include "path.h" #include "param.h" #include "vnode.h" +#include "file.h" #include #include #include @@ -48,6 +50,8 @@ struct vfs_exec_info { int is_dyn; /* Dynamically linked executable */ int elf_main_fd; /* Dyn: FD of main program execuatble */ char execname[PATH_MAX]; /* Full executable invocation */ + int procfd; + int procfd_used; }; static void lock_exec(void); @@ -185,6 +189,27 @@ static int get_read_vp(struct vfs_exec_info *execi, r=get_read_vp(&e,f,p,s,rs,fp); if(r != OK) { FAILCHECK(r); } \ } while(0) +static int vfs_memmap(struct exec_info *execi, + vir_bytes vaddr, vir_bytes len, vir_bytes foffset, u16_t clearend, + int protflags) +{ + struct vfs_exec_info *vi = (struct vfs_exec_info *) execi->opaque; + struct vnode *vp = ((struct vfs_exec_info *) execi->opaque)->vp; + int r; + u16_t flags = 0; + + if(protflags & PROT_WRITE) + flags |= MVM_WRITABLE; + + r = minix_vfs_mmap(execi->proc_e, foffset, len, + vp->v_dev, vp->v_inode_nr, vi->procfd, vaddr, clearend, flags); + if(r == OK) { + vi->procfd_used = 1; + } + + return r; +} + /*===========================================================================* * pm_exec * *===========================================================================*/ @@ -205,14 +230,18 @@ int pm_exec(endpoint_t proc_e, vir_bytes path, size_t path_len, int i; static char fullpath[PATH_MAX], elf_interpreter[PATH_MAX], + firstexec[PATH_MAX], finalexec[PATH_MAX]; struct lookup resolve; stackhook_t makestack = NULL; + static int n; + n++; lock_exec(); /* unset execi values are 0. */ memset(&execi, 0, sizeof(execi)); + execi.procfd = -1; /* passed from exec() libc code */ execi.userflags = user_exec_flags; @@ -223,6 +252,7 @@ int pm_exec(endpoint_t proc_e, vir_bytes path, size_t path_len, rfp = fp = &fproc[slot]; lookup_init(&resolve, fullpath, PATH_NOFLAGS, &execi.vmp, &execi.vp); + resolve.l_vmnt_lock = VMNT_READ; resolve.l_vnode_lock = VNODE_READ; @@ -244,6 +274,7 @@ int pm_exec(endpoint_t proc_e, vir_bytes path, size_t path_len, /* Get the exec file name. */ FAILCHECK(fetch_name(path, path_len, fullpath)); strlcpy(finalexec, fullpath, PATH_MAX); + strlcpy(firstexec, fullpath, PATH_MAX); /* Get_read_vp will return an opened vn in execi. * if necessary it releases the existing vp so we can @@ -264,6 +295,7 @@ int pm_exec(endpoint_t proc_e, vir_bytes path, size_t path_len, FAILCHECK(fetch_name(path, path_len, fullpath)); FAILCHECK(patch_stack(execi.vp, mbuf, &frame_len, fullpath)); strlcpy(finalexec, fullpath, PATH_MAX); + strlcpy(firstexec, fullpath, PATH_MAX); Get_read_vp(execi, fullpath, 1, 0, &resolve, fp); } @@ -299,9 +331,28 @@ int pm_exec(endpoint_t proc_e, vir_bytes path, size_t path_len, * be looked up */ strlcpy(fullpath, elf_interpreter, PATH_MAX); + strlcpy(firstexec, elf_interpreter, PATH_MAX); Get_read_vp(execi, fullpath, 0, 0, &resolve, fp); } + /* We also want an FD so VM can mmap() the process in if possible. */ + { + int openr; + fp->fp_fdscan = OPEN_MAX/2; + if((openr=execi.procfd=common_open(firstexec, O_RDONLY, 0)) < 0) { + printf("vfs: exec: open failed of %s (%d), can't mmap\n", + firstexec, openr); + execi.procfd = -1; + } else { + if(fp->fp_filp[openr]->filp_vno->v_vmnt->m_haspeek && + major(fp->fp_filp[openr]->filp_vno->v_dev) != MEMORY_MAJOR) { + FD_SET(openr, &fp->fp_filp_system); /* systemfd */ + execi.args.memmap = vfs_memmap; + } + } + fp->fp_fdscan = 0; + } + /* callback functions and data */ execi.args.copymem = read_seg; execi.args.clearproc = libexec_clearproc_vm_procctl; @@ -358,7 +409,13 @@ pm_execfinal: unlock_vnode(execi.vp); put_vnode(execi.vp); } + if(execi.procfd >= 0 && !execi.procfd_used) { + int r; + r = close_fd(rfp, execi.procfd, 0); + } + unlock_exec(); + return(r); } @@ -651,7 +708,7 @@ static void clo_exec(struct fproc *rfp) /* Check the file desriptors one by one for presence of FD_CLOEXEC. */ for (i = 0; i < OPEN_MAX; i++) if ( FD_ISSET(i, &rfp->fp_cloexec_set)) - (void) close_fd(rfp, i); + (void) close_fd(rfp, i, 0); } /*===========================================================================* diff --git a/servers/vfs/fdset.h b/servers/vfs/fdset.h new file mode 100644 index 000000000..80c8f6384 --- /dev/null +++ b/servers/vfs/fdset.h @@ -0,0 +1,11 @@ +#ifndef _FDSET_H +#define _FDSET_H 1 + +#include "const.h" +#ifdef FD_SETSIZE +#error FD_SETSIZE already set +#endif +#define FD_SETSIZE FDS_PER_PROCESS +#include + +#endif diff --git a/servers/vfs/filedes.c b/servers/vfs/filedes.c index 1ccdb8eb9..6cac9fabe 100644 --- a/servers/vfs/filedes.c +++ b/servers/vfs/filedes.c @@ -203,7 +203,6 @@ int fild; /* file descriptor */ tll_access_t locktype; { /* See if 'fild' refers to a valid file descr. If so, return its filp ptr. */ - return get_filp2(fp, fild, locktype); } @@ -227,8 +226,15 @@ tll_access_t locktype; */ else if ((filp = rfp->fp_filp[fild]) == NULL) err_code = EBADF; - else + else { + if (FD_ISSET(fild, &fp->fp_filp_system)) { +#if 0 + printf("VFS: warning: doing get_filp of system fd %d: %d\n", + fp->fp_endpoint, fild); +#endif + } lock_filp(filp, locktype); /* All is fine */ + } return(filp); /* may also be NULL */ } diff --git a/servers/vfs/fproc.h b/servers/vfs/fproc.h index 9ef7acbd0..634296186 100644 --- a/servers/vfs/fproc.h +++ b/servers/vfs/fproc.h @@ -2,6 +2,7 @@ #define __VFS_FPROC_H__ #include "threads.h" +#include "fdset.h" #include #include @@ -23,6 +24,7 @@ EXTERN struct fproc { struct filp *fp_filp[OPEN_MAX];/* the file descriptor table */ fd_set fp_filp_inuse; /* which fd's are in use? */ fd_set fp_cloexec_set; /* bit map for POSIX Table 6-2 FD_CLOEXEC */ + fd_set fp_filp_system; /* system-reserved; user process can't touch it */ dev_t fp_tty; /* major/minor of controlling tty */ @@ -46,6 +48,7 @@ EXTERN struct fproc { struct job fp_job; /* pending job */ thread_t fp_wtid; /* Thread ID of worker */ char fp_name[PROC_NAME_LEN]; /* Last exec() */ + int fp_fdscan; /* Start scanning fd's here */ #if LOCK_DEBUG int fp_vp_rdlocks; /* number of read-only locks on vnodes */ int fp_vmnt_rdlocks; /* number of read-only locks on vmnts */ diff --git a/servers/vfs/fs.h b/servers/vfs/fs.h index 3c95e26d5..b628f69d0 100644 --- a/servers/vfs/fs.h +++ b/servers/vfs/fs.h @@ -6,6 +6,9 @@ */ #define _SYSTEM 1 /* tell headers that this is the kernel */ +/* Get the right-sized fd_set */ +#include "fdset.h" + /* The following are so basic, all the *.c files get them automatically. */ #include /* MUST be first */ #include diff --git a/servers/vfs/main.c b/servers/vfs/main.c index d18cb752f..437e794de 100644 --- a/servers/vfs/main.c +++ b/servers/vfs/main.c @@ -758,8 +758,9 @@ static void get_work() } proc_p = _ENDPOINT_P(m_in.m_source); - if (proc_p < 0 || proc_p >= NR_PROCS) fp = NULL; - else fp = &fproc[proc_p]; + if (proc_p < 0 || proc_p >= NR_PROCS) { + fp = NULL; + } else fp = &fproc[proc_p]; if (m_in.m_type == EDEADSRCDST) return; /* Failed 'sendrec' */ diff --git a/servers/vfs/misc.c b/servers/vfs/misc.c index 44d9d53b7..e11e08a6f 100644 --- a/servers/vfs/misc.c +++ b/servers/vfs/misc.c @@ -311,6 +311,160 @@ int do_fsync(message *UNUSED(m_out)) return(r); } +/*===========================================================================* + * do_vm_call * + *===========================================================================*/ +int do_vm_call(message *m_out) +{ +/* A call that VM does to VFS. + * We must reply with the fixed type VM_VFS_REPLY (and put our result info + * in the rest of the message) so VM can tell the difference between a + * request from VFS and a reply to this call. + */ + int req = job_m_in.VFS_VMCALL_REQ; + int req_fd = job_m_in.VFS_VMCALL_FD; + u32_t req_id = job_m_in.VFS_VMCALL_REQID; + endpoint_t ep = job_m_in.VFS_VMCALL_ENDPOINT; + u64_t offset = make64(job_m_in.VFS_VMCALL_OFFSET_LO, + job_m_in.VFS_VMCALL_OFFSET_HI); + u32_t length = job_m_in.VFS_VMCALL_LENGTH; + int result = OK; + int slot; + struct fproc *fdsrc, *vmf; + struct filp *f = NULL; + + if(job_m_in.m_source != VM_PROC_NR) + return ENOSYS; + + okendpt(ep, &slot); + + fdsrc = &fproc[slot]; + vmf = &fproc[VM_PROC_NR]; + assert(fp == vmf); + + /* We do everything on behalf of the target process. */ + fp = fdsrc; + + if(fp == vmf) { + printf("VFS: huh, performing request for VM?\n"); + } + + switch(req) { + case VMVFSREQ_FDLOOKUP: + { + int procfd; + + /* Lookup fd in referenced process. */ + if ((f = get_filp(req_fd, VNODE_READ)) == NULL) { + result = err_code; + goto reqdone; + } + + assert(f->filp_vno); + assert(f->filp_vno->v_vmnt); + + if(!f->filp_vno->v_vmnt->m_haspeek) { + result = EINVAL; + goto reqdone; + } + + if (!S_ISREG(f->filp_vno->v_mode) && + !S_ISBLK(f->filp_vno->v_mode)) { + printf("VFS: mmap regular/blockdev only; dev 0x%x ino %d has mode 0%o\n", + f->filp_vno->v_dev, f->filp_vno->v_inode_nr, f->filp_vno->v_mode); + result = EINVAL; + goto reqdone; + } + + if((result=get_fd(OPEN_MAX/2, 0, &procfd, NULL)) != OK) + goto reqdone; + + if(S_ISBLK(f->filp_vno->v_mode)) { + assert(f->filp_vno->v_sdev != NO_DEV); + m_out->VMV_DEV = f->filp_vno->v_sdev; + m_out->VMV_INO = VMC_NO_INODE; + m_out->VMV_SIZE_PAGES = LONG_MAX; + } else { + f->filp_vno->v_mmapped = 1; + m_out->VMV_DEV = f->filp_vno->v_dev; + m_out->VMV_INO = f->filp_vno->v_inode_nr; + m_out->VMV_SIZE_PAGES = + roundup(f->filp_vno->v_size, + PAGE_SIZE)/PAGE_SIZE; + } +#if 1 + m_out->VMV_FD = procfd; + + f->filp_count++; + fp->fp_filp[procfd] = f; + + /* mmap FD's are inuse, system-owned (user can't + * close it), and close-on-exec + */ + FD_SET(procfd, &fp->fp_filp_inuse); + FD_SET(procfd, &fp->fp_filp_system); + FD_SET(procfd, &fp->fp_cloexec_set); +#else + m_out->VMV_FD = req_fd; +#endif + +#if 0 + printf("%d: fd %d is system now\n", fp->fp_endpoint, procfd); +#endif + result = OK; + +#if 0 + printf("VFS: FDLOOKUP: result %d dev 0x%x ino %ld; procfd %d\n", + result, m_out->VMV_DEV, m_out->VMV_INO, procfd); +#endif + break; + } + case VMVFSREQ_FDCLOSE: + { + int procfd = req_fd; + scratch(fp).file.fd_nr = procfd; + result = close_fd(fp, scratch(fp).file.fd_nr, 0); + if(result != OK) { + printf("VFS: VM fd close for fd %d, %d (%d)\n", + procfd, fp->fp_endpoint, result); + } + break; + } + case VMVFSREQ_FDIO: + { + message dummy_out; + result = actual_llseek(&dummy_out, req_fd, SEEK_SET, + offset); + + if(result != OK) { + printf("VFS: llseek for mmap i/o %d (%llu) failed (%d)\n", + fp->fp_endpoint, offset, result); + } else { + fp = fdsrc; + result = do_read_write_peek(PEEKING, + req_fd, NULL, length); + } + + break; + } + default: + panic("VFS: bad request code from VM\n"); + break; + } + +reqdone: + if(f) + unlock_filp(f); + + /* We're VM now. */ + fp = vmf; + m_out->VMV_ENDPOINT = ep; + m_out->VMV_RESULT = result; + m_out->VMV_REQID = req_id; + + return VM_VFS_REPLY; +} + /*===========================================================================* * pm_reboot * *===========================================================================*/ @@ -442,7 +596,7 @@ static void free_proc(struct fproc *exiter, int flags) /* Loop on file descriptors, closing any that are open. */ for (i = 0; i < OPEN_MAX; i++) { - (void) close_fd(exiter, i); + (void) close_fd(exiter, i, 0); } /* Release root and working directories. */ @@ -685,6 +839,8 @@ int pm_dumpcore(endpoint_t proc_e, int csig, vir_bytes exe_name) struct filp *f; char core_path[PATH_MAX]; char proc_name[PROC_NAME_LEN]; + static int seq = 0; + seq++; okendpt(proc_e, &slot); fp = &fproc[slot]; @@ -696,20 +852,25 @@ int pm_dumpcore(endpoint_t proc_e, int csig, vir_bytes exe_name) unpause(fp->fp_endpoint); /* open core file */ - snprintf(core_path, PATH_MAX, "%s.%d", CORE_NAME, fp->fp_pid); + snprintf(core_path, PATH_MAX, "%s.%d.%d", CORE_NAME, fp->fp_pid, seq); core_fd = common_open(core_path, O_WRONLY | O_CREAT | O_TRUNC, CORE_MODE); if (core_fd < 0) { r = core_fd; goto core_exit; } /* get process' name */ +#if 0 r = sys_datacopy(PM_PROC_NR, exe_name, VFS_PROC_NR, (vir_bytes) proc_name, PROC_NAME_LEN); if (r != OK) goto core_exit; +#else + exe_name = 0; + strcpy(proc_name, "coreprog"); +#endif proc_name[PROC_NAME_LEN - 1] = '\0'; if ((f = get_filp(core_fd, VNODE_WRITE)) == NULL) { r=EBADF; goto core_exit; } write_elf_core_file(f, csig, proc_name); unlock_filp(f); - (void) close_fd(fp, core_fd); /* ignore failure, we're exiting anyway */ + (void) close_fd(fp, core_fd, 0); /* ignore failure, we're exiting anyway */ core_exit: if(csig) diff --git a/servers/vfs/open.c b/servers/vfs/open.c index 36b722660..d574c4a28 100644 --- a/servers/vfs/open.c +++ b/servers/vfs/open.c @@ -37,6 +37,37 @@ static struct vnode *new_node(struct lookup *resolve, int oflags, mode_t bits); static int pipe_open(struct vnode *vp, mode_t bits, int oflags); +/*===========================================================================* + * lock_close * + *===========================================================================*/ +static void lock_close(void) +{ + struct fproc *org_fp; + struct worker_thread *org_self; + + /* First try to get it right off the bat */ + if (mutex_trylock(&closefd_lock) == 0) + return; + + org_fp = fp; + org_self = self; + + if (mutex_lock(&closefd_lock) != 0) + panic("Could not obtain lock on close"); + + fp = org_fp; + self = org_self; +} + +/*===========================================================================* + * unlock_close * + *===========================================================================*/ +static void unlock_close(void) +{ + if (mutex_unlock(&closefd_lock) != 0) + panic("Could not release lock on closefd_lock"); +} + /*===========================================================================* * do_open * *===========================================================================*/ @@ -96,7 +127,7 @@ int common_open(char path[PATH_MAX], int oflags, mode_t omode) if (!bits) return(EINVAL); /* See if file descriptor and filp slots are available. */ - if ((r = get_fd(0, bits, &(scratch(fp).file.fd_nr), &filp)) != OK) return(r); + if ((r = get_fd(fp->fp_fdscan, bits, &(scratch(fp).file.fd_nr), &filp)) != OK) return(r); lookup_init(&resolve, path, PATH_NOFLAGS, &vmp, &vp); @@ -587,21 +618,13 @@ int do_mkdir(message *UNUSED(m_out)) return(r); } -/*===========================================================================* - * do_lseek * - *===========================================================================*/ -int do_lseek(message *m_out) +int actual_lseek(message *m_out, int seekfd, int seekwhence, off_t offset) { /* Perform the lseek(ls_fd, offset, whence) system call. */ register struct filp *rfilp; - int r = OK, seekfd, seekwhence; - off_t offset; + int r = OK; u64_t pos, newpos; - seekfd = job_m_in.ls_fd; - seekwhence = job_m_in.whence; - offset = (off_t) job_m_in.offset_lo; - /* Check to see if the file descriptor is valid. */ if ( (rfilp = get_filp(seekfd, VNODE_READ)) == NULL) return(err_code); @@ -647,20 +670,24 @@ int do_lseek(message *m_out) } /*===========================================================================* - * do_llseek * + * do_lseek * *===========================================================================*/ -int do_llseek(message *m_out) +int do_lseek(message *m_out) +{ + return actual_lseek(m_out, job_m_in.ls_fd, job_m_in.whence, + (off_t) job_m_in.offset_lo); +} + +/*===========================================================================* + * actual_llseek * + *===========================================================================*/ +int actual_llseek(message *m_out, int seekfd, int seekwhence, u64_t offset) { /* Perform the llseek(ls_fd, offset, whence) system call. */ register struct filp *rfilp; u64_t pos, newpos; - int r = OK, seekfd, seekwhence; - long off_hi, off_lo; - - seekfd = job_m_in.ls_fd; - seekwhence = job_m_in.whence; - off_hi = job_m_in.offset_high; - off_lo = job_m_in.offset_lo; + int r = OK; + long off_hi = ex64hi(offset); /* Check to see if the file descriptor is valid. */ if ( (rfilp = get_filp(seekfd, VNODE_READ)) == NULL) return(err_code); @@ -679,7 +706,7 @@ int do_llseek(message *m_out) default: unlock_filp(rfilp); return(EINVAL); } - newpos = add64(pos, make64(off_lo, off_hi)); + newpos = pos + offset; /* Check for overflow. */ if ((off_hi > 0) && cmp64(newpos, pos) < 0) @@ -704,24 +731,31 @@ int do_llseek(message *m_out) return(r); } +int do_llseek(message *m_out) +{ + return actual_llseek(m_out, job_m_in.ls_fd, job_m_in.whence, + make64(job_m_in.offset_lo, job_m_in.offset_high)); +} + /*===========================================================================* * do_close * *===========================================================================*/ int do_close(message *UNUSED(m_out)) { /* Perform the close(fd) system call. */ - - scratch(fp).file.fd_nr = job_m_in.fd; - return close_fd(fp, scratch(fp).file.fd_nr); + int thefd = job_m_in.fd; + scratch(fp).file.fd_nr = thefd; + return close_fd(fp, scratch(fp).file.fd_nr, 1); } /*===========================================================================* * close_fd * *===========================================================================*/ -int close_fd(rfp, fd_nr) +int close_fd(rfp, fd_nr, user_request) struct fproc *rfp; int fd_nr; +int user_request; { /* Perform the close(fd) system call. */ register struct filp *rfilp; @@ -729,8 +763,24 @@ int fd_nr; struct file_lock *flp; int lock_count; + lock_close(); + + if(user_request && + fd_nr >= 0 && fd_nr < OPEN_MAX && FD_ISSET(fd_nr, &fp->fp_filp_system)) { +#if 0 + printf("VFS: close_fd: closing system FD %d from %d, not doing it\n", + fd_nr, rfp->fp_endpoint); +#endif + unlock_close(); + return EBADF; + } + /* First locate the vnode that belongs to the file descriptor. */ - if ( (rfilp = get_filp2(rfp, fd_nr, VNODE_OPCL)) == NULL) return(err_code); + if ( (rfilp = get_filp2(rfp, fd_nr, VNODE_OPCL)) == NULL) { + unlock_close(); + return(err_code); + } + vp = rfilp->filp_vno; close_filp(rfilp); @@ -752,6 +802,8 @@ int fd_nr; lock_revive(); /* one or more locks released */ } + unlock_close(); + return(OK); } diff --git a/servers/vfs/proto.h b/servers/vfs/proto.h index a2fe8728b..49f6ad156 100644 --- a/servers/vfs/proto.h +++ b/servers/vfs/proto.h @@ -143,6 +143,7 @@ int do_fsync(message *m_out); void pm_reboot(void); int do_svrctl(message *m_out); int do_getsysinfo(void); +int do_vm_call(message *m_out); int pm_dumpcore(endpoint_t proc_e, int sig, vir_bytes exe_name); void * ds_event(void *arg); @@ -159,7 +160,7 @@ void unmount_all(int force); /* open.c */ int do_close(message *m_out); -int close_fd(struct fproc *rfp, int fd_nr); +int close_fd(struct fproc *rfp, int fd_nr, int flag); void close_reply(void); int common_open(char path[PATH_MAX], int oflags, mode_t omode); int do_creat(void); @@ -169,6 +170,8 @@ int do_mknod(message *m_out); int do_mkdir(message *m_out); int do_open(message *m_out); int do_slink(message *m_out); +int actual_lseek(message *m_out, int seekfd, int seekwhence, off_t offset); +int actual_llseek(message *m_out, int seekfd, int seekwhence, u64_t offset); int do_vm_open(void); int do_vm_close(void); diff --git a/servers/vfs/read.c b/servers/vfs/read.c index b278738f0..66c921631 100644 --- a/servers/vfs/read.c +++ b/servers/vfs/read.c @@ -125,7 +125,7 @@ int read_write(int rw_flag, struct filp *f, char *buf, size_t size, endpoint_t for_e) { register struct vnode *vp; - u64_t position, res_pos, new_pos; + u64_t position, res_pos; unsigned int cum_io, cum_io_incr, res_cum_io; int op, r; @@ -136,7 +136,10 @@ int read_write(int rw_flag, struct filp *f, char *buf, size_t size, assert(rw_flag == READING || rw_flag == WRITING || rw_flag == PEEKING); - if (size > SSIZE_MAX) return(EINVAL); + if (size > SSIZE_MAX) { + printf("read_write: size too big, %zu\n", size); + return(EINVAL); + } op = (rw_flag == READING ? VFS_DEV_READ : VFS_DEV_WRITE); @@ -144,14 +147,20 @@ int read_write(int rw_flag, struct filp *f, char *buf, size_t size, if (fp->fp_cum_io_partial != 0) { panic("VFS: read_write: fp_cum_io_partial not clear"); } - if(rw_flag == PEEKING) return EINVAL; + if(rw_flag == PEEKING) { + printf("read_write: peek on pipe makes no sense\n"); + return EINVAL; + } r = rw_pipe(rw_flag, for_e, f, buf, size); } else if (S_ISCHR(vp->v_mode)) { /* Character special files. */ dev_t dev; int suspend_reopen; int op = (rw_flag == READING ? VFS_DEV_READ : VFS_DEV_WRITE); - if(rw_flag == PEEKING) return EINVAL; + if(rw_flag == PEEKING) { + printf("read_write: peek on char device makes no sense\n"); + return EINVAL; + } if (vp->v_sdev == NO_DEV) panic("VFS: read_write tries to access char dev NO_DEV"); @@ -170,15 +179,17 @@ int read_write(int rw_flag, struct filp *f, char *buf, size_t size, if (vp->v_sdev == NO_DEV) panic("VFS: read_write tries to access block dev NO_DEV"); - if(rw_flag == PEEKING) return EINVAL; - lock_bsf(); - r = req_breadwrite(vp->v_bfs_e, for_e, vp->v_sdev, position, size, - buf, rw_flag, &res_pos, &res_cum_io); - if (r == OK) { - position = res_pos; - cum_io += res_cum_io; + if(rw_flag == PEEKING) { + r = req_bpeek(vp->v_bfs_e, vp->v_sdev, position, size); + } else { + r = req_breadwrite(vp->v_bfs_e, for_e, vp->v_sdev, + position, size, buf, rw_flag, &res_pos, &res_cum_io); + if (r == OK) { + position = res_pos; + cum_io += res_cum_io; + } } unlock_bsf(); @@ -189,16 +200,25 @@ int read_write(int rw_flag, struct filp *f, char *buf, size_t size, } /* Issue request */ - r = req_readwrite(vp->v_fs_e, vp->v_inode_nr, position, rw_flag, for_e, - buf, size, &new_pos, &cum_io_incr); + if(rw_flag == PEEKING) { + r = req_peek(vp->v_fs_e, vp->v_inode_nr, position, size); +#if 0 + printf("vfs: req_peek ino %d position 0x%llx len 0x%x returns %d\n", + vp->v_inode_nr, position, size, r); +#endif + } else { + u64_t new_pos; + r = req_readwrite(vp->v_fs_e, vp->v_inode_nr, position, + rw_flag, for_e, buf, size, &new_pos, &cum_io_incr); - if (r >= 0) { - if (ex64hi(new_pos)) - panic("read_write: bad new pos"); + if (r >= 0) { + if (ex64hi(new_pos)) + panic("read_write: bad new pos"); - position = new_pos; - cum_io += cum_io_incr; - } + position = new_pos; + cum_io += cum_io_incr; + } + } } /* On write, update file size and access time. */ @@ -292,6 +312,8 @@ size_t req_size; vp = f->filp_vno; position = cvu64(0); /* Not actually used */ + assert(rw_flag == READING || rw_flag == WRITING); + /* fp->fp_cum_io_partial is only nonzero when doing partial writes */ cum_io = fp->fp_cum_io_partial; diff --git a/servers/vfs/table.c b/servers/vfs/table.c index 5fe5b817e..6aa4f734b 100644 --- a/servers/vfs/table.c +++ b/servers/vfs/table.c @@ -132,6 +132,7 @@ int (*call_vec[])(message *m_out) = { no_sys, /* 114 = (clock_getres) */ no_sys, /* 115 = (clock_gettime) */ no_sys, /* 116 = (clock_settime) */ + do_vm_call, /* 117 = call from vm */ }; /* This should not fail with "array size is negative": */ extern int dummy[sizeof(call_vec) == NCALLS * sizeof(call_vec[0]) ? 1 : -1]; diff --git a/servers/vfs/vnode.h b/servers/vfs/vnode.h index 2f01d7379..accfae711 100644 --- a/servers/vfs/vnode.h +++ b/servers/vfs/vnode.h @@ -23,6 +23,7 @@ EXTERN struct vnode { dev_t v_sdev; /* device number for special files */ struct vmnt *v_vmnt; /* vmnt object of the partition */ tll_t v_lock; /* three-level-lock */ + int v_mmapped; /* inuse for mmap -> inform vm of shrinkage */ } vnode[NR_VNODES]; /* vnode lock types mapping */ diff --git a/servers/vfs/worker.c b/servers/vfs/worker.c index add1d3712..8b9f20f27 100644 --- a/servers/vfs/worker.c +++ b/servers/vfs/worker.c @@ -16,9 +16,9 @@ static int init = 0; static mthread_attr_t tattr; #ifdef MKCOVERAGE -# define TH_STACKSIZE (10 * 1024) +# define TH_STACKSIZE (40 * 1024) #else -# define TH_STACKSIZE (7 * 1024) +# define TH_STACKSIZE (28 * 1024) #endif #define ASSERTW(w) assert((w) == &sys_worker || (w) == &dl_worker || \ diff --git a/sys/sys/mman.h b/sys/sys/mman.h index 2c37e14c4..c4d1f838a 100644 --- a/sys/sys/mman.h +++ b/sys/sys/mman.h @@ -92,6 +92,7 @@ typedef __off_t off_t; /* file offset */ #define MAP_FIXED 0x0200 /* require mapping to happen at hint */ #define MAP_THIRDPARTY 0x0400 /* perform on behalf of any process */ #define MAP_UNINITIALIZED 0x0800 /* do not clear memory */ +#define MAP_FILE 0x1000 /* it's a file */ /* * Error indicator returned by mmap(2)