diff --git a/commands/service/parse.c b/commands/service/parse.c index a18ebef3e..ea805a073 100644 --- a/commands/service/parse.c +++ b/commands/service/parse.c @@ -732,6 +732,8 @@ struct { "RS_UPDATE", VM_RS_UPDATE }, { "RS_MEMCTL", VM_RS_MEMCTL }, { "PROCCTL", VM_PROCCTL }, + { "MAPCACHEPAGE", VM_MAPCACHEPAGE }, + { "SETCACHEPAGE", VM_SETCACHEPAGE }, { NULL, 0 }, }; diff --git a/etc/system.conf b/etc/system.conf index 022492521..496888114 100644 --- a/etc/system.conf +++ b/etc/system.conf @@ -107,7 +107,7 @@ service mfs { ipc ALL_SYS; # All system ipc targets allowed system BASIC; # Only basic kernel calls allowed - vm BASIC; # Only basic VM calls allowed + vm MAPCACHEPAGE SETCACHEPAGE; io NONE; # No I/O range allowed irq NONE; # No IRQ allowed sigmgr rs; # Signal manager is RS @@ -134,7 +134,7 @@ service ext2 { ipc ALL_SYS; # All system ipc targets allowed system BASIC; # Only basic kernel calls allowed - vm BASIC; # Only basic VM calls allowed + vm MAPCACHEPAGE SETCACHEPAGE; io NONE; # No I/O range allowed irq NONE; # No IRQ allowed sigmgr rs; # Signal manager is RS @@ -147,7 +147,7 @@ service pfs { ipc ALL_SYS; # All system ipc targets allowed system BASIC; # Only basic kernel calls allowed - vm BASIC; # Only basic VM calls allowed + vm MAPCACHEPAGE SETCACHEPAGE; io NONE; # No I/O range allowed irq NONE; # No IRQ allowed sigmgr rs; # Signal manager is RS diff --git a/include/minix/com.h b/include/minix/com.h index 2c03147c5..d2bcf9ab4 100644 --- a/include/minix/com.h +++ b/include/minix/com.h @@ -985,12 +985,31 @@ /* To VM: map in cache block by FS */ #define VM_MAPCACHEPAGE (VM_RQ_BASE+26) +/* To VM: identify cache block in FS */ +#define VM_SETCACHEPAGE (VM_RQ_BASE+27) + +/* To VFS: fields for request from VM. */ +# define VFS_VMCALL_REQ m10_i1 +# define VFS_VMCALL_FD m10_i2 +# define VFS_VMCALL_REQID m10_i3 +# define VFS_VMCALL_ENDPOINT m10_i4 +# define VFS_VMCALL_OFFSET_LO m10_l1 +# define VFS_VMCALL_OFFSET_HI m10_l2 +# define VFS_VMCALL_LENGTH m10_l3 + +/* Request codes to from VM to VFS */ +#define VMVFSREQ_FDLOOKUP 101 +#define VMVFSREQ_FDCLOSE 102 +#define VMVFSREQ_FDIO 103 + /* Calls from VFS. */ -# define VMV_ENDPOINT m1_i1 /* for all VM_VFS_REPLY_* */ -#define VM_VFS_REPLY_OPEN (VM_RQ_BASE+30) -# define VMVRO_FD m1_i2 -#define VM_VFS_REPLY_MMAP (VM_RQ_BASE+31) -#define VM_VFS_REPLY_CLOSE (VM_RQ_BASE+32) +#define VM_VFS_REPLY (VM_RQ_BASE+30) +# define VMV_ENDPOINT m10_i1 +# define VMV_RESULT m10_i2 +# define VMV_REQID m10_i3 +# define VMV_DEV m10_i4 +# define VMV_INO m10_l1 +# define VMV_FD m10_l2 #define VM_REMAP (VM_RQ_BASE+33) # define VMRE_D m1_i1 diff --git a/include/minix/ipc.h b/include/minix/ipc.h index a0f881365..fc61d6c09 100644 --- a/include/minix/ipc.h +++ b/include/minix/ipc.h @@ -30,6 +30,30 @@ typedef struct {long m9l1, m9l2, m9l3, m9l4, m9l5; typedef struct {int m10i1, m10i2, m10i3, m10i4; long m10l1, m10l2, m10l3; } mess_10; +typedef struct { + void *block; + u32_t dev_offset_pages; + u32_t ino_offset_pages; + u32_t ino; + u32_t *flags_ptr; + u32_t dev; + u8_t pages; + u8_t flags; +} mess_vmmcp __packed; + +typedef struct { + endpoint_t who; + u32_t offset; + u32_t dev; + u32_t ino; + u32_t vaddr; + u32_t len; + u16_t fd; + u16_t clearend_and_flags; /* low 12 bits are clearend, rest flags */ +} mess_vm_vfs_mmap __packed; + +typedef struct { u8_t flags; void *addr; } mess_vmmcp_reply __packed; + typedef struct { endpoint_t m_source; /* who sent the message */ int m_type; /* what kind of message is it */ @@ -44,6 +68,9 @@ typedef struct { mess_6 m_m6; mess_9 m_m9; mess_10 m_m10; + mess_vmmcp m_vmmcp; + mess_vmmcp_reply m_vmmcp_reply; + mess_vm_vfs_mmap m_vm_vfs; } m_u; } message __aligned(16); diff --git a/include/minix/libminixfs.h b/include/minix/libminixfs.h index d68ef5e79..1f32bc166 100644 --- a/include/minix/libminixfs.h +++ b/include/minix/libminixfs.h @@ -17,9 +17,16 @@ struct buf { struct buf *lmfs_hash; /* used to link bufs on hash chains */ block_t lmfs_blocknr; /* block number of its (minor) device */ dev_t lmfs_dev; /* major | minor device where block resides */ - char lmfs_dirt; /* BP_CLEAN or BP_DIRTY */ char lmfs_count; /* number of users of this buffer */ + char lmfs_needsetcache; /* to be identified to VM */ unsigned int lmfs_bytes; /* Number of bytes allocated in bp */ + u32_t lmfs_flags; /* Flags shared between VM and FS */ + + /* If any, which inode & offset does this block correspond to? + * If none, VMC_NO_INODE + */ + ino_t lmfs_inode; + u64_t lmfs_inode_offset; }; int fs_lookup_credentials(vfs_ucred_t *credentials, @@ -42,9 +49,12 @@ void lmfs_reset_rdwt_err(void); int lmfs_rdwt_err(void); void lmfs_buf_pool(int new_nr_bufs); struct buf *lmfs_get_block(dev_t dev, block_t block,int only_search); +struct buf *lmfs_get_block_ino(dev_t dev, block_t block,int only_search, + ino_t ino, u64_t off); void lmfs_invalidate(dev_t device); void lmfs_put_block(struct buf *bp, int block_type); void lmfs_rw_scattered(dev_t, struct buf **, int, int); +int lmfs_do_bpeek(message *); /* calls that libminixfs does into fs */ void fs_blockstats(u32_t *blocks, u32_t *free, u32_t *used); diff --git a/include/minix/vm.h b/include/minix/vm.h index acb0a901a..698905954 100644 --- a/include/minix/vm.h +++ b/include/minix/vm.h @@ -65,5 +65,20 @@ int vm_info_region(endpoint_t who, struct vm_region_info *vri, int count, vir_bytes *next); int vm_procctl(endpoint_t ep, int param); +int vm_set_cacheblock(void *block, u32_t dev, u64_t dev_offset, + u64_t ino, u64_t ino_offset, u32_t *flags, int blocksize); + +void *vm_map_cacheblock(u32_t dev, u64_t dev_offset, + u64_t ino, u64_t ino_offset, u32_t *flags, int blocksize); + +/* flags for vm cache functions */ +#define VMMC_FLAGS_LOCKED 0x01 /* someone is updating the flags; don't read/write */ +#define VMMC_DIRTY 0x02 /* dirty buffer and it may not be evicted */ +#define VMMC_EVICTED 0x04 /* VM has evicted the buffer and it's invalid */ +#define VMMC_BLOCK_LOCKED 0x08 /* client is using it and it may not be evicted */ + +/* special inode number for vm cache functions */ +#define VMC_NO_INODE 0 /* to reference a disk block, no associated file */ + #endif /* _MINIX_VM_H */ diff --git a/lib/libminixfs/Makefile b/lib/libminixfs/Makefile index ade070374..ae0324ebc 100644 --- a/lib/libminixfs/Makefile +++ b/lib/libminixfs/Makefile @@ -1,6 +1,5 @@ # Makefile for libminixfs .include - LIB= minixfs SRCS= fetch_credentials.c cache.c diff --git a/lib/libminixfs/cache.c b/lib/libminixfs/cache.c index fae021087..57aeeaa97 100644 --- a/lib/libminixfs/cache.c +++ b/lib/libminixfs/cache.c @@ -7,7 +7,7 @@ #include #include -#include +#include #include #include @@ -16,9 +16,6 @@ #include #include -#define BP_CLEAN 0 /* on-disk block and memory copies identical */ -#define BP_DIRTY 1 /* on-disk block and memory copies differ */ - #define BUFHASH(b) ((b) % nr_bufs) #define MARKCLEAN lmfs_markclean @@ -31,6 +28,7 @@ static unsigned int bufs_in_use;/* # bufs currently in use (not on free list)*/ static void rm_lru(struct buf *bp); static void read_block(struct buf *); static void flushall(dev_t dev); +static void freeblock(struct buf *bp); static int vmcache = 0; /* are we using vm's secondary cache? (initially not) */ @@ -53,13 +51,6 @@ u32_t fs_bufs_heuristic(int minbufs, u32_t btotal, u32_t bfree, bused = btotal-bfree; - /* but we simply need minbufs no matter what, and we don't - * want more than that if we're a memory device - */ - if(majordev == MEMORY_MAJOR) { - return minbufs; - } - /* set a reasonable cache size; cache at most a certain * portion of the used FS, and at most a certain %age of remaining * memory @@ -96,19 +87,19 @@ u32_t fs_bufs_heuristic(int minbufs, u32_t btotal, u32_t bfree, void lmfs_markdirty(struct buf *bp) { - bp->lmfs_dirt = BP_DIRTY; + bp->lmfs_flags |= VMMC_DIRTY; } void lmfs_markclean(struct buf *bp) { - bp->lmfs_dirt = BP_CLEAN; + bp->lmfs_flags &= ~VMMC_DIRTY; } int lmfs_isclean(struct buf *bp) { - return bp->lmfs_dirt == BP_CLEAN; + return !(bp->lmfs_flags & VMMC_DIRTY); } dev_t @@ -122,14 +113,109 @@ int lmfs_bytes(struct buf *bp) return bp->lmfs_bytes; } +static void +free_unused_blocks(void) +{ + struct buf *bp; + + int freed = 0, bytes = 0; + printf("libminixfs: freeing; %d blocks in use\n", bufs_in_use); + for(bp = &buf[0]; bp < &buf[nr_bufs]; bp++) { + if(bp->lmfs_bytes > 0 && bp->lmfs_count == 0) { + freed++; + bytes += bp->lmfs_bytes; + freeblock(bp); + } + } + printf("libminixfs: freeing; %d blocks, %d bytes\n", freed, bytes); +} + +static void +lmfs_alloc_block(struct buf *bp) +{ + ASSERT(!bp->data); + ASSERT(bp->lmfs_bytes == 0); + ASSERT(!(fs_block_size % PAGE_SIZE)); + if((bp->data = minix_mmap(0, fs_block_size, + PROT_READ|PROT_WRITE, MAP_PREALLOC|MAP_ANON, -1, 0)) == MAP_FAILED) { + free_unused_blocks(); + if((bp->data = minix_mmap(0, fs_block_size, PROT_READ|PROT_WRITE, + MAP_PREALLOC|MAP_ANON, -1, 0)) == MAP_FAILED) { + panic("libminixfs: could not allocate block"); + } + } + assert(bp->data); + bp->lmfs_bytes = fs_block_size; + bp->lmfs_needsetcache = 1; +} + /*===========================================================================* * lmfs_get_block * *===========================================================================*/ -struct buf *lmfs_get_block( - register dev_t dev, /* on which device is the block? */ - register block_t block, /* which block is wanted? */ - int only_search /* if NO_READ, don't read, else act normal */ -) +struct buf *lmfs_get_block(register dev_t dev, register block_t block, + int only_search) +{ + return lmfs_get_block_ino(dev, block, only_search, VMC_NO_INODE, 0); +} + +void minix_munmap_t(void *a, int len) +{ + vir_bytes av = (vir_bytes) a; + assert(a); + assert(a != MAP_FAILED); + assert(len > 0); + assert(!(len % PAGE_SIZE)); + assert(!(av % PAGE_SIZE)); + + if(minix_munmap(a, len) < 0) + panic("libminixfs cache: munmap failed"); +} + +static void raisecount(struct buf *bp) +{ + assert(bufs_in_use >= 0); + ASSERT(bp->lmfs_count >= 0); + bp->lmfs_count++; + if(bp->lmfs_count == 1) bufs_in_use++; + assert(bufs_in_use > 0); +} + +static void lowercount(struct buf *bp) +{ + assert(bufs_in_use > 0); + ASSERT(bp->lmfs_count > 0); + bp->lmfs_count--; + if(bp->lmfs_count == 0) bufs_in_use--; + assert(bufs_in_use >= 0); +} + +static void freeblock(struct buf *bp) +{ + ASSERT(bp->lmfs_count == 0); + /* If the block taken is dirty, make it clean by writing it to the disk. + * Avoid hysteresis by flushing all other dirty blocks for the same device. + */ + if (bp->lmfs_dev != NO_DEV) { + if (!lmfs_isclean(bp)) flushall(bp->lmfs_dev); + assert(bp->lmfs_bytes == fs_block_size); + bp->lmfs_dev = NO_DEV; + } + + /* Fill in block's parameters and add it to the hash chain where it goes. */ + MARKCLEAN(bp); /* NO_DEV blocks may be marked dirty */ + if(bp->lmfs_bytes > 0) { + assert(bp->data); + minix_munmap_t(bp->data, bp->lmfs_bytes); + bp->lmfs_bytes = 0; + bp->data = NULL; + } else assert(!bp->data); +} + +/*===========================================================================* + * lmfs_get_block_ino * + *===========================================================================*/ +struct buf *lmfs_get_block_ino(dev_t dev, block_t block, int only_search, + ino_t ino, u64_t ino_off) { /* Check to see if the requested block is in the block cache. If so, return * a pointer to it. If not, evict some other block and fetch it (unless @@ -147,8 +233,9 @@ struct buf *lmfs_get_block( */ int b; - static struct buf *bp, *prev_ptr; - u64_t yieldid = VM_BLOCKID_NONE /*, getid = make64(dev, block) */; + static struct buf *bp; + u64_t dev_off = (u64_t) block * fs_block_size; + struct buf *prev_ptr; assert(buf_hash); assert(buf); @@ -158,22 +245,52 @@ struct buf *lmfs_get_block( assert(dev != NO_DEV); - /* Search the hash chain for (dev, block). Do_read() can use - * lmfs_get_block(NO_DEV ...) to get an unnamed block to fill with zeros when - * someone wants to read from a hole in a file, in which case this search - * is skipped - */ + if((ino_off % fs_block_size)) { + + printf("cache: unaligned lmfs_get_block_ino ino_off %llu\n", + ino_off); + util_stacktrace(); + } + + /* Search the hash chain for (dev, block). */ b = BUFHASH(block); bp = buf_hash[b]; while (bp != NULL) { if (bp->lmfs_blocknr == block && bp->lmfs_dev == dev) { + if(bp->lmfs_flags & VMMC_EVICTED) { + /* We had it but VM evicted it; invalidate it. */ + ASSERT(bp->lmfs_count == 0); + ASSERT(!(bp->lmfs_flags & VMMC_BLOCK_LOCKED)); + ASSERT(!(bp->lmfs_flags & VMMC_DIRTY)); + bp->lmfs_dev = NO_DEV; + bp->lmfs_bytes = 0; + bp->data = NULL; + break; + } + ASSERT(bp->lmfs_needsetcache == 0); /* Block needed has been found. */ - if (bp->lmfs_count == 0) rm_lru(bp); - bp->lmfs_count++; /* record that block is in use */ + if (bp->lmfs_count == 0) { + rm_lru(bp); + ASSERT(!(bp->lmfs_flags & VMMC_BLOCK_LOCKED)); + bp->lmfs_flags |= VMMC_BLOCK_LOCKED; + } + raisecount(bp); ASSERT(bp->lmfs_bytes == fs_block_size); ASSERT(bp->lmfs_dev == dev); ASSERT(bp->lmfs_dev != NO_DEV); + ASSERT(bp->lmfs_flags & VMMC_BLOCK_LOCKED); ASSERT(bp->data); + + if(ino != VMC_NO_INODE) { + if(bp->lmfs_inode == VMC_NO_INODE + || bp->lmfs_inode != ino + || bp->lmfs_inode_offset != ino_off) { + bp->lmfs_inode = ino; + bp->lmfs_inode_offset = ino_off; + bp->lmfs_needsetcache = 1; + } + } + return(bp); } else { /* This block is not the one sought. */ @@ -181,29 +298,13 @@ struct buf *lmfs_get_block( } } - /* Desired block is not on available chain. Take oldest block ('front'). */ - if ((bp = front) == NULL) panic("all buffers in use: %d", nr_bufs); - - if(bp->lmfs_bytes < fs_block_size) { - ASSERT(!bp->data); - ASSERT(bp->lmfs_bytes == 0); - if(!(bp->data = alloc_contig( (size_t) fs_block_size, 0, NULL))) { - printf("fs cache: couldn't allocate a new block.\n"); - for(bp = front; - bp && bp->lmfs_bytes < fs_block_size; bp = bp->lmfs_next) - ; - if(!bp) { - panic("no buffer available"); - } - } else { - bp->lmfs_bytes = fs_block_size; - } + /* Desired block is not on available chain. Find a free block to use. */ + if(bp) { + ASSERT(bp->lmfs_flags & VMMC_EVICTED); + } else { + if ((bp = front) == NULL) panic("all buffers in use: %d", nr_bufs); } - - ASSERT(bp); - ASSERT(bp->data); - ASSERT(bp->lmfs_bytes == fs_block_size); - ASSERT(bp->lmfs_count == 0); + assert(bp); rm_lru(bp); @@ -223,25 +324,17 @@ struct buf *lmfs_get_block( } } - /* If the block taken is dirty, make it clean by writing it to the disk. - * Avoid hysteresis by flushing all other dirty blocks for the same device. - */ - if (bp->lmfs_dev != NO_DEV) { - if (bp->lmfs_dirt == BP_DIRTY) flushall(bp->lmfs_dev); + freeblock(bp); - /* Are we throwing out a block that contained something? - * Give it to VM for the second-layer cache. - */ - yieldid = make64(bp->lmfs_dev, bp->lmfs_blocknr); - assert(bp->lmfs_bytes == fs_block_size); - bp->lmfs_dev = NO_DEV; - } + bp->lmfs_inode = ino; + bp->lmfs_inode_offset = ino_off; - /* Fill in block's parameters and add it to the hash chain where it goes. */ - MARKCLEAN(bp); /* NO_DEV blocks may be marked dirty */ + bp->lmfs_flags = VMMC_BLOCK_LOCKED; + bp->lmfs_needsetcache = 0; bp->lmfs_dev = dev; /* fill in device number */ bp->lmfs_blocknr = block; /* fill in block number */ - bp->lmfs_count++; /* record that block is being used */ + ASSERT(bp->lmfs_count == 0); + raisecount(bp); b = BUFHASH(bp->lmfs_blocknr); bp->lmfs_hash = buf_hash[b]; @@ -249,23 +342,26 @@ struct buf *lmfs_get_block( assert(dev != NO_DEV); - /* Go get the requested block unless searching or prefetching. */ - if(only_search == PREFETCH || only_search == NORMAL) { - /* Block is not found in our cache, but we do want it - * if it's in the vm cache. - */ - if(vmcache) { - /* If we can satisfy the PREFETCH or NORMAL request - * from the vm cache, work is done. - */ -#if 0 - if(vm_yield_block_get_block(yieldid, getid, - bp->data, fs_block_size) == OK) { - return bp; - } -#endif + /* Block is not found in our cache, but we do want it + * if it's in the vm cache. + */ + assert(!bp->data); + assert(!bp->lmfs_bytes); + if(vmcache) { + if((bp->data = vm_map_cacheblock(dev, dev_off, ino, ino_off, + &bp->lmfs_flags, fs_block_size)) != MAP_FAILED) { + bp->lmfs_bytes = fs_block_size; + ASSERT(!bp->lmfs_needsetcache); + return bp; } } + bp->data = NULL; + + /* Not in the cache; reserve memory for its contents. */ + + lmfs_alloc_block(bp); + + assert(bp->data); if(only_search == PREFETCH) { /* PREFETCH: don't do i/o. */ @@ -273,15 +369,7 @@ struct buf *lmfs_get_block( } else if (only_search == NORMAL) { read_block(bp); } else if(only_search == NO_READ) { - /* we want this block, but its contents - * will be overwritten. VM has to forget - * about it. - */ -#if 0 - if(vmcache) { - vm_forgetblock(getid); - } -#endif + /* This block will be overwritten by new contents. */ } else panic("unexpected only_search value: %d", only_search); @@ -305,15 +393,21 @@ int block_type; /* INODE_BLOCK, DIRECTORY_BLOCK, or whatever */ * the integrity of the file system (e.g., inode blocks) are written to * disk immediately if they are dirty. */ + dev_t dev; + u64_t dev_off; + int r; + if (bp == NULL) return; /* it is easier to check here than in caller */ - bp->lmfs_count--; /* there is one use fewer now */ + dev = bp->lmfs_dev; + + dev_off = (u64_t) bp->lmfs_blocknr * fs_block_size; + + lowercount(bp); if (bp->lmfs_count != 0) return; /* block is still in use */ - bufs_in_use--; /* one fewer block buffers in use */ - /* Put this block back on the LRU chain. */ - if (bp->lmfs_dev == DEV_RAM || (block_type & ONE_SHOT)) { + if (dev == DEV_RAM || (block_type & ONE_SHOT)) { /* Block probably won't be needed quickly. Put it on front of chain. * It will be the next block to be evicted from the cache. */ @@ -337,6 +431,25 @@ int block_type; /* INODE_BLOCK, DIRECTORY_BLOCK, or whatever */ rear->lmfs_next = bp; rear = bp; } + + assert(bp->lmfs_flags & VMMC_BLOCK_LOCKED); + bp->lmfs_flags &= ~VMMC_BLOCK_LOCKED; + + /* block has sensible content - if necesary, identify it to VM */ + if(vmcache && bp->lmfs_needsetcache && dev != NO_DEV) { + if((r=vm_set_cacheblock(bp->data, dev, dev_off, + bp->lmfs_inode, bp->lmfs_inode_offset, + &bp->lmfs_flags, fs_block_size)) != OK) { + if(r == ENOSYS) { + printf("libminixfs: ENOSYS, disabling VM calls\n"); + vmcache = 0; + } else { + panic("libminixfs: setblock of 0x%lx dev 0x%x off " + "0x%llx failed\n", bp->data, dev, dev_off); + } + } + } + bp->lmfs_needsetcache = 0; } /*===========================================================================* @@ -358,9 +471,28 @@ register struct buf *bp; /* buffer pointer */ assert(dev != NO_DEV); + ASSERT(bp->lmfs_bytes == fs_block_size); + ASSERT(fs_block_size > 0); + ASSERT(!(fs_block_size % PAGE_SIZE)); + pos = mul64u(bp->lmfs_blocknr, fs_block_size); - r = bdev_read(dev, pos, bp->data, fs_block_size, - BDEV_NOFLAGS); + if(fs_block_size > PAGE_SIZE) { +#define MAXPAGES 20 + vir_bytes vaddr = (vir_bytes) bp->data; + int p; + static iovec_t iovec[MAXPAGES]; + int pages = fs_block_size/PAGE_SIZE; + ASSERT(pages > 1 && pages < MAXPAGES); + for(p = 0; p < pages; p++) { + iovec[p].iov_addr = vaddr; + iovec[p].iov_size = PAGE_SIZE; + vaddr += PAGE_SIZE; + } + r = bdev_gather(dev, pos, iovec, pages, BDEV_NOFLAGS); + } else { + r = bdev_read(dev, pos, bp->data, fs_block_size, + BDEV_NOFLAGS); + } if (r < 0) { printf("fs cache: I/O error on device %d/%d, block %u\n", major(dev), minor(dev), bp->lmfs_blocknr); @@ -376,6 +508,7 @@ register struct buf *bp; /* buffer pointer */ /* Report read errors to interested parties. */ rdwt_err = r; } + } /*===========================================================================* @@ -389,10 +522,16 @@ void lmfs_invalidate( register struct buf *bp; - for (bp = &buf[0]; bp < &buf[nr_bufs]; bp++) - if (bp->lmfs_dev == device) bp->lmfs_dev = NO_DEV; - - /* vm_forgetblocks(); */ + for (bp = &buf[0]; bp < &buf[nr_bufs]; bp++) { + if (bp->lmfs_dev == device) { + assert(bp->data); + assert(bp->lmfs_bytes > 0); + minix_munmap_t(bp->data, bp->lmfs_bytes); + bp->lmfs_dev = NO_DEV; + bp->lmfs_bytes = 0; + bp->data = NULL; + } + } } /*===========================================================================* @@ -418,7 +557,7 @@ static void flushall(dev_t dev) } for (bp = &buf[0], ndirty = 0; bp < &buf[nr_bufs]; bp++) { - if (bp->lmfs_dirt == BP_DIRTY && bp->lmfs_dev == dev) { + if (!lmfs_isclean(bp) && bp->lmfs_dev == dev) { dirty[ndirty++] = bp; } } @@ -444,16 +583,22 @@ void lmfs_rw_scattered( register iovec_t *iop; static iovec_t *iovec = NULL; u64_t pos; - int j, r; + int iov_per_block; STATICINIT(iovec, NR_IOREQS); + assert(dev != NO_DEV); + assert(!(fs_block_size % PAGE_SIZE)); + assert(fs_block_size > 0); + iov_per_block = fs_block_size / PAGE_SIZE; + /* (Shell) sort buffers on lmfs_blocknr. */ gap = 1; do gap = 3 * gap + 1; while (gap <= bufqsize); while (gap != 1) { + int j; gap /= 3; for (j = gap; j < bufqsize; j++) { for (i = j - gap; @@ -470,17 +615,33 @@ void lmfs_rw_scattered( * went fine, otherwise the error code for the first failed transfer. */ while (bufqsize > 0) { - for (j = 0, iop = iovec; j < NR_IOREQS && j < bufqsize; j++, iop++) { - bp = bufq[j]; - if (bp->lmfs_blocknr != (block_t) bufq[0]->lmfs_blocknr + j) break; - iop->iov_addr = (vir_bytes) bp->data; - iop->iov_size = (vir_bytes) fs_block_size; + int nblocks = 0, niovecs = 0; + int r; + for (iop = iovec; nblocks < bufqsize; nblocks++) { + int p; + vir_bytes vdata; + bp = bufq[nblocks]; + if (bp->lmfs_blocknr != (block_t) bufq[0]->lmfs_blocknr + nblocks) + break; + if(niovecs >= NR_IOREQS-iov_per_block) break; + vdata = (vir_bytes) bp->data; + for(p = 0; p < iov_per_block; p++) { + iop->iov_addr = vdata; + iop->iov_size = PAGE_SIZE; + vdata += PAGE_SIZE; + iop++; + niovecs++; + } } + + assert(nblocks > 0); + assert(niovecs > 0); + pos = mul64u(bufq[0]->lmfs_blocknr, fs_block_size); if (rw_flag == READING) - r = bdev_gather(dev, pos, iovec, j, BDEV_NOFLAGS); + r = bdev_gather(dev, pos, iovec, niovecs, BDEV_NOFLAGS); else - r = bdev_scatter(dev, pos, iovec, j, BDEV_NOFLAGS); + r = bdev_scatter(dev, pos, iovec, niovecs, BDEV_NOFLAGS); /* Harvest the results. The driver may have returned an error, or it * may have done less than what we asked for. @@ -489,13 +650,12 @@ void lmfs_rw_scattered( printf("fs cache: I/O error %d on device %d/%d, block %u\n", r, major(dev), minor(dev), bufq[0]->lmfs_blocknr); } - for (i = 0; i < j; i++) { + for (i = 0; i < nblocks; i++) { bp = bufq[i]; if (r < (ssize_t) fs_block_size) { /* Transfer failed. */ if (i == 0) { bp->lmfs_dev = NO_DEV; /* Invalidate block */ - /* vm_forgetblocks(); */ } break; } @@ -507,8 +667,8 @@ void lmfs_rw_scattered( } r -= fs_block_size; } - bufq += i; - bufqsize -= i; + bufq += nblocks; + bufqsize -= nblocks; if (rw_flag == READING) { /* Don't bother reading more than the device is willing to * give at this time. Don't forget to release those extras. @@ -538,7 +698,6 @@ struct buf *bp; /* Remove a block from its LRU chain. */ struct buf *next_ptr, *prev_ptr; - bufs_in_use++; next_ptr = bp->lmfs_next; /* successor on LRU chain */ prev_ptr = bp->lmfs_prev; /* predecessor on LRU chain */ if (prev_ptr != NULL) @@ -594,13 +753,10 @@ void lmfs_set_blocksize(int new_block_size, int major) * - our main FS device isn't a memory device */ -#if 0 vmcache = 0; - if(vm_forgetblock(VM_BLOCKID_NONE) != ENOSYS && - may_use_vmcache && major != MEMORY_MAJOR) { + + if(may_use_vmcache) vmcache = 1; - } -#endif } /*===========================================================================* @@ -619,7 +775,7 @@ void lmfs_buf_pool(int new_nr_bufs) for (bp = &buf[0]; bp < &buf[nr_bufs]; bp++) { if(bp->data) { assert(bp->lmfs_bytes > 0); - free_contig(bp->data, bp->lmfs_bytes); + minix_munmap_t(bp->data, bp->lmfs_bytes); } } } @@ -654,8 +810,6 @@ void lmfs_buf_pool(int new_nr_bufs) for (bp = &buf[0]; bp < &buf[nr_bufs]; bp++) bp->lmfs_hash = bp->lmfs_next; buf_hash[0] = front; - - /* vm_forgetblocks(); */ } int lmfs_bufs_in_use(void) @@ -672,7 +826,7 @@ void lmfs_flushall(void) { struct buf *bp; for(bp = &buf[0]; bp < &buf[nr_bufs]; bp++) - if(bp->lmfs_dev != NO_DEV && bp->lmfs_dirt == BP_DIRTY) + if(bp->lmfs_dev != NO_DEV && !lmfs_isclean(bp)) flushall(bp->lmfs_dev); } @@ -695,3 +849,4 @@ int lmfs_rdwt_err(void) { return rdwt_err; } + diff --git a/lib/libsys/Makefile b/lib/libsys/Makefile index 1231a1b66..b604e1433 100644 --- a/lib/libsys/Makefile +++ b/lib/libsys/Makefile @@ -78,6 +78,7 @@ SRCS+= \ tickdelay.c \ timers.c \ vm_brk.c \ + vm_cache.c \ vm_exit.c \ vm_fork.c \ vm_info.c \ diff --git a/lib/libsys/vm_cache.c b/lib/libsys/vm_cache.c new file mode 100644 index 000000000..73883203c --- /dev/null +++ b/lib/libsys/vm_cache.c @@ -0,0 +1,62 @@ + +#include "syslib.h" + +#include +#include + +#include +#include +#include +#include + +int vm_cachecall(message *m, int call, void *addr, u32_t dev, u64_t dev_offset, + u64_t ino, u64_t ino_offset, u32_t *flags, int blocksize) +{ + if(blocksize % PAGE_SIZE) + panic("blocksize %d should be a multiple of pagesize %d\n", + blocksize, PAGE_SIZE); + + if(ino_offset % PAGE_SIZE) + panic("inode offset %d should be a multiple of pagesize %d\n", + ino_offset, PAGE_SIZE); + + if(dev_offset % PAGE_SIZE) + panic("dev offset offset %d should be a multiple of pagesize %d\n", + dev_offset, PAGE_SIZE); + + memset(m, 0, sizeof(*m)); + + assert(dev != NO_DEV); + + m->m_u.m_vmmcp.dev_offset_pages = dev_offset/PAGE_SIZE; + m->m_u.m_vmmcp.ino_offset_pages = ino_offset/PAGE_SIZE; + m->m_u.m_vmmcp.ino = ino; + m->m_u.m_vmmcp.block = addr; + m->m_u.m_vmmcp.flags_ptr = flags; + m->m_u.m_vmmcp.dev = dev; + m->m_u.m_vmmcp.pages = blocksize / PAGE_SIZE; + m->m_u.m_vmmcp.flags = 0; + + return _taskcall(VM_PROC_NR, call, m); +} + +void *vm_map_cacheblock(u32_t dev, u64_t dev_offset, + u64_t ino, u64_t ino_offset, u32_t *flags, int blocksize) +{ + message m; + + if(vm_cachecall(&m, VM_MAPCACHEPAGE, NULL, dev, dev_offset, + ino, ino_offset, flags, blocksize) != OK) + return MAP_FAILED; + + return m.m_u.m_vmmcp_reply.addr; +} + +int vm_set_cacheblock(void *block, u32_t dev, u64_t dev_offset, + u64_t ino, u64_t ino_offset, u32_t *flags, int blocksize) +{ + message m; + + return vm_cachecall(&m, VM_SETCACHEPAGE, block, dev, dev_offset, + ino, ino_offset, flags, blocksize); +}