diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index 5f426228f4c76a..97f7cdcab43289 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -726,6 +726,10 @@ static int fuse_create_open(struct mnt_idmap *idmap, struct inode *dir, memset(&inarg, 0, sizeof(inarg)); memset(&outentry, 0, sizeof(outentry)); inarg.flags = flags; + + /* The kernel owns append positioning; see fuse_send_open() */ + if (fm->fc->writeback_cache) + inarg.flags &= ~O_APPEND; inarg.mode = mode; inarg.umask = current_umask(); @@ -2106,34 +2110,24 @@ int fuse_do_setattr(struct mnt_idmap *idmap, struct dentry *dentry, WARN_ON(!(attr->ia_valid & ATTR_SIZE)); WARN_ON(attr->ia_size != 0); if (fc->atomic_o_trunc) { - struct percpu_rw_semaphore *wb_sem = fi->wb_inval_rwsem; - /* * No need to send request to userspace, since actual * truncation has already been done by OPEN. But still * need to truncate page cache. * - * Revoke and drop under the coherency gate write side, - * like the NOTIFY invalidate path: a gate reader that - * already re-validated its grant must not have the - * lock tree and the cache yanked mid-hold, or it - * would repopulate the truncated range trusting a - * grant that no longer exists. Waiting for gate - * readers here is safe: we hold i_rwsem exclusive, so - * no gate holder can be waiting on it (the write path - * takes i_rwsem before the gate, the read path never - * takes it). + * Dropping every grant here does not need a reader or + * writer fenced out: truncate_pagecache() discards the + * folios rather than writing them, and a write racing + * this is a write racing an O_TRUNC open, which has no + * order to preserve. i_rwsem is held exclusive + * anyway, so no cached write is in progress. */ - if (wb_sem) - percpu_down_write(wb_sem); if (fc->dlm && fc->writeback_cache) fuse_dlm_cache_release_locks(fi); spin_lock(&fi->lock); i_size_write(inode, 0); spin_unlock(&fi->lock); truncate_pagecache(inode, 0); - if (wb_sem) - percpu_up_write(wb_sem); goto out; } file = NULL; @@ -2237,23 +2231,17 @@ int fuse_do_setattr(struct mnt_idmap *idmap, struct dentry *dentry, */ if ((is_truncate || !is_wb) && S_ISREG(inode->i_mode) && oldsize != outarg.attr.size) { - struct percpu_rw_semaphore *wb_sem = fi->wb_inval_rwsem; - /* - * Revoke and drop under the coherency gate write side; see - * the atomic-O_TRUNC branch above. i_rwsem is held - * exclusive here as well (setattr), so waiting out gate - * readers cannot deadlock. + * Revoke past the new size and drop what is beyond it; see + * the atomic-O_TRUNC branch above for why this needs nothing + * fenced out. i_rwsem is held exclusive here as well. */ - if (wb_sem) - percpu_down_write(wb_sem); if (fc->dlm && fc->writeback_cache) - fuse_dlm_unlock_range(fi, outarg.attr.size & PAGE_MASK, -1); + fuse_dlm_unlock_range(fi, outarg.attr.size & PAGE_MASK, + U64_MAX); truncate_pagecache(inode, outarg.attr.size); invalidate_inode_pages2(mapping); - if (wb_sem) - percpu_up_write(wb_sem); } clear_bit(FUSE_I_SIZE_UNSTABLE, &fi->state); diff --git a/fs/fuse/file.c b/fs/fuse/file.c index 7c5b5e8b4ba268..e21a08dd5704df 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -71,6 +71,17 @@ static int fuse_send_open(struct fuse_mount *fm, u64 nodeid, if (!fm->fc->atomic_o_trunc) inarg.flags &= ~O_TRUNC; + /* + * With the writeback cache the kernel owns append positioning: + * writeback sends FUSE_WRITE with explicit offsets, and a server + * that opens its backing file O_APPEND has pwrite(2) ignore them + * (Linux appends regardless of offset). Any re-sent or reordered + * run is then placed at EOF: duplicated data and a growing file. + * Do not hand the flag to the server at all. + */ + if (fm->fc->writeback_cache) + inarg.flags &= ~O_APPEND; + if (fm->fc->handle_killpriv_v2 && (inarg.flags & O_TRUNC) && !capable(CAP_FSETID)) { inarg.open_flags |= FUSE_OPEN_KILL_SUIDGID; @@ -174,6 +185,10 @@ static int fuse_compound_open_getattr(struct fuse_mount *fm, u64 nodeid, if (!fm->fc->atomic_o_trunc) open_in.flags &= ~O_TRUNC; + /* The kernel owns append positioning; see fuse_send_open() */ + if (fm->fc->writeback_cache) + open_in.flags &= ~O_APPEND; + if (fm->fc->handle_killpriv_v2 && (open_in.flags & O_TRUNC) && !capable(CAP_FSETID)) open_in.open_flags |= FUSE_OPEN_KILL_SUIDGID; @@ -384,10 +399,20 @@ static int fuse_open(struct inode *inode, struct file *file) if (is_wb_truncate || dax_truncate) fuse_release_nowrite(inode); if (!err) { - if (is_truncate) + if (is_truncate) { + /* + * Every grant goes with the cache, as on the + * fuse_do_setattr() O_TRUNC path: a record left + * behind would keep naming bytes the folios no + * longer hold. i_rwsem is held exclusive + * (is_wb_truncate), so no cached write is mid-record. + */ + if (fc->dlm && fc->writeback_cache) + fuse_dlm_cache_release_locks(fi); truncate_pagecache(inode, 0); - else if (!(ff->open_flags & FOPEN_KEEP_CACHE)) + } else if (!(ff->open_flags & FOPEN_KEEP_CACHE)) { invalidate_inode_pages2(inode->i_mapping); + } } if (dax_truncate) filemap_invalidate_unlock(inode->i_mapping); @@ -469,8 +494,8 @@ void fuse_file_release(struct inode *inode, struct fuse_file *ff, * If this release dropped the last writer, fuse_prepare_release() * cleared the forced-direct-IO latch (under fi->lock). Drop any clean * folios a read racing the latch may have repopulated so they cannot be - * served stale once caching mode resumes. No inode lock or - * wb_inval_rwsem: release may run on the fuse server thread (async fput + * served stale once caching mode resumes. No inode lock: release may + * run on the fuse server thread (async fput * from aio completion), where blocking on a contended inode lock could * stall the connection. Writes were routed direct while latched, so * only clean folios exist and this invalidate is server-free; the last @@ -556,11 +581,15 @@ u64 fuse_lock_owner_id(struct fuse_conn *fc, fl_owner_t id) return (u64) v0 + ((u64) v1 << 32); } +struct fuse_wb_token; + struct fuse_writepage_args { struct fuse_io_args ia; struct list_head queue_entry; struct inode *inode; struct fuse_sync_bucket *bucket; + /* One per entry of ia.ap.folios, see struct fuse_wb_token */ + struct fuse_wb_token **tokens; }; /* @@ -1007,16 +1036,137 @@ static int fuse_do_readfolio(struct file *file, struct folio *folio, return 0; } +/** + * fuse_read_folio_range - read part of a folio from the server + * @file: file to read through + * @folio: the folio to fill + * @off: offset within @folio to start at + * @len: bytes to read + * + * fuse_do_readfolio() cannot serve a partial folio: it asks for + * page_zeroing, and fuse_copy_folio() answers that by zeroing the whole + * folio whenever the request covers less than all of it. Ask without it + * and zero exactly what the reply left short, which is the server saying + * the file ends there. + * + * Return: 0, AOP_TRUNCATED_PAGE, or a negative error. + */ +static int fuse_read_folio_range(struct file *file, struct folio *folio, + size_t off, size_t len) +{ + struct inode *inode = folio->mapping->host; + struct fuse_mount *fm = get_fuse_mount(inode); + loff_t pos = folio_pos(folio) + off; + struct fuse_folio_desc desc = { + .offset = off, + .length = len, + }; + struct fuse_io_args ia = { + .ap.args.out_pages = true, + .ap.num_folios = 1, + .ap.folios = &folio, + .ap.descs = &desc, + }; + ssize_t res; + + /* Don't overflow end offset */ + if (pos + (desc.length - 1) == LLONG_MAX) + desc.length--; + + fuse_read_args_fill(&ia, file, pos, desc.length, FUSE_READ); + res = fuse_simple_request(fm, &ia.ap.args); + if (res < 0) { + /* See fuse_do_readfolio() for why READ can return -EDEADLK */ + if ((res == -EDEADLK || res == -EAGAIN) && fm->fc->dlm) + res = AOP_TRUNCATED_PAGE; + return res; + } + + if (res < desc.length) + folio_zero_range(folio, off + res, desc.length - res); + + return 0; +} + +/** + * fuse_read_folio_merge - fill @folio without disturbing what it holds + * @file: file to read through + * @folio: the folio to fill + * + * iomap tracks a folio a block at a time, and a write that covered some + * of its blocks and not others leaves it valid in the ones it covered + * and not in the rest. The valid ones hold what that write put there, + * which writeback may not have sent yet; reading over them would lose + * it. Fetch the rest, in as few requests as the gaps allow. + * + * A folio the page cache tracks in one piece has no per block state to + * ask, and none to have: it is dirty only if a write covered it whole, + * and then it is valid and never reaches here. + * + * Return: 0, AOP_TRUNCATED_PAGE, or a negative error. + */ +static int fuse_read_folio_merge(struct file *file, struct folio *folio) +{ + size_t bsize = i_blocksize(folio->mapping->host); + size_t size = folio_size(folio); + size_t off = 0; + + while (off < size) { + size_t run = 0; + int err; + + /* Skip what the folio already holds */ + while (off < size && + iomap_is_partially_uptodate(folio, off, bsize)) + off += bsize; + + /* Take the gap behind it in one request */ + while (off + run < size && + !iomap_is_partially_uptodate(folio, off + run, bsize)) + run += bsize; + + if (!run) + break; + + err = fuse_read_folio_range(file, folio, off, run); + if (err) + return err; + off += run; + } + + return 0; +} + static int fuse_read_folio(struct file *file, struct folio *folio) { struct inode *inode = folio->mapping->host; + struct fuse_conn *fc = get_fuse_conn(inode); int err; err = -EIO; if (fuse_is_bad(inode)) goto out; - err = fuse_do_readfolio(file, folio, 0, folio_size(folio)); + /* + * Writeback unlocks a folio as soon as it has handed it over, with + * the writeback flag still on it, so this can be reached while a + * FUSE_WRITE is still reading out of it. Filling it now would + * rewrite what is being sent, and past the end of the file the reply + * comes back short and zeroes it. Nothing reached here before a + * partial write started leaving folios invalid, because a dirty + * folio was always valid and never came this way. + */ + folio_wait_writeback(folio); + + /* + * Only a folio still holding what a write put in it has anything to + * keep. One that was merely being written back is clean by now, + * waited out just above, and reads whole. + */ + if (fc->writeback_cache && folio_test_dirty(folio)) + err = fuse_read_folio_merge(file, folio); + else + err = fuse_do_readfolio(file, folio, 0, folio_size(folio)); if (!err) folio_mark_uptodate(folio); @@ -1036,7 +1186,7 @@ static int fuse_iomap_read_folio_range(const struct iomap_iter *iter, size_t off = offset_in_folio(folio, pos); int ret; - ret = fuse_do_readfolio(file, folio, off, len); + ret = fuse_read_folio_range(file, folio, off, len); /* * TEMPORARY WORKAROUND for iomap write deadlock: @@ -1154,6 +1304,32 @@ static void fuse_readahead(struct readahead_control *rac) if (fuse_is_bad(inode)) return; + /* + * Readahead fills the page cache past the range the reader locked, + * so take a DLM read grant over the whole window here too. Folios + * the server handed out no lock for are folios it will not revoke + * when a remote node writes them, and a later read would be served + * from stale cache. Take the grant before any folio is pulled off + * @rac, so the window is either fully covered or not populated. + * + * Speculative pages are not worth serving uncovered: on a failed + * request drop the window and let read_pages() clean up the folios + * left in @rac. A server without DLM support answers -ENOSYS and + * clears fc->dlm, which is not a failure. + * + * The round trip is taken before any folio of the window is locked + * and with nothing fenced out, so it holds up this reader and + * nothing else. + */ + if (fc->writeback_cache && fc->dlm) { + int err = fuse_get_dlm_lock(rac->file, readahead_pos(rac), + readahead_length(rac), + FUSE_PAGE_LOCK_READ); + + if (err < 0 && err != -ENOSYS) + return; + } + max_pages = min_t(unsigned int, fc->max_pages, fc->max_read / PAGE_SIZE); @@ -1227,21 +1403,12 @@ static void fuse_readahead(struct readahead_control *rac) static ssize_t fuse_direct_read_iter(struct kiocb *iocb, struct iov_iter *to); -/* - * Bound on re-requesting a revoked DLM grant before a cached read is - * served unlocked; see fuse_cache_read_iter(). - */ -#define FUSE_DLM_READ_RETRIES 3 - static ssize_t fuse_cache_read_iter(struct kiocb *iocb, struct iov_iter *to) { struct file *file = iocb->ki_filp; struct inode *inode = file->f_mapping->host; struct fuse_conn *fc = get_fuse_conn(inode); - struct fuse_inode *fi = get_fuse_inode(inode); - struct percpu_rw_semaphore *wb_sem = fi->wb_inval_rwsem; ssize_t res; - int lock_err = 0; /* * In auto invalidate mode, always update attributes on read. @@ -1259,65 +1426,35 @@ static ssize_t fuse_cache_read_iter(struct kiocb *iocb, struct iov_iter *to) /* if we have dlm support acquire a read lock for the area * we are reading from. */ if (fc->writeback_cache && fc->dlm) - lock_err = fuse_get_dlm_lock(file, iocb->ki_pos, - iov_iter_count(to), - FUSE_PAGE_LOCK_READ); + fuse_get_dlm_lock(file, iocb->ki_pos, iov_iter_count(to), + FUSE_PAGE_LOCK_READ); /* - * Fence the cache-serving read against a NOTIFY invalidate so we never - * hand back a folio the server has just superseded. The gate read side - * is per-CPU cheap; the NOTIFY holds the write side with priority. - * Re-check the forced-DIO latch under it: if a storm latched us while we - * waited on a pending writer, reroute to direct like the buffered write - * path, so we do not repopulate the cache the latch just dropped. - * wb_sem is NULL on non-writeback+dlm mounts (gate inactive). + * A NOTIFY invalidate racing this read drops the folios it + * supersedes, so the read either misses and refetches or returns + * data that was current when it was copied. There is nothing to + * fence: unlike a write, a read leaves nothing behind that could + * reach the server under a grant it no longer holds. */ - if (wb_sem) { - int tries = FUSE_DLM_READ_RETRIES; + if (fuse_inode_force_dio(inode)) { + size_t count = iov_iter_count(to); -retry: - percpu_down_read(wb_sem); - if (fuse_inode_force_dio(inode)) { - percpu_up_read(wb_sem); - return fuse_direct_read_iter(iocb, to); - } /* - * The DLM lock was requested before entering the gate, and - * the NOTIFY invalidate we may just have waited on revokes - * locks under the gate write side. Re-check the grant here - * and re-request with the gate dropped, so a - * FUSE_DLM_WB_LOCK round trip never parks a pending - * invalidate behind our own gate hold. Once the check - * passes the lock cannot go away for the rest of the gate - * hold. A failed or unrecorded request falls through - * unlocked, as before: the retry is taken even then (the - * latch must be re-checked under the re-entered gate), so - * lock_err has to stay sticky across it -- seeded by the - * pre-gate request above -- or a grant that failed would - * be re-requested forever. The retry is also bounded: a - * remote writer can revoke each successful grant before - * the gate is re-entered, and a reader-only inode has no - * force-DIO latch to end such a storm, so after - * FUSE_DLM_READ_RETRIES re-requests the read is served - * unlocked rather than looping without bound. + * A write that passed this same check just before the latch + * took hold dirtied the page cache after the notify dropped + * it, and a direct read does not look there. Send it first. */ - if (!lock_err && fc->dlm && tries-- > 0 && - !fuse_dlm_lock_is_held(fi, iocb->ki_pos, - iov_iter_count(to), - FUSE_PAGE_LOCK_READ)) { - percpu_up_read(wb_sem); - lock_err = fuse_get_dlm_lock(file, iocb->ki_pos, - iov_iter_count(to), - FUSE_PAGE_LOCK_READ); - goto retry; + if (count) { + res = filemap_write_and_wait_range(inode->i_mapping, + iocb->ki_pos, iocb->ki_pos + count - 1); + if (res) + return res; } + return fuse_direct_read_iter(iocb, to); } res = generic_file_read_iter(iocb, to); - if (wb_sem) - percpu_up_read(wb_sem); - return res; } @@ -1805,18 +1942,25 @@ static ssize_t fuse_dlm_write_chunk(struct kiocb *iocb, struct iov_iter *from, } /* - * Buffered write under DLM. A partial page dirtied for writeback would - * have to be completed by reading the untouched remainder back from the - * server, and for a write past the server EOF that READ can only return - * zero bytes: a wasted round trip per unaligned edge. So cache only the - * page-aligned interior, whole pages need no read-modify-write, and - * route the unaligned head and tail straight through to the server. The - * writethrough path writes just those bytes and leaves the page - * non-uptodate, doing no read, and each edge lands as an independent - * FUSE_WRITE carrying FUSE_WRITE_CACHE like the writeback it replaces, - * so writers sharing a boundary page accumulate their bytes on the - * server. Aligned writes take the interior path whole; a - * sub-page write with no aligned interior goes fully through. + * Buffered write under DLM. A partly written page dirtied for + * writeback would have to be completed by reading the untouched + * remainder back from the server, and for a write past the server EOF + * that READ can only return zero bytes: a wasted round trip per + * unaligned edge. So cache only the page-aligned interior, whole pages + * need no read-modify-write, and route the unaligned head and tail + * straight through to the server. The writethrough path writes just + * those bytes and leaves the page non-uptodate, doing no read, and each + * edge lands as an independent FUSE_WRITE carrying FUSE_WRITE_CACHE + * like the writeback it replaces, so writers sharing a boundary page + * accumulate their bytes on the server. Aligned writes take the + * interior path whole; a sub-page write with no aligned interior goes + * fully through. + * + * The page is the block here: iomap tracks a folio a block at a time + * and __iomap_write_begin() skips the fill only for a block the write + * covers whole, so the cut has to land on block bounds. A writeback + * connection is refused unless the two are the same size; see + * process_init_reply(). */ static ssize_t fuse_dlm_buffered_write(struct kiocb *iocb, struct iov_iter *from, @@ -1826,32 +1970,37 @@ static ssize_t fuse_dlm_buffered_write(struct kiocb *iocb, loff_t end = pos + iov_iter_count(from); loff_t mid_start = round_up(pos, PAGE_SIZE); loff_t mid_end = round_down(end, PAGE_SIZE); + /* Unaligned head, cached interior of whole pages, unaligned tail */ + const struct { + loff_t len; + bool through; + } chunk[] = { + { mid_start - pos, true }, + { mid_end - mid_start, false }, + { end - mid_end, true }, + }; ssize_t res, total = 0; + unsigned int i; /* No whole page inside the write: nothing cacheable, all through. */ if (mid_end <= mid_start) return fuse_perform_write(iocb, from, true); - /* Unaligned head [pos, mid_start): through. */ - res = fuse_dlm_write_chunk(iocb, from, file, mid_start - pos, true); - if (res < 0) - return res; - total += res; - if (res < mid_start - pos) - return total; - - /* Aligned interior [mid_start, mid_end): cached whole pages. */ - res = fuse_dlm_write_chunk(iocb, from, file, mid_end - mid_start, false); - if (res < 0) - return total; - total += res; - if (res < mid_end - mid_start) - return total; - - /* Unaligned tail [mid_end, end): through. */ - res = fuse_dlm_write_chunk(iocb, from, file, end - mid_end, true); - if (res > 0) + /* + * Every chunk reports a failure the same way: the error while + * nothing has landed, a short write once something has. Returning + * what has landed when that is nothing reports no error and no + * bytes, which fuse_cache_write_iter() turns into the full count. + */ + for (i = 0; i < ARRAY_SIZE(chunk); i++) { + res = fuse_dlm_write_chunk(iocb, from, file, chunk[i].len, + chunk[i].through); + if (res < 0) + return total ? total : res; total += res; + if (res < chunk[i].len) + break; + } return total; } @@ -1902,19 +2051,13 @@ static void fuse_cache_wr_unlock(struct inode *inode, bool exclusive) * fc->dlm: the server has no DLM, proceed as a plain cached write. Any * other failure means the cache would be dirtied without DLM coverage - * the caller must fail the write instead. A granted-but-unrecorded - * lock (positive return) is covered cluster-wide; proceed, but flag it - * so the in-gate re-validation skips a check an invisible grant could - * never pass. + * lock (positive return) is covered cluster-wide; proceed. */ -static int fuse_cache_wr_dlm_lock(struct file *file, loff_t pos, size_t len, - bool *unrecorded) +static int fuse_cache_wr_dlm_lock(struct file *file, loff_t pos, size_t len) { int err = fuse_get_dlm_lock(file, pos, len, FUSE_PAGE_LOCK_WRITE); - if (err < 0 && err != -ENOSYS) - return err; - *unrecorded = err > 0; - return 0; + return (err < 0 && err != -ENOSYS) ? err : 0; } static ssize_t fuse_cache_write_iter(struct kiocb *iocb, struct iov_iter *from) @@ -1927,13 +2070,8 @@ static ssize_t fuse_cache_write_iter(struct kiocb *iocb, struct iov_iter *from) ssize_t err, count; struct fuse_conn *fc = get_fuse_conn(inode); struct fuse_inode *fi = get_fuse_inode(inode); - struct percpu_rw_semaphore *wb_sem = fi->wb_inval_rwsem; bool writeback = false; - bool wb_guard = false; - bool exclusive = true; - bool dlm_unrecorded = false; - loff_t dlm_pos = 0; - size_t dlm_len = 0; + bool exclusive; if (fuse_inode_force_dio(inode)) return fuse_direct_write_iter(iocb, from); @@ -1977,98 +2115,64 @@ static ssize_t fuse_cache_write_iter(struct kiocb *iocb, struct iov_iter *from) writeback = true; } - exclusive = fuse_cache_wr_exclusive_lock(iocb, writeback); - /* * Request the DLM write lock before taking i_rwsem: the request is * an unbounded cluster round trip, and holding the writer-priority * rwsem across it would park a truncate -- and behind it every - * later writer -- for the duration. The grant-to-use window this - * leaves open is closed by the in-gate re-validation below. Only - * the append case must wait for the lock: its range depends on - * i_size, which is stable only under the exclusive inode lock. + * later writer -- for the duration. Only the append case must wait + * for the lock: its range depends on i_size, which is settled by + * generic_write_checks() under the exclusive inode lock. + * + * The request may find that the server has no DLM at all and clear + * fc->dlm, so pick the lock mode after it rather than before. The + * relaxed shared lock is only sound under DLM: the shared path + * claims the i_size extension up front, which stops iomap from + * zeroing beyond EOF, and the zero-fill that replaces it in + * fuse_iomap_read_folio_range() is itself gated on fc->dlm. Chosen + * too early, an expanding write would fall through to a READ of a + * range that cannot hold data -- which fails outright on a handle + * the client opened write-only. */ if (writeback && fc->dlm && !(iocb->ki_flags & IOCB_APPEND)) { - dlm_pos = iocb->ki_pos; - dlm_len = iov_iter_count(from); - - err = fuse_cache_wr_dlm_lock(file, dlm_pos, dlm_len, - &dlm_unrecorded); + err = fuse_cache_wr_dlm_lock(file, iocb->ki_pos, + iov_iter_count(from)); if (err) return err; - - /* - * The request above may have found that the server has no DLM - * at all, in which case it cleared fc->dlm. The relaxed shared - * lock was chosen just before, while fc->dlm still read 1, and - * it is only sound under DLM: the shared path claims the i_size - * extension up front, which stops iomap from zeroing beyond - * EOF, and the zero-fill that replaces it in - * fuse_iomap_read_folio_range() is itself gated on fc->dlm. - * Left as chosen, an expanding write would fall through to a - * READ of a range that cannot hold data -- which fails outright - * on a handle the client opened write-only. Re-decide now, - * while no lock is held yet. - */ - exclusive = fuse_cache_wr_exclusive_lock(iocb, writeback); } + exclusive = fuse_cache_wr_exclusive_lock(iocb, writeback); + if (exclusive) inode_lock(inode); else inode_lock_shared(inode); - /* note that this small code dup will save us a lot of headache later - * when appends are done concurrently without using parallel direct writes */ - if (writeback && fc->dlm && (iocb->ki_flags & IOCB_APPEND)) { - /* - * An append write lands at the current EOF no matter what - * ki_pos holds: generic_write_checks() rewrites ki_pos to - * i_size for IOCB_APPEND. Lock where the data will land. - */ - dlm_pos = i_size_read(inode); - dlm_len = iov_iter_count(from); - - err = fuse_cache_wr_dlm_lock(file, dlm_pos, dlm_len, - &dlm_unrecorded); - if (err) - goto out; - } - err = count = generic_write_checks(iocb, from); if (err <= 0) goto out; /* - * The exclusive inode lock does not pin i_size for the append: - * attribute replies move it under fi->lock alone, so - * generic_write_checks() may have put ki_pos past the granted - * range. Re-lock where the write really lands; dlm_pos tracks it - * so the in-gate re-validation below guards the same range. + * An append lands at the EOF generic_write_checks() has just written + * into ki_pos, not where the caller pointed, and the exclusive inode + * lock does not pin i_size either: attribute replies move it under + * fi->lock alone. Take the grant here, where the range is settled. */ - if (writeback && fc->dlm && (iocb->ki_flags & IOCB_APPEND) && - iocb->ki_pos != dlm_pos) { - dlm_pos = iocb->ki_pos; - dlm_len = count; - - err = fuse_cache_wr_dlm_lock(file, dlm_pos, dlm_len, - &dlm_unrecorded); + if (writeback && fc->dlm && (iocb->ki_flags & IOCB_APPEND)) { + err = fuse_cache_wr_dlm_lock(file, iocb->ki_pos, count); if (err) goto out; } /* - * Kill suid/sgid and stamp the timestamps here, before the gate, - * instead of leaving them next to the write itself. kiocb_modified() - * -> file_remove_privs() is the one that reaches the server: without - * handle_killpriv[_v2] fuse_setattr() kills the bits by asking it (a - * FUSE_GETATTR to refresh the mode, then a FUSE_SETATTR, which for a - * writeback inode first flushes and freezes writepages), and - * security_inode_killpriv() can drop the capability xattr with another - * round trip. A server may have to invalidate this inode from inside - * such a handler; its NOTIFY_INVAL_INODE then blocks in - * percpu_down_write() draining a gate reader that is itself waiting for - * the reply. Nothing held under the gate may wait for the server. + * Kill suid/sgid and stamp the timestamps here, ahead of the write + * itself. kiocb_modified() -> file_remove_privs() is the one that + * reaches the server: without handle_killpriv[_v2] fuse_setattr() + * kills the bits by asking it (a FUSE_GETATTR to refresh the mode, + * then a FUSE_SETATTR, which for a writeback inode first flushes and + * freezes writepages), and security_inode_killpriv() can drop the + * capability xattr with another round trip. A server may have to + * invalidate this inode from inside such a handler, and it must not + * find this write holding anything it needs. * * This also runs before the forced-DIO re-route below, so a re-routed * write repeats it; there is nothing left to do the second time. @@ -2077,31 +2181,32 @@ static ssize_t fuse_cache_write_iter(struct kiocb *iocb, struct iov_iter *from) if (err) goto out; - wb_guard = !!wb_sem; - if (wb_guard) { -retry: - percpu_down_read(wb_sem); - if (fuse_inode_force_dio(inode)) { - percpu_up_read(wb_sem); - fuse_cache_wr_unlock(inode, exclusive); - return fuse_direct_write_iter(iocb, from); - } - if (writeback && fc->dlm && !dlm_unrecorded && - !fuse_dlm_lock_is_held(fi, dlm_pos, dlm_len, - FUSE_PAGE_LOCK_WRITE)) { - percpu_up_read(wb_sem); - err = fuse_cache_wr_dlm_lock(file, dlm_pos, dlm_len, - &dlm_unrecorded); - if (err) { - /* The gate is already dropped; funnel the - * failure through the one audited exit. */ - wb_guard = false; - goto out; - } - goto retry; - } + if (fuse_inode_force_dio(inode)) { + /* + * As on the read side, only worse: the direct write would + * land under whatever a write racing the latch left dirty, + * and the invalidate fuse_direct_write_iter() does after it + * launders rather than drops, putting that folio on the + * server on top. Send it first and the order is ordinary. + */ + if (count) + err = filemap_write_and_wait_range(inode->i_mapping, + iocb->ki_pos, iocb->ki_pos + count - 1); + fuse_cache_wr_unlock(inode, exclusive); + if (err) + return err; + return fuse_direct_write_iter(iocb, from); } + /* + * A NOTIFY invalidate can revoke the grant requested above between + * here and the dirtying below, and nothing stops it: the bytes are + * caught on the way out instead. fuse_dlm_unlock_range() keeps a + * revoked range for as long as there is page cache under it, and + * writeback holds the range again before sending anything. So a + * write racing a revoke costs a round trip, not coverage. + */ + task_io_account_write(count); if (iocb->ki_flags & IOCB_DIRECT) { @@ -2146,7 +2251,13 @@ static ssize_t fuse_cache_write_iter(struct kiocb *iocb, struct iov_iter *from) } spin_unlock(&fi->lock); - /* Zero the tail of the folio straddling the old EOF. */ + /* + * Zero the tail of the folio straddling the old EOF. + * Inert while the fuse block size is PAGE_SIZE, which + * it always is, and nothing is recorded for it either + * way: claiming the whole gap as written would hand + * writeback bytes no one wrote. + */ if (extended && orig_size < pos) pagecache_isize_extended(inode, orig_size, pos); } @@ -2154,7 +2265,7 @@ static ssize_t fuse_cache_write_iter(struct kiocb *iocb, struct iov_iter *from) /* * Under DLM the unaligned edges go through to the server * instead of being completed by a read-modify-write READ - * (see fuse_dlm_buffered_write()); only whole pages are + * (see fuse_dlm_buffered_write()); only whole blocks are * cached for writeback. */ if (fc->dlm) @@ -2189,8 +2300,6 @@ static ssize_t fuse_cache_write_iter(struct kiocb *iocb, struct iov_iter *from) written = fuse_perform_write(iocb, from, false); } out: - if (wb_guard) - percpu_up_read(wb_sem); fuse_cache_wr_unlock(inode, exclusive); if (written > 0) written = generic_write_sync(iocb, written); @@ -2549,6 +2658,77 @@ static ssize_t fuse_splice_write(struct pipe_inode_info *pipe, struct file *out, return iter_file_splice_write(pipe, out, ppos, len, flags); } +/* + * A folio is written back one recorded run at a time, and a run that does + * not reach both folio edges forces a new request, so the runs of one folio + * end up in requests that complete independently. iomap counts a folio's + * outstanding writes in ifs->write_bytes_pending, but a folio of a single + * block carries no iomap_folio_state, and there iomap_finish_folio_write() + * ends the writeback on every call. Count the runs here instead and end + * the folio writeback once, on the last one. + */ +struct fuse_wb_token { + refcount_t refs; + struct inode *inode; + struct folio *folio; +}; + +/* + * Take @folio into writeback and open the count. The caller keeps the + * returned reference as a bias, so the count cannot reach zero while + * further runs of the same folio are still being queued. + */ +static struct fuse_wb_token *fuse_wb_token_alloc(struct inode *inode, + struct folio *folio) +{ + struct fuse_wb_token *token; + + /* As iomap allocates the state this stands in for */ + token = kmalloc(sizeof(*token), GFP_NOFS | __GFP_NOFAIL); + refcount_set(&token->refs, 1); + token->inode = inode; + token->folio = folio; + iomap_start_folio_write(inode, folio, 1); + + return token; +} + +static struct fuse_wb_token *fuse_wb_token_get(struct fuse_wb_token *token) +{ + refcount_inc(&token->refs); + return token; +} + +static void fuse_wb_token_put(struct fuse_wb_token *token) +{ + if (token && refcount_dec_and_test(&token->refs)) { + iomap_finish_folio_write(token->inode, token->folio, 1); + kfree(token); + } +} + +/* + * The folios, descs and tokens of a writeback request come from one + * allocation, which kfree(ap->folios) releases. + */ +static struct folio **fuse_wb_folios_alloc(unsigned int nfolios, gfp_t flags, + struct fuse_folio_desc **descs, + struct fuse_wb_token ***tokens) +{ + struct folio **folios; + + folios = kzalloc(nfolios * (sizeof(struct folio *) + + sizeof(struct fuse_folio_desc) + + sizeof(struct fuse_wb_token *)), flags); + if (!folios) + return NULL; + + *descs = (void *) (folios + nfolios); + *tokens = (void *) (*descs + nfolios); + + return folios; +} + static void fuse_writepage_free(struct fuse_writepage_args *wpa) { struct fuse_args_pages *ap = &wpa->ia.ap; @@ -2575,7 +2755,7 @@ static void fuse_writepage_finish(struct fuse_writepage_args *wpa) * scope of the fi->lock alleviates xarray lock * contention and noticeably improves performance. */ - iomap_finish_folio_write(inode, ap->folios[i], 1); + fuse_wb_token_put(wpa->tokens[i]); wake_up(&fi->page_waitq); } @@ -2723,7 +2903,8 @@ static struct fuse_writepage_args *fuse_writepage_args_alloc(void) if (wpa) { ap = &wpa->ia.ap; ap->num_folios = 0; - ap->folios = fuse_folios_alloc(1, GFP_NOFS, &ap->descs); + ap->folios = fuse_wb_folios_alloc(1, GFP_NOFS, &ap->descs, + &wpa->tokens); if (!ap->folios) { kfree(wpa); wpa = NULL; @@ -2748,13 +2929,15 @@ static void fuse_writepage_add_to_bucket(struct fuse_conn *fc, } static void fuse_writepage_args_page_fill(struct fuse_writepage_args *wpa, struct folio *folio, - uint32_t folio_index, loff_t offset, unsigned len) + uint32_t folio_index, loff_t offset, unsigned int len, + struct fuse_wb_token *token) { struct fuse_args_pages *ap = &wpa->ia.ap; ap->folios[folio_index] = folio; ap->descs[folio_index].offset = offset; ap->descs[folio_index].length = len; + wpa->tokens[folio_index] = fuse_wb_token_get(token); } static struct fuse_writepage_args *fuse_writepage_args_setup(struct folio *folio, @@ -2787,6 +2970,12 @@ struct fuse_fill_wb_data { struct fuse_writepage_args *wpa; struct fuse_file *ff; unsigned int max_folios; + /* + * The folio currently being split into runs, and the count that + * holds its writeback open until the last run has been queued. + */ + struct folio *wb_folio; + struct fuse_wb_token *wb_token; /* * nr_bytes won't overflow since fuse_writepage_need_send() caps * wb requests to never exceed fc->max_pages (which has an upper bound @@ -2801,21 +2990,25 @@ static bool fuse_pages_realloc(struct fuse_fill_wb_data *data, struct fuse_args_pages *ap = &data->wpa->ia.ap; struct folio **folios; struct fuse_folio_desc *descs; + struct fuse_wb_token **tokens; unsigned int nfolios = min_t(unsigned int, max_t(unsigned int, data->max_folios * 2, FUSE_DEFAULT_MAX_PAGES_PER_REQ), max_pages); WARN_ON(nfolios <= data->max_folios); - folios = fuse_folios_alloc(nfolios, GFP_NOFS, &descs); + folios = fuse_wb_folios_alloc(nfolios, GFP_NOFS, &descs, &tokens); if (!folios) return false; memcpy(folios, ap->folios, sizeof(struct folio *) * ap->num_folios); memcpy(descs, ap->descs, sizeof(struct fuse_folio_desc) * ap->num_folios); + memcpy(tokens, data->wpa->tokens, + sizeof(struct fuse_wb_token *) * ap->num_folios); kfree(ap->folios); ap->folios = folios; ap->descs = descs; + data->wpa->tokens = tokens; data->max_folios = nfolios; return true; @@ -2897,7 +3090,7 @@ static ssize_t fuse_iomap_writeback_range(struct iomap_writepage_ctx *wpc, struct inode *inode = wpc->inode; struct fuse_inode *fi = get_fuse_inode(inode); struct fuse_conn *fc = get_fuse_conn(inode); - loff_t offset = offset_in_folio(folio, pos); + loff_t offset; WARN_ON_ONCE(!data); @@ -2907,6 +3100,39 @@ static ssize_t fuse_iomap_writeback_range(struct iomap_writepage_ctx *wpc, return -EIO; } + /* + * A folio iomap has not asked about before: the one before it has all + * of its runs queued, so let go of the bias holding its count open. + */ + if (data->wb_folio != folio) { + fuse_wb_token_put(data->wb_token); + data->wb_token = NULL; + data->wb_folio = folio; + } + + /* + * Every run iomap reports dirty was written whole by this client: + * the unaligned edges of a cached write go to the server directly + * and the interior covers whole blocks, so nothing partly written + * is ever dirtied. There is nothing to classify, only the grant to + * make sure of: a revoke may have arrived since the write, and + * these bytes must not go out from under one. + * + * fuse_dlm_regrant_range() takes the range back when it has gone, + * and walks the record once under the lock held for read when it + * has not. A failure leaves the folio dirty, so the next writeback + * tries again; only a hard error stops it. + */ + if (fc->dlm && fc->writeback_cache) { + int err = fuse_dlm_regrant_range(data->ff, inode, pos, + pos + len - 1); + + if (err < 0 && err != -ENOSYS) + return err; + } + + offset = offset_in_folio(folio, pos); + if (wpa && fuse_writepage_need_send(fc, pos, len, ap, data, wpc->wbc)) { fuse_writepages_send(inode, data); data->wpa = NULL; @@ -2922,9 +3148,16 @@ static ssize_t fuse_iomap_writeback_range(struct iomap_writepage_ctx *wpc, ap = &wpa->ia.ap; } - iomap_start_folio_write(inode, folio, 1); + /* + * The first run of this folio that is actually sent takes it into + * writeback. A folio with no run at all never gets here, and iomap + * ends its writeback itself. + */ + if (!data->wb_token) + data->wb_token = fuse_wb_token_alloc(inode, folio); + fuse_writepage_args_page_fill(wpa, folio, ap->num_folios, - offset, len); + offset, len, data->wb_token); data->nr_bytes += len; ap->num_folios++; @@ -2941,6 +3174,11 @@ static int fuse_iomap_writeback_submit(struct iomap_writepage_ctx *wpc, WARN_ON_ONCE(!data); + /* No more runs are coming for the folio last seen */ + fuse_wb_token_put(data->wb_token); + data->wb_token = NULL; + data->wb_folio = NULL; + if (data->wpa) { WARN_ON(!data->wpa->ia.ap.num_folios); fuse_writepages_send(wpc->inode, data); @@ -3065,6 +3303,7 @@ static int fuse_get_page_mkwrite_lock(struct file *file, loff_t offset, size_t l fuse_abort_conn(fc); err = -EINVAL; } + return err; } /* @@ -3140,7 +3379,7 @@ static int fuse_file_mmap(struct file *file, struct vm_area_struct *vma) /* * If the inode was latched into forced direct IO after a remote-modify * notification, a mapping needs the page cache, so revert to caching - * mode. Revert without the inode lock or wb_inval_rwsem: ->mmap runs + * mode. Revert without the inode lock: ->mmap runs * under mmap_lock and the buffered write path holds both across a fault * on the user buffer (which takes mmap_lock), so taking either here * would invert lock order (ABBA). Clearing the latch and dropping the @@ -3984,23 +4223,6 @@ void fuse_init_file_inode(struct inode *inode, unsigned int flags) fi->iocachectr = 0; init_waitqueue_head(&fi->page_waitq); init_waitqueue_head(&fi->direct_io_waitq); - /* - * Coherency gate for the forced-direct-IO feature; only writeback+dlm - * regular files need it. A percpu_rw_semaphore embeds per-CPU state, - * so allocate it out of line and only when the mount can use it rather - * than paying it on every inode. On failure leave it NULL: the gate - * stays inactive (best-effort invalidate) and the inode is still usable. - */ - fi->wb_inval_rwsem = NULL; - if (fc->writeback_cache && fc->dlm) { - struct percpu_rw_semaphore *sem = kmalloc(sizeof(*sem), GFP_KERNEL); - - if (sem && percpu_init_rwsem(sem)) { - kfree(sem); - sem = NULL; - } - fi->wb_inval_rwsem = sem; - } fi->notify_stamp = jiffies; fi->notify_interval_ewma = FUSE_NOTIFY_EWMA_SEED << FUSE_NOTIFY_EWMA_SHIFT; diff --git a/fs/fuse/fuse_dlm_cache.c b/fs/fuse/fuse_dlm_cache.c index bc6dbae2d5aeb0..a953ae8f1b16ef 100644 --- a/fs/fuse/fuse_dlm_cache.c +++ b/fs/fuse/fuse_dlm_cache.c @@ -1,35 +1,75 @@ // SPDX-License-Identifier: GPL-2.0-only /* * FUSE page lock cache implementation + * + * cache->ranges records the grants the server has given this client, + * each with the mode it is held in. A grant still on the wire covers + * nothing and must not appear there, but a revoke has to be able to find + * it: otherwise a revoke processed before the grant is recorded removes + * nothing, and the grant recorded afterwards is never taken back. A + * request in flight therefore waits on cache->pending, where a revoke + * marks it killed and fuse_dlm_request_commit() drops the grant instead + * of recording it. + * + * Keeping requests off the tree leaves every tree walker looking at + * grants alone. + * + * The record says nothing about the page cache under a range. What is + * cached there, and whether the server has seen it, is what the page + * cache itself answers. A revoked grant is therefore forgotten, not + * kept: writeback holds the range again for every run it sends, and an + * absent record and a revoked one both make it ask. */ #include "fuse_i.h" #include "fuse_dlm_cache.h" #include +#include #include #include #include +/* + * How often to ask again for a grant a revoke killed while it was in + * flight, before giving up on the range. Each pass is a round trip. + */ +#define FUSE_DLM_GRANT_RETRIES 16 + /* A range of pages with a lock */ struct fuse_dlm_range { - /* Interval tree node */ + /* Interval tree node; only linked once granted */ struct rb_node rb; - /* Start page offset (inclusive) */ + /* + * The range, as byte offsets, both inclusive. Grants arrive page + * aligned, and a range is split only at the bounds of another, so + * these are page aligned too. + */ uint64_t start; - /* End page offset (inclusive) */ uint64_t end; /* Subtree end value for interval tree */ uint64_t __subtree_end; - /* Lock mode */ + /* The mode a grant in cache->ranges is held in */ enum fuse_page_lock_mode mode; - /* Temporary list entry for operations */ + /* A revoke overlapped this request in flight; cache->pending only */ + bool killed; + /* Temporary list entry for operations, and the cache->pending link */ struct list_head list; }; -/* Lock modes for FUSE page cache */ -#define FUSE_PCACHE_LK_READ 1 /* Shared read lock */ -#define FUSE_PCACHE_LK_WRITE 2 /* Exclusive write lock */ +/** + * fuse_dlm_mode_satisfies - is a grant in @held usable for @want + * @held: the mode a range in the tree is held in + * @want: the mode being asked for + * + * A WRITE grant is exclusive and so covers a READ request; a READ grant + * covers only READ. + */ +static inline bool fuse_dlm_mode_satisfies(enum fuse_page_lock_mode held, + enum fuse_page_lock_mode want) +{ + return held == want || held == FUSE_PAGE_LOCK_WRITE; +} /* Interval tree definitions for page ranges */ static inline uint64_t fuse_dlm_range_start(struct fuse_dlm_range *range) @@ -43,34 +83,122 @@ static inline uint64_t fuse_dlm_range_last(struct fuse_dlm_range *range) } INTERVAL_TREE_DEFINE(struct fuse_dlm_range, rb, uint64_t, __subtree_end, - fuse_dlm_range_start, fuse_dlm_range_last, static, - fuse_page_it); + fuse_dlm_range_start, fuse_dlm_range_last, static, + fuse_page_it); /** - * fuse_page_cache_init - Initialize a page cache lock manager - * @cache: The cache to initialize + * fuse_dlm_split_at - make @off start a range + * @cache: The page cache + * @off: byte offset to split at * - * Initialize a page cache lock manager for a FUSE inode. + * Splits the range containing @off in two, both halves keeping the mode + * of the original, so a revoke can apply to one side only. A no-op when + * @off already starts a range or falls in a gap. * - * Return: 0 on success, negative error code on failure + * Caller holds @cache->lock for write. + * + * Cannot fail: the split decides which bytes a caller goes on to name, + * and both naming more than was written and lowering more than was sent + * lose data. iomap allocates the state it keeps per folio the same way. */ -int fuse_dlm_cache_init(struct fuse_inode *inode) +static void fuse_dlm_split_at(struct fuse_dlm_cache *cache, uint64_t off) { - struct fuse_dlm_cache *cache = &inode->dlm_locked_areas; + struct fuse_dlm_range *range, *tail; - if (!cache) - return -EINVAL; + if (!off) + return; + + range = fuse_page_it_iter_first(&cache->ranges, off, off); + if (!range || range->start == off) + return; + + tail = kmalloc(sizeof(*tail), GFP_NOFS | __GFP_NOFAIL); + + *tail = *range; + INIT_LIST_HEAD(&tail->list); + tail->start = off; + + /* + * Bounds are never edited in place: the interval tree caches a + * subtree end that only insertion recomputes. + */ + fuse_page_it_remove(range, &cache->ranges); + range->end = off - 1; + fuse_page_it_insert(range, &cache->ranges); + fuse_page_it_insert(tail, &cache->ranges); +} + +/* + * Make @start and @end + 1 range bounds, so every range overlapping + * [start, end] lies wholly inside it and can be revoked or upgraded + * whole. Caller holds @cache->lock for write. + */ +static void fuse_dlm_split_bounds(struct fuse_dlm_cache *cache, uint64_t start, + uint64_t end) +{ + fuse_dlm_split_at(cache, start); + if (end < U64_MAX) + fuse_dlm_split_at(cache, end + 1); +} + +/* A grant record for [start, end] held in @mode, not yet in the tree */ +static struct fuse_dlm_range *fuse_dlm_range_new(uint64_t start, uint64_t end, + enum fuse_page_lock_mode mode) +{ + struct fuse_dlm_range *range = kmalloc(sizeof(*range), GFP_NOFS); + + if (!range) + return NULL; + + range->start = start; + range->end = end; + range->mode = mode; + INIT_LIST_HEAD(&range->list); + + return range; +} + +/** + * fuse_dlm_kill_pending - mark in-flight requests overlapping [start, end] + * @cache: The page cache + * @start: Start byte offset of the revoked region + * @end: End byte offset of the revoked region + * + * A revoke overlapping a request still on the wire has nothing to remove + * from the tree, since that grant is not recorded yet. Marking it makes + * fuse_dlm_request_commit() drop the grant instead of recording it. + * + * The nodes are owned by the threads waiting on their replies: mark + * only, never remove or free. + * + * Caller holds @cache->lock for write. + */ +static void fuse_dlm_kill_pending(struct fuse_dlm_cache *cache, + uint64_t start, uint64_t end) +{ + struct fuse_dlm_range *req; + + list_for_each_entry(req, &cache->pending, list) + if (req->start <= end && start <= req->end) + req->killed = true; +} + +/** + * fuse_dlm_cache_init - Initialize a page cache lock manager + * @inode: The fuse inode to initialize the cache of + */ +void fuse_dlm_cache_init(struct fuse_inode *inode) +{ + struct fuse_dlm_cache *cache = &inode->dlm_locked_areas; init_rwsem(&cache->lock); cache->ranges = RB_ROOT_CACHED; - cache->revoke_gen = 0; - - return 0; + INIT_LIST_HEAD(&cache->pending); } /** - * fuse_page_cache_destroy - Clean up a page cache lock manager - * @cache: The cache to clean up + * fuse_dlm_cache_release_locks - Clean up a page cache lock manager + * @inode: The fuse inode to clean up the cache of * * Release all locks and free all resources associated with the cache. */ @@ -80,12 +208,13 @@ void fuse_dlm_cache_release_locks(struct fuse_inode *inode) struct fuse_dlm_range *range; struct rb_node *node; - if (!cache) - return; - /* Release all locks */ down_write(&cache->lock); - WRITE_ONCE(cache->revoke_gen, cache->revoke_gen + 1); + /* + * Every grant goes, so every request in flight is revoked. Mark + * only; each node is owned by the thread waiting on its reply. + */ + fuse_dlm_kill_pending(cache, 0, U64_MAX); while ((node = rb_first_cached(&cache->ranges)) != NULL) { range = rb_entry(node, struct fuse_dlm_range, rb); fuse_page_it_remove(range, &cache->ranges); @@ -95,25 +224,10 @@ void fuse_dlm_cache_release_locks(struct fuse_inode *inode) } /** - * fuse_dlm_find_overlapping - Find a range that overlaps with [start, end] + * fuse_dlm_try_merge - Try to merge ranges within a specific region * @cache: The page cache - * @start: Start page offset - * @end: End page offset - * - * Return: Pointer to the first overlapping range, or NULL if none found - */ -static struct fuse_dlm_range * -fuse_dlm_find_overlapping(struct fuse_dlm_cache *cache, uint64_t start, - uint64_t end) -{ - return fuse_page_it_iter_first(&cache->ranges, start, end); -} - -/** - * fuse_page_try_merge - Try to merge ranges within a specific region - * @cache: The page cache - * @start: Start page offset - * @end: End page offset + * @start: Start byte offset + * @end: End byte offset * * Attempt to merge ranges within and adjacent to the specified region * that have the same lock mode. @@ -125,9 +239,6 @@ static void fuse_dlm_try_merge(struct fuse_dlm_cache *cache, uint64_t start, uint64_t first = start ? start - 1 : start; uint64_t last = end < U64_MAX ? end + 1 : end; - if (!cache) - return; - /* * Find the first range that might need merging. Directly adjacent * ranges can merge, hence the region is widened by one unit to each @@ -148,7 +259,7 @@ static void fuse_dlm_try_merge(struct fuse_dlm_cache *cache, uint64_t start, struct fuse_dlm_range, rb); } - /* Try to merge with next range if adjacent and same mode */ + /* Merge neighbours the server has given us on the same terms */ if (next && range->mode == next->mode && range->end + 1 == next->start) { /* Merge ranges: re-insert so __subtree_end is updated */ @@ -168,13 +279,11 @@ static void fuse_dlm_try_merge(struct fuse_dlm_cache *cache, uint64_t start, } /** - * __fuse_dlm_lock_range - Lock a range of pages - * @cache: The page cache - * @start: Start page offset - * @end: End page offset + * fuse_dlm_lock_range_locked - Record a granted range of pages + * @inode: The fuse inode + * @start: Start byte offset + * @end: End byte offset * @mode: Lock mode (read or write) - * @genp: If non-NULL, the revocation generation sampled before the grant - * was requested; recording fails with -EAGAIN if it has moved * * Add a locked range on the specified range of pages. * If parts of the range are already locked, only add the remaining parts. @@ -183,40 +292,33 @@ static void fuse_dlm_try_merge(struct fuse_dlm_cache *cache, uint64_t start, * - READ locks are compatible with existing WRITE locks (downgrade not needed) * - WRITE locks need to upgrade existing READ locks * + * Caller holds the cache lock for write. + * * Return: 0 on success, negative error code on failure */ -static int __fuse_dlm_lock_range(struct fuse_inode *inode, uint64_t start, - uint64_t end, enum fuse_page_lock_mode mode, - const uint64_t *genp) +static int fuse_dlm_lock_range_locked(struct fuse_inode *inode, uint64_t start, + uint64_t end, + enum fuse_page_lock_mode mode) { struct fuse_dlm_cache *cache = &inode->dlm_locked_areas; struct fuse_dlm_range *range, *new_range, *next; - int lock_mode; bool covered_to_end = false; int ret = 0; LIST_HEAD(to_lock); LIST_HEAD(to_upgrade); uint64_t current_start = start; - if (!cache || start > end) + if (start > end) return -EINVAL; - /* Convert to lock mode */ - lock_mode = (mode == FUSE_PAGE_LOCK_READ) ? FUSE_PCACHE_LK_READ : - FUSE_PCACHE_LK_WRITE; - - down_write(&cache->lock); - /* - * A revoke was processed after @genp was sampled; the grant this - * record carries may be the very one it targeted (a revoke of a - * not-yet-recorded grant removes nothing and would never be - * retried). Refuse, the caller re-requests. + * Ranges are upgraded whole below, so split at the grant bounds + * first: a range extending past the grant would otherwise have its + * uncovered part upgraded with it, recording coverage the server + * never gave, and a read range half-covered by a write grant would + * report the other half held for write. */ - if (genp && cache->revoke_gen != *genp) { - up_write(&cache->lock); - return -EAGAIN; - } + fuse_dlm_split_bounds(cache, start, end); /* Find all ranges that overlap with [start, end] */ range = fuse_page_it_iter_first(&cache->ranges, start, end); @@ -224,27 +326,19 @@ static int __fuse_dlm_lock_range(struct fuse_inode *inode, uint64_t start, /* Get next overlapping range before we potentially modify the tree */ next = fuse_page_it_iter_next(range, start, end); - /* Check lock compatibility */ - if (lock_mode == FUSE_PCACHE_LK_WRITE && - lock_mode != range->mode) { - /* we own the lock but have to update it. */ + /* A read range needs upgrading when a write is granted */ + if (!fuse_dlm_mode_satisfies(range->mode, mode)) list_add_tail(&range->list, &to_upgrade); - } - /* If WRITE lock already exists - nothing to do */ /* If there's a gap before this range, we need to add the missing range */ if (current_start < range->start) { - new_range = kmalloc(sizeof(*new_range), GFP_KERNEL); + new_range = fuse_dlm_range_new(current_start, + range->start - 1, mode); if (!new_range) { ret = -ENOMEM; goto out_free; } - new_range->start = current_start; - new_range->end = range->start - 1; - new_range->mode = lock_mode; - INIT_LIST_HEAD(&new_range->list); - list_add_tail(&new_range->list, &to_lock); } @@ -260,25 +354,18 @@ static int __fuse_dlm_lock_range(struct fuse_inode *inode, uint64_t start, /* If there's a gap after the last range to the end, extend the range */ if (!covered_to_end && current_start <= end) { - new_range = kmalloc(sizeof(*new_range), GFP_KERNEL); + new_range = fuse_dlm_range_new(current_start, end, mode); if (!new_range) { ret = -ENOMEM; goto out_free; } - new_range->start = current_start; - new_range->end = end; - new_range->mode = lock_mode; - INIT_LIST_HEAD(&new_range->list); - list_add_tail(&new_range->list, &to_lock); } - /* update locks, if any lock is in this list it has the wrong mode */ - list_for_each_entry(range, &to_upgrade, list) { - /* Update the lock mode */ - range->mode = lock_mode; - } + /* Everything on this list is now covered in @mode */ + list_for_each_entry(range, &to_upgrade, list) + range->mode = mode; /* Add all new ranges to the tree */ list_for_each_entry(new_range, &to_lock, list) { @@ -289,7 +376,6 @@ static int __fuse_dlm_lock_range(struct fuse_inode *inode, uint64_t start, /* Try to merge adjacent ranges with the same mode */ fuse_dlm_try_merge(cache, start, end); - up_write(&cache->lock); return 0; out_free: @@ -301,296 +387,207 @@ static int __fuse_dlm_lock_range(struct fuse_inode *inode, uint64_t start, kfree(new_range); } - /* Restore original lock modes for any partially upgraded locks */ - list_for_each_entry(range, &to_upgrade, list) { - if (lock_mode == FUSE_PCACHE_LK_WRITE) { - /* We upgraded this lock but failed later, downgrade it back */ - range->mode = FUSE_PCACHE_LK_READ; - } - } - - up_write(&cache->lock); + /* + * Nothing to undo on @to_upgrade: every goto here is taken before + * the loop above runs, so no state has been changed yet. + */ return ret; } -int fuse_dlm_lock_range(struct fuse_inode *inode, uint64_t start, - uint64_t end, enum fuse_page_lock_mode mode) -{ - return __fuse_dlm_lock_range(inode, start, end, mode, NULL); -} - -int fuse_dlm_lock_range_gen(struct fuse_inode *inode, uint64_t start, - uint64_t end, enum fuse_page_lock_mode mode, - uint64_t gen) -{ - return __fuse_dlm_lock_range(inode, start, end, mode, &gen); -} - /** - * fuse_dlm_revoke_gen - sample the revocation generation + * fuse_dlm_request_begin - publish a lock request before it is sent * @inode: the fuse inode + * @req: caller-owned storage for the request, live until commit or abort + * @start: start byte offset being requested (inclusive) + * @end: end byte offset being requested (inclusive) * - * Sampled before a FUSE_DLM_WB_LOCK request leaves the client. The - * reply and a NOTIFY revoke can be serviced on different threads, so a - * revoke may be processed between the reply arriving and its grant - * being recorded. fuse_dlm_lock_range_gen() re-checks the generation - * under the cache lock and refuses to record a grant such a revoke may - * have already killed. + * The mode is not recorded here: until the server answers the range is + * held in neither, and the mode that reaches the tree is the one passed + * to fuse_dlm_request_commit(). + * + * A FUSE_DLM_WB_LOCK reply and a NOTIFY revoke are serviced on different + * threads, so a revoke can be processed before the grant the reply + * carries is recorded. Publishing the request before it leaves gives + * that revoke a node to mark; without one it removes nothing, and the + * grant recorded afterwards is never taken back by any later NOTIFY. + * + * The request covers nothing while in flight, so it is kept off the + * tree. @req is reachable only through cache->pending, which both + * fuse_dlm_request_commit() and fuse_dlm_request_abort() unlink under + * the cache lock before the caller returns; stack storage is therefore + * fine and nothing is allocated here. */ -uint64_t fuse_dlm_revoke_gen(struct fuse_inode *inode) +void fuse_dlm_request_begin(struct fuse_inode *inode, + struct fuse_dlm_range *req, uint64_t start, + uint64_t end) { - return READ_ONCE(inode->dlm_locked_areas.revoke_gen); + struct fuse_dlm_cache *cache = &inode->dlm_locked_areas; + + RB_CLEAR_NODE(&req->rb); + req->start = start; + req->end = end; + req->killed = false; + /* Nothing reads the mode while the request is pending */ + + down_write(&cache->lock); + list_add_tail(&req->list, &cache->pending); + up_write(&cache->lock); } /** - * fuse_dlm_punch_hole - Punch a hole in a locked range - * @cache: The page cache - * @start: Start page offset of the hole - * @end: End page offset of the hole + * fuse_dlm_request_commit - retire a request and record its grant + * @inode: the fuse inode + * @req: the request published by fuse_dlm_request_begin() + * @start: start byte offset the server granted (inclusive) + * @end: end byte offset the server granted (inclusive) + * @mode: the mode that was requested * - * Create a hole in a locked range by splitting it into two ranges. + * Unlinking @req and recording the grant are one step under the cache + * lock, so a revoke lands either before it and is seen on @req, or after + * it and finds the grant in the tree. * - * Return: 0 on success, negative error code on failure + * @req is retired in every case and may be reused. + * + * Return: -EAGAIN if a revoke overlapped @req while it was in flight, + * nothing recorded; otherwise the result of recording the grant. */ -static int fuse_dlm_punch_hole(struct fuse_dlm_cache *cache, uint64_t start, - uint64_t end) +int fuse_dlm_request_commit(struct fuse_inode *inode, + struct fuse_dlm_range *req, uint64_t start, + uint64_t end, enum fuse_page_lock_mode mode) { - struct fuse_dlm_range *range, *new_range; + struct fuse_dlm_cache *cache = &inode->dlm_locked_areas; + bool revoked; int ret = 0; - if (!cache || start > end) - return -EINVAL; - - /* Find a range that contains [start, end] */ - range = fuse_dlm_find_overlapping(cache, start, end); - if (!range) { - ret = -EINVAL; - goto out; - } - - /* If the hole is at the beginning of the range */ - if (start == range->start) { - fuse_page_it_remove(range, &cache->ranges); - range->start = end + 1; - fuse_page_it_insert(range, &cache->ranges); - goto out; - } - - /* If the hole is at the end of the range */ - if (end == range->end) { - fuse_page_it_remove(range, &cache->ranges); - range->end = start - 1; - fuse_page_it_insert(range, &cache->ranges); - goto out; - } - - /* The hole is in the middle, need to split */ - new_range = kmalloc(sizeof(*new_range), GFP_KERNEL); - if (!new_range) { - ret = -ENOMEM; - goto out; - } - - /* Copy properties from original range */ - *new_range = *range; - INIT_LIST_HEAD(&new_range->list); + down_write(&cache->lock); + list_del(&req->list); + revoked = req->killed; + if (!revoked) + ret = fuse_dlm_lock_range_locked(inode, start, end, mode); + up_write(&cache->lock); - /* Adjust ranges */ - new_range->start = end + 1; - range->end = start - 1; + return revoked ? -EAGAIN : ret; +} - /* Update interval tree */ - fuse_page_it_remove(range, &cache->ranges); - fuse_page_it_insert(range, &cache->ranges); - fuse_page_it_insert(new_range, &cache->ranges); +/** + * fuse_dlm_request_abort - retire a request that got no usable reply + * @inode: the fuse inode + * @req: the request published by fuse_dlm_request_begin() + * + * Nothing is recorded, so a mark left by a revoke does not matter. + */ +void fuse_dlm_request_abort(struct fuse_inode *inode, + struct fuse_dlm_range *req) +{ + struct fuse_dlm_cache *cache = &inode->dlm_locked_areas; -out: - return ret; + down_write(&cache->lock); + list_del(&req->list); + up_write(&cache->lock); } /** - * fuse_dlm_unlock_range - Unlock a range of pages - * @cache: The page cache - * @start: Start page offset - * @end: End page offset + * fuse_dlm_unlock_range - Revoke the grants over a range of pages + * @inode: The fuse inode + * @start: Start byte offset + * @end: End byte offset * - * Release locks on the specified range of pages. An inverted range is - * rejected rather than silently removing nothing: the callers revoke - * coverage, and a revoke that quietly keeps the grant alive would let - * the re-validating IO paths trust a lock the server has taken away. - * To drop every grant use fuse_dlm_cache_release_locks() (there is no - * in-band sentinel range for it). + * The server has taken [start, end] back, so the grants over it are + * removed and the IO paths ask again. Page cache dirtied under a grant + * that has gone is not lost by this: writeback takes the range again for + * every run it sends, and a range it finds unrecorded is a range it asks + * for. + * + * An inverted range is rejected rather than silently revoking nothing: + * the callers revoke coverage, and a revoke that quietly keeps the grant + * alive would let the re-validating IO paths trust a lock the server has + * taken away. To drop every grant use fuse_dlm_cache_release_locks() + * (there is no in-band sentinel range for it). * * Return: 0 on success, negative error code on failure */ -int fuse_dlm_unlock_range(struct fuse_inode *inode, - uint64_t start, uint64_t end) +int fuse_dlm_unlock_range(struct fuse_inode *inode, uint64_t start, + uint64_t end) { struct fuse_dlm_cache *cache = &inode->dlm_locked_areas; struct fuse_dlm_range *range, *next; - int ret = 0; - if (!cache || start > end) + if (start > end) return -EINVAL; down_write(&cache->lock); /* - * Unconditional, even when nothing overlaps: the revoke racing - * with an in-flight grant finds an empty tree precisely because - * the grant is not recorded yet, and the bump is what makes the - * recording side notice (see fuse_dlm_lock_range_gen()). + * Before touching the tree, and even when nothing in the tree + * overlaps: a revoke racing an in-flight grant finds no overlap + * because that grant is not recorded yet. */ - WRITE_ONCE(cache->revoke_gen, cache->revoke_gen + 1); + fuse_dlm_kill_pending(cache, start, end); + + /* Split so the revoked region has its own ranges */ + fuse_dlm_split_bounds(cache, start, end); - /* Find all ranges that overlap with [start, end] */ range = fuse_page_it_iter_first(&cache->ranges, start, end); while (range) { - /* Get next overlapping range before we potentially modify the tree */ + /* Get next overlapping range before we modify the tree */ next = fuse_page_it_iter_next(range, start, end); - /* Check if we need to punch a hole */ - if (start > range->start && end < range->end) { - /* Punch a hole in the middle */ - ret = fuse_dlm_punch_hole(cache, start, end); - if (ret) - goto out; - /* After punching a hole, we're done */ - break; - } else if (start > range->start) { - /* Adjust the end of the range */ - fuse_page_it_remove(range, &cache->ranges); - range->end = start - 1; - fuse_page_it_insert(range, &cache->ranges); - } else if (end < range->end) { - /* Adjust the start of the range */ - fuse_page_it_remove(range, &cache->ranges); - range->start = end + 1; - fuse_page_it_insert(range, &cache->ranges); - } else { - /* Complete overlap, remove the range */ - fuse_page_it_remove(range, &cache->ranges); - kfree(range); - } + fuse_page_it_remove(range, &cache->ranges); + kfree(range); range = next; } -out: up_write(&cache->lock); - return ret; + return 0; } /** - * fuse_dlm_range_is_locked - Check if a page range is already locked - * @cache: The page cache - * @start: Start page offset - * @end: End page offset - * @mode: Lock mode to check for (or NULL to check for any lock) - * - * Check if the specified range of pages is already locked. - * The entire range must be locked for this to return true. + * fuse_dlm_range_is_locked - Check if a byte range is already locked + * @inode: The fuse inode + * @start: Start byte offset + * @end: End byte offset + * @mode: Lock mode to check for * * Return: true if the entire range is locked, false otherwise */ -bool fuse_dlm_range_is_locked(struct fuse_inode *inode, uint64_t start, - uint64_t end, enum fuse_page_lock_mode mode) +static bool fuse_dlm_range_is_locked(struct fuse_inode *inode, uint64_t start, + uint64_t end, + enum fuse_page_lock_mode mode) { struct fuse_dlm_cache *cache = &inode->dlm_locked_areas; struct fuse_dlm_range *range; - int lock_mode = 0; uint64_t current_start = start; + bool covered = false; - if (!cache || start > end) + if (start > end) return false; - /* Convert to lock mode if specified */ - if (mode == FUSE_PAGE_LOCK_READ) - lock_mode = FUSE_PCACHE_LK_READ; - else if (mode == FUSE_PAGE_LOCK_WRITE) - lock_mode = FUSE_PCACHE_LK_WRITE; - down_read(&cache->lock); - /* Find the first range that overlaps with [start, end] */ - range = fuse_dlm_find_overlapping(cache, start, end); - - /* Check if the entire range is covered */ - while (range && current_start <= end) { + for (range = fuse_page_it_iter_first(&cache->ranges, start, end); range; + range = fuse_page_it_iter_next(range, start, end)) { /* - * The held lock must be at least as strong as the one - * requested. A WRITE lock (exclusive) satisfies a READ - * request, so only treat the range as uncovered when the - * held mode is weaker than what we ask for. This avoids - * re-requesting a READ lock for a range we already hold - * a WRITE lock on (e.g. read-after-write). + * A gap before this range leaves the request uncovered, and + * so does a range held less strongly than it asks for. A + * WRITE grant is exclusive and does cover a READ request, + * which keeps a read-after-write from asking again. */ - if (lock_mode && range->mode < lock_mode) { - /* Held lock is weaker than requested */ - up_read(&cache->lock); - return false; - } - - /* Check if there's a gap before this range */ - if (current_start < range->start) { - /* Found a gap */ - up_read(&cache->lock); - return false; - } + if (current_start < range->start || + !fuse_dlm_mode_satisfies(range->mode, mode)) + break; - /* Covered through the end of the requested range? */ if (range->end >= end) { - up_read(&cache->lock); - return true; + covered = true; + break; } - /* Move current_start past this range */ current_start = range->end + 1; - - /* Get next overlapping range */ - range = fuse_page_it_iter_next(range, start, end); - } - - /* Check if we covered the entire range */ - if (current_start <= end) { - /* There's a gap at the end */ - up_read(&cache->lock); - return false; } up_read(&cache->lock); - return true; -} - -/** - * fuse_dlm_write_grant_exists - does the inode hold an exclusive grant anywhere - * @fi: the fuse inode - * - * Unlike fuse_dlm_range_is_locked(), which asks whether one range is fully - * covered, this asks whether any part of the file is held exclusively. A - * client that holds a write grant may be sitting on dirty page cache the - * server has not seen, so its mtime and ctime run ahead of anything the - * server can report. - * - * Return: true if at least one recorded range is held for write - */ -bool fuse_dlm_write_grant_exists(struct fuse_inode *fi) -{ - struct fuse_dlm_cache *cache = &fi->dlm_locked_areas; - struct fuse_dlm_range *range; - bool held = false; - down_read(&cache->lock); - for (range = fuse_dlm_find_overlapping(cache, 0, U64_MAX); range; - range = fuse_page_it_iter_next(range, 0, U64_MAX)) { - if (range->mode == FUSE_PCACHE_LK_WRITE) { - held = true; - break; - } - } - up_read(&cache->lock); - - return held; + return covered; } /** @@ -634,11 +631,10 @@ bool fuse_dlm_lock_is_held(struct fuse_inode *fi, loff_t offset, * re-validating the grant must not re-request on a nonzero return or * they would spin. */ -int fuse_get_dlm_lock(struct file *file, loff_t offset, - size_t length, enum fuse_page_lock_mode mode) +static int __fuse_get_dlm_lock(struct fuse_file *ff, struct inode *inode, + loff_t offset, size_t length, + enum fuse_page_lock_mode mode) { - struct fuse_file *ff = file->private_data; - struct inode *inode = file_inode(file); struct fuse_conn *fc = get_fuse_conn(inode); struct fuse_inode *fi = get_fuse_inode(inode); struct fuse_mount *fm = ff->fm; @@ -646,13 +642,23 @@ int fuse_get_dlm_lock(struct file *file, loff_t offset, FUSE_ARGS(args); struct fuse_dlm_lock_in inarg; struct fuse_dlm_lock_out outarg; - uint64_t gen; + struct fuse_dlm_range req; + uint64_t pg_start, pg_end; + int tries = FUSE_DLM_GRANT_RETRIES; int err; /* An empty range needs no lock. */ if (!length) return 0; + /* + * note that the offset and length don't have to be page aligned + * here but since we only get here on writeback caching we will + * send out page aligned requests + */ + pg_start = (uint64_t)offset & PAGE_MASK; + pg_end = ((uint64_t)offset + length - 1) | (PAGE_SIZE - 1); + restart: /* note that this can be run from different processes * at the same time. It is intentionally not protected @@ -661,27 +667,19 @@ int fuse_get_dlm_lock(struct file *file, loff_t offset, * The early exit uses the same helper the callers re-validate * with, so this check and a later fuse_dlm_lock_is_held() can * never disagree about what counts as covered. */ - if (fuse_dlm_lock_is_held(fi, offset, length, mode)) - return 0; /* we already have this area locked */ - - /* - * Sample the revocation generation before the request leaves. - * The reply and a NOTIFY revoke are serviced on different - * threads, so a revoke aimed at the grant this request returns - * can be processed before the grant is recorded below -- - * recording it anyway would resurrect a dead grant that no later - * NOTIFY will ever remove. - */ - gen = fuse_dlm_revoke_gen(fi); + if (fuse_dlm_lock_is_held(fi, offset, length, mode)) { + /* + * Already covered, and the record says nothing beyond that, + * so this is one shared acquisition end to end. + */ + return 0; + } memset(&inarg, 0, sizeof(inarg)); inarg.fh = ff->fh; - /* note that the offset and length don't have to be page aligned - * here but since we only get here on writeback caching we will - * send out page aligned requests */ - inarg.start = offset & PAGE_MASK; - inarg.end = (offset + length - 1) | (PAGE_SIZE - 1); + inarg.start = pg_start; + inarg.end = pg_end; inarg.type = (mode == FUSE_PAGE_LOCK_WRITE) ? FUSE_DLM_LOCK_WRITE : FUSE_DLM_LOCK_READ; @@ -693,40 +691,48 @@ int fuse_get_dlm_lock(struct file *file, loff_t offset, args.out_numargs = 1; args.out_args[0].size = sizeof(outarg); args.out_args[0].value = &outarg; + + /* Publish before sending; see fuse_dlm_request_begin() */ + fuse_dlm_request_begin(fi, &req, inarg.start, inarg.end); + err = fuse_simple_request(fm, &args); - if (err == -ENOSYS) { - /* fuse server does not support dlm, save the info */ - fc->dlm = 0; + if (err) { + fuse_dlm_request_abort(fi, &req); + if (err == -ENOSYS) { + /* fuse server does not support dlm, save the info */ + fc->dlm = 0; + } return err; } - if (err) - return err; - if (inarg.start < outarg.start || inarg.end > outarg.end) { /* fuse server is seriously broken */ + fuse_dlm_request_abort(fi, &req); pr_warn("fuse: dlm lock request for %llu:%llu returned %llu:%llu bytes\n", inarg.start, inarg.end, outarg.start, outarg.end); fuse_abort_conn(fc); return -EIO; } - /* - * The server granted the lock; record it so - * fuse_dlm_lock_is_held() sees it. - */ - err = fuse_dlm_lock_range_gen(fi, outarg.start, outarg.end, mode, gen); + /* Retire the request and record the grant */ + err = fuse_dlm_request_commit(fi, &req, outarg.start, outarg.end, mode); if (err == -EAGAIN) { /* - * A revoke was processed while the request was in flight; - * the grant may already be dead, so re-request instead of - * recording it. Retry until a grant survives long enough to - * be recorded: giving up here would hand the caller an error - * for a range no one else holds, and the write path turns - * that into a failed write. Each pass makes a fresh server - * round trip, so a revoke storm throttles this loop rather - * than spinning it. + * A revoke overlapping this range was processed while the + * request was in flight, so the grant is dead. Retry + * rather than fail: no one else holds the range, and the + * write path turns an error into a failed write. + * + * Not forever, though. Every pass is a whole round trip, + * which throttles the loop but does not end it, and + * writeback asks for a grant with a folio locked, so a node + * revoking as fast as the grants arrive would hold that + * folio and this task for as long as it kept going. */ + if (fatal_signal_pending(current)) + return -EINTR; + if (!tries--) + return -EIO; goto restart; } @@ -743,3 +749,32 @@ int fuse_get_dlm_lock(struct file *file, loff_t offset, return 0; } + +int fuse_get_dlm_lock(struct file *file, loff_t offset, + size_t length, enum fuse_page_lock_mode mode) +{ + return __fuse_get_dlm_lock(file->private_data, file_inode(file), + offset, length, mode); +} + +/** + * fuse_dlm_regrant_range - hold [start, end] again for writeback + * @ff: a fuse file open for writing on @inode + * @inode: the inode + * @start: start byte offset (inclusive) + * @end: end byte offset (inclusive) + * + * Writeback holds the range again before sending a folio, since a revoke + * may have arrived between the write and the send. Whatever the other + * holder wrote in between is overwritten, which for two writers that + * never synchronised is a legitimate order. + * + * A range still held is the ordinary case: the grant is found recorded + * and nothing is sent to the server. + */ +int fuse_dlm_regrant_range(struct fuse_file *ff, struct inode *inode, + uint64_t start, uint64_t end) +{ + return __fuse_get_dlm_lock(ff, inode, start, end - start + 1, + FUSE_PAGE_LOCK_WRITE); +} diff --git a/fs/fuse/fuse_dlm_cache.h b/fs/fuse/fuse_dlm_cache.h index 30fdbb26bd3daf..98cc834d76521e 100644 --- a/fs/fuse/fuse_dlm_cache.h +++ b/fs/fuse/fuse_dlm_cache.h @@ -13,6 +13,8 @@ struct fuse_inode; +struct fuse_dlm_range; +struct fuse_file; /* Lock modes for page ranges */ enum fuse_page_lock_mode { FUSE_PAGE_LOCK_READ, FUSE_PAGE_LOCK_WRITE }; @@ -26,53 +28,67 @@ enum fuse_page_lock_mode { FUSE_PAGE_LOCK_READ, FUSE_PAGE_LOCK_WRITE }; */ #define FUSE_DLM_GRANT_UNRECORDED 1 -/* Page cache lock manager */ +/* + * Page cache lock manager. + * + * @ranges holds the grants the client has been given. A request still on + * the wire covers nothing and lives on @pending instead, so tree walkers + * see grants only. See struct fuse_dlm_range in fuse_dlm_cache.c. + */ struct fuse_dlm_cache { - /* Lock protecting the tree */ + /* Lock protecting the tree and the pending list */ struct rw_semaphore lock; - /* Interval tree of locked ranges */ + /* Interval tree of the grants held */ struct rb_root_cached ranges; /* - * Bumped under @lock by every revocation - * (fuse_dlm_unlock_range(), fuse_dlm_cache_release_locks()); - * lets fuse_get_dlm_lock() order recording a reply's grant - * against revokes processed while the reply was in flight. + * FUSE_DLM_WB_LOCK requests in flight. Owned by the queueing + * thread; the revoke paths only mark them killed. */ - uint64_t revoke_gen; + struct list_head pending; }; /* Initialize a page cache lock manager */ -int fuse_dlm_cache_init(struct fuse_inode *inode); +void fuse_dlm_cache_init(struct fuse_inode *inode); /* Clean up a page cache lock manager */ void fuse_dlm_cache_release_locks(struct fuse_inode *inode); -/* Lock a range of pages */ -int fuse_dlm_lock_range(struct fuse_inode *inode, uint64_t start, - uint64_t end, enum fuse_page_lock_mode mode); +/* + * Publish a FUSE_DLM_WB_LOCK for [start, end] before it is sent, so a + * revoke processed while the reply is on the wire can mark it. @req is + * caller-owned storage, live until the matching commit or abort. The + * mode is not recorded until the grant is, so only the commit takes it. + */ +void fuse_dlm_request_begin(struct fuse_inode *inode, + struct fuse_dlm_range *req, uint64_t start, + uint64_t end); -/* As above, but refuse (-EAGAIN) if a revoke ran since @gen was sampled */ -int fuse_dlm_lock_range_gen(struct fuse_inode *inode, uint64_t start, - uint64_t end, enum fuse_page_lock_mode mode, - uint64_t gen); +/* + * Retire @req and record the grant [start, end] as one step under the + * cache lock. -EAGAIN means a revoke overlapped @req in flight and + * nothing was recorded; the caller must request again. @req is retired + * either way. + */ +int fuse_dlm_request_commit(struct fuse_inode *inode, + struct fuse_dlm_range *req, uint64_t start, + uint64_t end, enum fuse_page_lock_mode mode); -/* Sample the revocation generation (see fuse_dlm_lock_range_gen()) */ -uint64_t fuse_dlm_revoke_gen(struct fuse_inode *inode); +/* Retire @req without recording anything */ +void fuse_dlm_request_abort(struct fuse_inode *inode, + struct fuse_dlm_range *req); /* Unlock a range of pages */ int fuse_dlm_unlock_range(struct fuse_inode *inode, uint64_t start, uint64_t end); -/* Check if a page range is already locked */ -bool fuse_dlm_range_is_locked(struct fuse_inode *inode, uint64_t start, - uint64_t end, enum fuse_page_lock_mode mode); - /* Re-validate a fuse_get_dlm_lock() grant against the live lock tree */ bool fuse_dlm_lock_is_held(struct fuse_inode *inode, loff_t offset, size_t length, enum fuse_page_lock_mode mode); -/* Is any part of the file held for write? */ -bool fuse_dlm_write_grant_exists(struct fuse_inode *inode); +/* Hold [start, end] again so writeback can send what it found revoked */ +int fuse_dlm_regrant_range(struct fuse_file *ff, struct inode *inode, + uint64_t start, uint64_t end); + /* This is the interface to the filesystem */ int fuse_get_dlm_lock(struct file *file, loff_t offset, diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 31580b7834e296..e3de135c291220 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -202,21 +202,6 @@ struct fuse_inode { /* dlm locked areas we have sent lock requests for */ struct fuse_dlm_cache dlm_locked_areas; - /* - * Serializes buffered-write page-cache dirtying against - * the forced-direct-IO latch transition driven by - * NOTIFY_INVAL_INODE (fuse_reverse_inval_inode()), which - * may be delivered by the same server thread that still - * owes a reply to an in-flight write holding the inode - * lock. The buffered writer holds this for read around - * the dirtying and re-checks the latch under it; the - * NOTIFY latch site takes it for write (trylock, never - * blocking) around its page-cache invalidate + latch set. - * Only regular files initialise it -- it shares storage - * with the readdir-cache union arm. - */ - struct percpu_rw_semaphore *wb_inval_rwsem; - /* * Rate of FUSE_NOTIFY_INVAL_INODE data invalidations * for this whole file: notify_stamp is the jiffies of diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index d54676b73abf9e..7e573b4e7e4004 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -220,23 +220,6 @@ static void fuse_evict_inode(struct inode *inode) WARN_ON(!list_empty(&fi->queued_writes)); fuse_dlm_cache_release_locks(fi); } - - /* - * Free the coherency gate here rather than in ->free_inode: that runs - * from an RCU callback, where percpu_free_rwsem() may sleep in - * rcu_sync_dtor() if the write side has not fully quiesced. No user - * can remain by eviction time: gate readers hold a file reference and - * a concurrent notify holds an inode reference. wb_inval_rwsem lives - * in the regular-file union arm and is only ever allocated for regular - * files, so gate on S_ISREG (but not fuse_is_bad() -- bad-marked - * regular files still own a gate); a directory's overlapping - * readdir-cache fields must not be misread. - */ - if (S_ISREG(inode->i_mode) && fi->wb_inval_rwsem) { - percpu_free_rwsem(fi->wb_inval_rwsem); - kfree(fi->wb_inval_rwsem); - fi->wb_inval_rwsem = NULL; - } } static int fuse_reconfigure(struct fs_context *fsc) @@ -371,8 +354,17 @@ static void fuse_change_attributes_common_sx(struct inode *inode, } } - /* Common fields for both statx and getattr */ - if (attr->blksize != 0) + /* + * Common fields for both statx and getattr. + * + * A writeback connection was refused at FUSE_INIT unless its block + * is a page, for the reasons given there, and a server naming a + * different one per inode does not get to take that back. What it + * named is still reported as st_blksize out of + * fi->cached_i_blkbits; this is only what the page cache is + * tracked in. + */ + if (attr->blksize != 0 && !fc->writeback_cache) inode->i_blkbits = ilog2(attr->blksize); else inode->i_blkbits = inode->i_sb->s_blocksize_bits; @@ -527,14 +519,14 @@ u32 fuse_get_cache_mask(struct inode *inode) * for exactly what the grant covers: * * - size, when the server reports less than i_size and the tail it does not - * know about, [srv_size, i_size), is entirely under a write grant. Taking - * the server's answer would shrink i_size and have truncate_pagecache() - * throw the unwritten tail away. - * - mtime and ctime, while a write grant covers unwritten data: our writes - * have stamped them locally and the server's stamps predate them. Only - * while the cache is actually dirty, not for as long as the grant lives: - * a grant is held until it is revoked or the inode is evicted, and past - * the writeback the server's stamps are the newer ones. Keeping ours + * know about, [attr->size, i_size), is entirely under a write grant. + * Taking the server's answer would shrink i_size and have + * truncate_pagecache() throw the unwritten tail away. + * - mtime and ctime, while the page cache is dirty or under writeback: our + * writes have stamped them locally and the server's stamps predate them. + * Only while the cache is actually dirty, not for as long as a grant + * lives: a grant is held until it is revoked or the inode is evicted, and + * past the writeback the server's stamps are the newer ones. Keeping ours * beyond that would hide a remote chown or chmod indefinitely. * * A remote truncate cannot slip through. It has to revoke the grant first, @@ -558,16 +550,31 @@ static u32 fuse_attr_cache_mask(struct inode *inode, struct fuse_attr *attr, !S_ISREG(inode->i_mode)) return cache_mask; - if (!fuse_dlm_write_grant_exists(fi)) - return cache_mask; - + /* + * A dirty mapping keeps the local attributes authoritative even + * when no grant is recorded: a fault dirties pages under a + * page-mkwrite lock that is never recorded, and a truncate revokes + * the tail grants itself while cached writes above the new size + * are still waiting for writeback. + */ if (mapping_tagged(inode->i_mapping, PAGECACHE_TAG_DIRTY) || mapping_tagged(inode->i_mapping, PAGECACHE_TAG_WRITEBACK)) cache_mask |= STATX_MTIME | STATX_CTIME; + /* + * The local size stays authoritative while the extension is + * covered by a write grant, and also while anything in + * [attr->size, size) is dirty or under writeback: those bytes + * exist only here, and taking the server's smaller size would + * truncate them away before they are ever sent. The grant check + * alone misses them, because a page-mkwrite grant is never + * recorded and a local truncate revokes its own tail grants. + */ if (have_size && size > (loff_t) attr->size && - fuse_dlm_lock_is_held(fi, attr->size, size - attr->size, - FUSE_PAGE_LOCK_WRITE)) + (fuse_dlm_lock_is_held(fi, attr->size, size - attr->size, + FUSE_PAGE_LOCK_WRITE) || + filemap_range_needs_writeback(inode->i_mapping, attr->size, + size - 1))) cache_mask |= STATX_SIZE; return cache_mask; @@ -901,22 +908,28 @@ static void fuse_dlm_revoke_inval_range(struct fuse_inode *fi, loff_t offset, * Drop a page-cache range on behalf of a NOTIFY invalidate. * * invalidate_inode_pages2_range() waits out folios under writeback and - * launders dirty ones, both of which need a FUSE_WRITE reply. While - * writepages are frozen (fuse_set_nowrite(): truncate, O_TRUNC open, fsync, - * pre-SETATTR flush) no reply can arrive, because fuse_flush_writepages() - * parks the request on fi->queued_writes until fuse_release_nowrite(). A - * server that revokes from inside the handler it is revoking for then - * deadlocks against its own reply. fuse_do_setattr() states the same rule - * for its own invalidate. + * launders dirty ones, both of which need a FUSE_WRITE reply. It is only + * needed when the range can hold data the server has not seen. + * + * @may_be_dirty false says it cannot, on the strength of the DLM range + * record: every way a folio gets dirtied under a grant raises that record + * before the data lands, so a range it reports clean has no dirty folio to + * launder. invalidate_mapping_pages() then drops the same folios without + * ever waiting for the server. * - * So while frozen use invalidate_mapping_pages(), which skips dirty and - * under-writeback folios and never blocks. The stale clean folios still - * go, and the freezes that span a request drop the cache themselves once - * they complete: fuse_do_setattr() invalidates the mapping after releasing - * the freeze, the O_TRUNC open path calls truncate_pagecache(). + * The same substitution is forced while writepages are frozen + * (fuse_set_nowrite(): truncate, O_TRUNC open, fsync, pre-SETATTR flush), + * where no reply can arrive because fuse_flush_writepages() parks the + * request on fi->queued_writes until fuse_release_nowrite(). A server that + * revokes from inside the handler it is revoking for would otherwise + * deadlock against its own reply. fuse_do_setattr() states the same rule + * for its own invalidate. There the dirty folios are left behind, and the + * freezes that span a request drop the cache themselves once they complete: + * fuse_do_setattr() invalidates the mapping after releasing the freeze, the + * O_TRUNC open path calls truncate_pagecache(). */ static void fuse_notify_invalidate_range(struct inode *inode, pgoff_t start, - pgoff_t end) + pgoff_t end, bool may_be_dirty) { struct fuse_inode *fi = get_fuse_inode(inode); bool frozen; @@ -925,7 +938,7 @@ static void fuse_notify_invalidate_range(struct inode *inode, pgoff_t start, frozen = fi->writectr < 0; spin_unlock(&fi->lock); - if (frozen) + if (frozen || !may_be_dirty) invalidate_mapping_pages(inode->i_mapping, start, end); else invalidate_inode_pages2_range(inode->i_mapping, start, end); @@ -934,11 +947,12 @@ static void fuse_notify_invalidate_range(struct inode *inode, pgoff_t start, int fuse_reverse_inval_inode(struct fuse_conn *fc, u64 nodeid, loff_t offset, loff_t len) { - struct percpu_rw_semaphore *wb_sem = NULL; struct fuse_inode *fi; struct inode *inode; + loff_t end_byte; pgoff_t pg_start; pgoff_t pg_end; + bool tracked; inode = fuse_ilookup(fc, nodeid, NULL); if (!inode) @@ -971,57 +985,54 @@ int fuse_reverse_inval_inode(struct fuse_conn *fc, u64 nodeid, else pg_end = (offset + len - 1) >> PAGE_SHIFT; + /* Byte bounds of the same region */ + end_byte = len <= 0 ? LLONG_MAX : offset + len - 1; + /* - * A data invalidation means another (remote) entity is modifying - * the file. Two things happen here: + * A data invalidation means another (remote) entity is + * modifying the file. Two things happen here: * - * 1. Coherency. Drop the affected page-cache range so no local - * read returns a folio the remote modify has superseded. This - * runs under the write side of the per-inode coherency gate - * (wb_inval_rwsem), which fences cache-serving buffered reads - * and buffered writes out for the whole invalidate. Unlike the - * old best-effort trylock this BLOCKS -- the notify has - * priority: percpu_down_write() parks new gate readers, drains - * in-flight ones, then invalidates. A blocking writer here is - * safe only under a server that services request replies on - * threads other than the one delivering this notify: the write - * side waits for gate readers to drain, and a cache-miss read - * holds the read side across its FUSE_READ round-trip. redfs' - * dlm server provides that contract; a server that cannot must - * not enable writeback+dlm. + * 1. Coherency. Drop the affected page-cache range so no + * local read returns a folio the remote modify has + * superseded. Nothing is fenced out for it. A read + * racing the drop either misses and refetches or returns + * data that was current when it was copied. A write + * racing it is caught on the way out instead: its bytes + * were recorded before they were dirtied, this revoke + * marks the range rather than forgetting it, and + * writeback holds the range again before sending + * anything it finds marked that way. * - * 2. Latch. Keep a moving average (fuse_notify_inval_hot(), under - * fi->lock, updated for every data invalidation) of how fast - * these arrive; when they come in a rapid stream -- a remote - * writer repeatedly invalidating -- and the inode is also open - * for writing here, latch it into direct IO until the last - * writer closes or it is mmapped. When latched, drop the whole - * mapping rather than just the notified range, or dirty folios - * outside it would be invisible to the forced direct reads - * (stale read / lost write). Latching is opt-in via the - * enable_notify_dio module parameter and off by default; the - * average is kept up to date either way, so enabling it at - * runtime takes effect on the next storm rather than after a - * warm-up. Clearing it at runtime stops new latches but lets + * 2. Latch. Keep a moving average (fuse_notify_inval_hot(), + * under fi->lock, updated for every data invalidation) of + * how fast these arrive; when they come in a rapid stream + * -- a remote writer repeatedly invalidating -- and the + * inode is also open for writing here, latch it into + * direct IO until the last writer closes or it is mmapped. + * When latched, drop the whole mapping rather than just + * the notified range, or dirty folios outside it would be + * invisible to the forced direct reads (stale read / lost + * write). Latching is opt-in via the enable_notify_dio + * module parameter and off by default; the average is kept + * up to date either way, so enabling it at runtime takes + * effect on the next storm rather than after a warm-up. + * Clearing it at runtime stops new latches but lets * already-latched inodes run out on the usual exits (last * writer closes, or mmap). * - * The gate (and the average) exist only for writeback+dlm regular - * files; elsewhere wb_sem is NULL and the invalidate runs - * unserialized (best-effort), as before. An mmapped inode - * keeps the gate -- fuse_cache_read_iter() and - * fuse_cache_write_iter() enter it unconditionally and rely - * on the revoke staying fenced -- but is never latched: - * a mapping needs the page cache, and fuse_file_mmap() - * reverts any latch it races with. + * The average and the latch exist only for writeback+dlm + * regular files; elsewhere there is no record to consult and + * the range is dropped as it always was. An mmapped inode is + * never latched: a mapping needs the page cache, and + * fuse_file_mmap() reverts any latch it races with. */ - if (S_ISREG(inode->i_mode) && fc->writeback_cache && - fc->dlm && !FUSE_IS_DAX(inode) && - !fuse_inode_backing(fi)) - wb_sem = fi->wb_inval_rwsem; + tracked = S_ISREG(inode->i_mode) && fc->writeback_cache && + fc->dlm && !FUSE_IS_DAX(inode) && + !fuse_inode_backing(fi); - if (wb_sem) { + if (tracked) { bool hot, has_writer, latched = false; + bool may_be_dirty, has_pages; spin_lock(&fi->lock); hot = fuse_notify_inval_hot(fi); @@ -1029,22 +1040,42 @@ int fuse_reverse_inval_inode(struct fuse_conn *fc, u64 nodeid, spin_unlock(&fi->lock); /* - * Priority write side: park new gate readers, - * drain in-flight ones, then invalidate. Blocks - * (unlike the old trylock) -- see the contract in - * the comment above. + * What this notify has to do. Nothing cached in the + * range means the drop is a no-op and the revoke is + * the whole job. Otherwise the page cache says + * whether the drop has to launder, which is what + * makes it wait for a FUSE_WRITE reply. */ - percpu_down_write(wb_sem); + has_pages = filemap_range_has_page(inode->i_mapping, + offset, end_byte); + may_be_dirty = filemap_range_needs_writeback( + inode->i_mapping, offset, end_byte); /* - * Revoke the DLM lock range under the gate write - * side, atomically with the page drop: gate readers - * re-validate their grant right after entering, and - * a grant that passed that check must stay visible - * for their whole gate hold. + * Put unwritten data on the server while the grant + * still covers it, rather than leaving it to the drop + * below. After the revoke writeback would have to + * take the range again to send those bytes: a DLM + * round trip from inside the handler the server is + * waiting on. do_writepages() runs in this context, + * so the grant is asked for before the revoke. + * + * Waited out here rather than left to the drop, which + * launders when the record says the range may be dirty + * and so waits for these same replies. One explicit + * wait, before the revoke, and the drop then finds + * nothing under writeback to block on. Either way a + * server that revokes from a thread it also needs to + * answer FUSE_WRITE on deadlocks here, the same + * contract fuse_notify_invalidate_range() states for a + * frozen inode. The error is left to the mapping, + * where fsync collects it. */ - if (fc->dlm && fc->writeback_cache) - fuse_dlm_revoke_inval_range(fi, offset, len); + if (has_pages && may_be_dirty) + filemap_write_and_wait_range(inode->i_mapping, + offset, end_byte); + + fuse_dlm_revoke_inval_range(fi, offset, len); if (enable_notify_dio && hot && has_writer && !mapping_mapped(inode->i_mapping) && @@ -1060,27 +1091,31 @@ int fuse_reverse_inval_inode(struct fuse_conn *fc, u64 nodeid, /* * Latched: drop the whole mapping (dirty folios * outside the notified range would be invisible to - * the forced direct reads). Otherwise just the - * notified range. + * the forced direct reads), and the record says + * nothing about the rest of the file, so launder. + * Otherwise just the notified range, and only if + * anything is cached there. */ if (fuse_inode_force_dio(inode)) - fuse_notify_invalidate_range(inode, 0, -1); - else + fuse_notify_invalidate_range(inode, 0, -1, true); + else if (has_pages) fuse_notify_invalidate_range(inode, pg_start, - pg_end); - - percpu_up_write(wb_sem); + pg_end, + may_be_dirty); if (latched) pr_info_ratelimited("FUSE: inode %llu latched to direct IO on invalidation notify storm\n", nodeid); } else { - /* No gate on this inode (DAX, backing, non-regular, - * or the gate allocation failed): drop the lock - * range unserialized (best-effort), as before. */ + /* + * No record on this inode (DAX, backing, non-regular, + * or no DLM), so assume the range can hold unwritten + * data and drop it as before. + */ if (fc->dlm && fc->writeback_cache) fuse_dlm_revoke_inval_range(fi, offset, len); - fuse_notify_invalidate_range(inode, pg_start, pg_end); + fuse_notify_invalidate_range(inode, pg_start, pg_end, + true); } } iput(inode); @@ -1869,8 +1904,31 @@ static void process_init_reply(struct fuse_mount *fm, struct fuse_args *args, } if (flags & FUSE_ASYNC_DIO) fc->async_dio = 1; - if (flags & FUSE_WRITEBACK_CACHE) + if (flags & FUSE_WRITEBACK_CACHE) { + /* + * A buffered write goes through iomap, which + * tracks a folio a block at a time and fills + * any block the write covers only part of. + * Writeback then sends whole dirty blocks. + * Both of those are cut at the page in this + * filesystem: fuse_dlm_buffered_write() sends + * the unaligned edges of a write to the server + * itself so that no partly written block is + * ever dirtied, and it cuts at PAGE_SIZE. + * + * A block that is not a page breaks that, and + * quietly: the fill would come back for the + * remainder of an edge block, from a server + * that need not hold anything there. Refuse + * the connection instead. + */ + if (fc->blkbits != PAGE_SHIFT) { + pr_err("fuse: writeback cache needs a page sized block, got %u\n", + 1U << fc->blkbits); + ok = false; + } fc->writeback_cache = 1; + } if (flags & FUSE_PARALLEL_DIROPS) fc->parallel_dirops = 1; if (flags & FUSE_HANDLE_KILLPRIV)