diff --git a/.github/workflows/create-redfs-pr.yml b/.github/workflows/create-redfs-pr.yml new file mode 100644 index 00000000000000..cc03d7e1219e9b --- /dev/null +++ b/.github/workflows/create-redfs-pr.yml @@ -0,0 +1,93 @@ +# Automatially run copy-from-linux-branch.sh on branches and create PR for redfs. +name: Sync to redfs repo +on: + # Triggers the workflow on pull request merged. + pull_request: + branches: [ "redfs-*" ] + types: [ "closed" ] + +jobs: + create-redfs-pr: + if: github.event.pull_request.merged == true + runs-on: ubuntu-latest + steps: + # Checks-out to a different directory to avoid following checkout removing it. + - uses: actions/checkout@v4 + with: + path: linux + + - name: Try to checkout sync-${{ github.ref_name }} if it exists + uses: actions/checkout@v4 + id: try-checkout + continue-on-error: true + with: + repository: DDNStorage/redfs + ref: sync-${{ github.ref_name }} + fetch-depth: 0 + path: redfs + token: ${{ secrets.REDFS_TOKEN }} + + - name: Fallback to checkout main + if: steps.try-checkout.outcome == 'failure' + uses: actions/checkout@v4 + with: + repository: DDNStorage/redfs + ref: main + fetch-depth: 0 + path: redfs + token: ${{ secrets.REDFS_TOKEN }} + + - name: Initialize git + run: | + git config --global user.name "DDNStorage RED Workflow" + git config --global user.email "red@ddn.com" + + - name: Create tracking branch based on main + if: steps.try-checkout.outcome == 'failure' + run: | + pushd redfs + git checkout -b sync-${{ github.ref_name }} + popd + + - name: Generate PR for redfs + run: | + declare -A MAP + MAP["redfs-rhel9_4-427.42.1"]="5.14.0-427.42.1.el9_4" + MAP["redfs-rhel9_5-503.40.1"]="5.14.0-503.40.1.el9_5" + MAP["redfs-rhel9_6-570.12.1"]="5.14.0-570.12.1.el9_6" + MAP["redfs-ubuntu-noble-6.8.0-58.60"]="6.8.0-58.60.ubuntu" + kerver=${MAP["${{ github.ref_name }}"]} + if [ -z ${kerver} ]; then + echo "Cannot find target kernel version" + exit 1 + fi + pushd redfs + ./copy-from-linux-branch.sh $GITHUB_WORKSPACE/linux ${kerver} + git add src/$kerver + echo -e "Sync with ${{ github.repository }} branch ${{ github.ref_name }}\n" > ../commit.msg + echo -e "Sync with ${{ github.repository }} branch ${{ github.ref_name }} by commit" >> ../commit.msg + echo -e "${{ github.sha }}" >> ../commit.msg + RET=0 + git commit -F ../commit.msg 2> ../commit.log || RET=$?; + if [ -s ../commit.log ]; then + echo "Error detcted in commit:" + cat ../commit.log + exit 1 + elif [ $RET -eq 0 ]; then + echo "Done. Push the code to remote:" + git push origin sync-${{ github.ref_name }} 2> ../push.log ||: + else + echo "No changes to existed codes. Still try with PR." + fi + if [ -s ../push.log ]; then + echo "Message detected in push:" + cat ../push.log + fi + gh pr create --base main --fill || RET=$? + if [ $RET -eq 1 ]; then + echo "No pending changes for PR, returning $RET." + fi + popd + env: + GH_TOKEN: ${{ secrets.REDFS_TOKEN }} + diff --git a/Documentation/filesystems/fuse/fuse-AOP_TRUNCATED_PAGE-reason.txt b/Documentation/filesystems/fuse/fuse-AOP_TRUNCATED_PAGE-reason.txt new file mode 100644 index 00000000000000..7ae02c1f1e88b7 --- /dev/null +++ b/Documentation/filesystems/fuse/fuse-AOP_TRUNCATED_PAGE-reason.txt @@ -0,0 +1,579 @@ +================================================================================= +WHY FUSE CONVERTS -EDEADLK TO AOP_TRUNCATED_PAGE IN fuse_read_folio() +================================================================================= + +TLDR: To prevent ABBA deadlock between page locks and DLM (cluster) locks. + +================================================================================= +THE FUNDAMENTAL DEADLOCK SCENARIO - DETAILED CODE FLOW +================================================================================= + +This deadlock occurs when two CPUs hold locks in opposite order, creating +a circular dependency. Below is the exact code flow showing how this happens. + +───────────────────────────────────────────────────────────────────────────── +TIMELINE: CORE-0 (Application Read Path - Waiting for DLM Lock) +───────────────────────────────────────────────────────────────────────────── + +T0: Application calls read() + ├─ sys_read() [fs/read_write.c] + └─ vfs_read() [fs/read_write.c:440] + └─ generic_file_read_iter() [mm/filemap.c:2850] + +T1: Enter page cache read path + └─ filemap_read() [mm/filemap.c:2675] + │ ACQUIRES: filemap invalidate_lock (shared) [line 2678] + │ STATUS: Holding invalidate_lock + │ + └─ filemap_get_pages() [mm/filemap.c:2712] + └─ filemap_update_page() [mm/filemap.c:2625] + +T2: Attempt to lock page + └─ folio_trylock() [mm/filemap.c:2466] + │ ACQUIRES: page lock + │ STATUS: Holding invalidate_lock + page lock + │ + └─ filemap_read_folio() [mm/filemap.c:2497] + └─ filler(file, folio) [mm/filemap.c:2413] + │ Calls address_space_operations->read_folio + │ + └─ ocfs2_read_folio() [fs/ocfs2/aops.c:262] + +T3: Attempt to acquire DLM lock (CRITICAL POINT) + └─ ocfs2_inode_lock_with_folio() [fs/ocfs2/aops.c:271] + │ STATUS: Holding invalidate_lock + page lock + │ WANTS: DLM inode lock + │ + └─ ocfs2_inode_lock_full() [fs/ocfs2/dlmglue.c:2552] + └─ __ocfs2_cluster_lock() [fs/ocfs2/dlmglue.c:2465] + └─ dlm_lock() [fs/dlm/lock.c:3372] + │ + ❌ BLOCKS HERE - DLM lock held by Core-1 + │ Waiting for DLM lock while holding page lock + │ + DEADLOCK CONDITION: Cannot proceed + +───────────────────────────────────────────────────────────────────────────── +TIMELINE: CORE-1 (Memory Reclaim Path - Waiting for Page Lock) +───────────────────────────────────────────────────────────────────────────── + +T0: Memory pressure triggers reclaim + └─ kswapd() or direct memory reclaim [mm/vmscan.c] + └─ shrink_node_memcgs() [mm/vmscan.c] + +T1: Start reclaiming pages + └─ shrink_inactive_list() [mm/vmscan.c:2004] + └─ shrink_folio_list() [mm/vmscan.c:1098] + │ Iterating through pages to evict + │ + └─ Check if folio needs writeback [mm/vmscan.c:1227] + +T2: DLM lock already held during downconvert/writeback + │ CONTEXT: This thread is handling: + │ - DLM lock downconvert request from another node + │ - Metadata update requiring DLM lock + │ - Page writeback with cluster coordination + │ + │ STATUS: Holding DLM INODE LOCK (exclusive or shared) + │ + └─ pageout(folio, mapping, ...) [mm/vmscan.c:1452] + │ May trigger writepage callback + │ + └─ Filesystem writepage operations + │ These operations may update metadata + │ Already holding DLM lock for coordination + +T3: Attempt to acquire page lock (CRITICAL POINT) + └─ folio_trylock() [mm/vmscan.c:1129] + │ STATUS: Holding DLM lock + │ WANTS: page lock + │ + ❌ BLOCKS HERE - Page lock held by Core-0 + │ Waiting for page lock while holding DLM lock + │ + DEADLOCK CONDITION: Cannot proceed + +───────────────────────────────────────────────────────────────────────────── +RESULT: ABBA DEADLOCK (Circular Wait) +───────────────────────────────────────────────────────────────────────────── + +Lock Acquisition Order Violation: + + Core-0: invalidate_lock → page lock → [WANTS] DLM lock + Core-1: DLM lock → [WANTS] page lock + +Circular Dependency: + Core-0 waits for DLM lock (held by Core-1) + Core-1 waits for page lock (held by Core-0) + + ┌──────────────┐ ┌──────────────┐ + │ CORE-0 │ │ CORE-1 │ + │ │ │ │ + │ Holds: page │◄───────────────┤ Wants: page │ + │ Wants: DLM │ │ Holds: DLM │ + │ │────────────────►│ │ + └──────────────┘ └──────────────┘ + ▲ │ + │ │ + └──────────── DEADLOCK ────────────┘ + +Neither Core-0 nor Core-1 can proceed → System deadlock + +================================================================================= +WHY PAGE LOCKS ARE HELD WHEN CALLING read_folio() +================================================================================= + +From mm/filemap.c:do_read_cache_folio(): + 1. Page is allocated/looked up in page cache + 2. folio_trylock() acquires the page lock + 3. filler (read_folio) is called WITH page locked + 4. Lock prevents concurrent modifications during I/O + 5. Lock ensures atomic read operation + +The page lock must be held to prevent: + - Concurrent writes while reading + - Page truncation during read + - Multiple simultaneous reads to the same page + +================================================================================= +THE LOCK ORDERING RULE (From GFS2 Documentation) +================================================================================= + +From Documentation/filesystems/gfs2-glocks.rst: + + Lock ordering within GFS2: + 1. i_rwsem (if required) + 2. Rename glock + 3. Inode glock(s) + 4. Rgrp glock(s) + 5. Transaction glock + 6. i_rw_mutex (if required) + 7. Page lock (ALWAYS LAST!) + +**Critical Rule: PAGE LOCK IS ALWAYS ACQUIRED LAST** + +This means: + - All filesystem/cluster locks must be acquired BEFORE page lock + - NO filesystem lock can be acquired while holding a page lock + - Prevents lock inversion with memory reclaim + +================================================================================= +OCFS2'S SOLUTION: Breaking the Deadlock with AOP_TRUNCATED_PAGE +================================================================================= + +Location: fs/ocfs2/dlmglue.c:2547-2568 +Function: ocfs2_inode_lock_with_folio() + +───────────────────────────────────────────────────────────────────────────── +CODE: Deadlock Prevention Pattern +───────────────────────────────────────────────────────────────────────────── + + int ocfs2_inode_lock_with_folio(struct inode *inode, + struct buffer_head **ret_bh, int ex, struct folio *folio) + { + int ret; + + // LINE 2552: Try non-blocking DLM lock first + ret = ocfs2_inode_lock_full(inode, ret_bh, ex, OCFS2_LOCK_NONBLOCK); + + if (ret == -EAGAIN) { + // DLM lock not immediately available + // Would block, risking deadlock with Core-1 + + // LINE 2554: ✓ CRITICAL - Unlock page BEFORE waiting for DLM lock + folio_unlock(folio); + + /* + * Wait for DLM lock WITHOUT holding page lock + * This breaks the Core-0 → Core-1 dependency + * + * From comment (lines 2555-2560): + * "If we can't get inode lock immediately, we should not return + * directly here, since this will lead to a softlockup problem. + * The method is to get a blocking lock and immediately unlock + * before returning, this can avoid CPU resource waste due to + * lots of retries, and benefits fairness in getting lock." + */ + + // LINE 2562: Acquire lock in blocking mode (without page lock) + if (ocfs2_inode_lock(inode, ret_bh, ex) == 0) + ocfs2_inode_unlock(inode, ex); // Immediately release + + // LINE 2564: Signal VFS to retry entire operation + ret = AOP_TRUNCATED_PAGE; + } + + return ret; + } + +───────────────────────────────────────────────────────────────────────────── +HOW THIS BREAKS THE DEADLOCK +───────────────────────────────────────────────────────────────────────────── + +BEFORE (would deadlock): + Core-0: page lock → [WAITS for DLM lock] → BLOCKED by Core-1 + Core-1: DLM lock → [WAITS for page lock] → BLOCKED by Core-0 + Result: Circular wait, system hangs + +AFTER (with fix): + Core-0 at T3: + 1. Detects DLM lock unavailable (-EAGAIN) + 2. ✓ Releases page lock (line 2554) + 3. Waits for DLM lock (now safe, no page lock held) + 4. Gets DLM lock, immediately releases it (fair cycling) + 5. Returns AOP_TRUNCATED_PAGE + + Core-0 caller (mm/filemap.c): + 6. Sees AOP_TRUNCATED_PAGE + 7. Drops folio reference + 8. Jumps to retry label (e.g., do_read_cache_folio line 3971) + 9. Re-acquires page from cache + 10. Tries read again + + Core-1: + - While Core-0 released page lock (step 2) + - Core-1 can now acquire page lock + - Core-1 completes its work + - Core-1 releases DLM lock + + Core-0 retry: + - Now DLM lock is available + - Successfully acquires DLM lock + - Completes read operation + + Result: Both cores make progress, no deadlock + +───────────────────────────────────────────────────────────────────────────── +KEY INSIGHT FROM OCFS2 COMMENTS +───────────────────────────────────────────────────────────────────────────── + +From fs/ocfs2/dlmglue.c:1627-1632: + + "This is helping work around a lock inversion between the page lock + and dlm locks. One path holds the page lock while calling aops + which block acquiring dlm locks. The voting thread holds dlm + locks while acquiring page locks while down converting data locks. + This block is helping an aop path notice the inversion and back + off to unlock its page lock before trying the dlm lock again." + +Translation: + - "One path" = Core-0 (application read) + - "voting thread" = Core-1 (DLM downconvert/memory reclaim) + - "lock inversion" = ABBA deadlock scenario + - "back off" = Release page lock, return AOP_TRUNCATED_PAGE + +================================================================================= +FUSE'S SPECIFIC CONTEXT: Same Pattern, Different DLM +================================================================================= + +FUSE implements a custom "DLM cache" (fs/fuse/fuse_dlm_cache.c): + - NOT the kernel's DLM subsystem (used by GFS2/OCFS2) + - Tracks page-level locks for distributed coordination + - Sends FUSE_DLM_WB_LOCK operations to userspace daemon + - Userspace daemon handles cluster lock negotiation + +───────────────────────────────────────────────────────────────────────────── +FUSE READ PATH - CODE FLOW WITH DEADLOCK RISK +───────────────────────────────────────────────────────────────────────────── + +Core-0: Application reading from FUSE filesystem + + filemap_read() [mm/filemap.c:2675] + └─ filemap_get_pages() [line 2712] + └─ filemap_update_page() [line 2625] + └─ folio_trylock() [line 2466] + │ ACQUIRES: page lock + │ + └─ filemap_read_folio() [line 2497] + └─ fuse_read_folio() [fs/fuse/file.c:947] + └─ fuse_do_readfolio() [fs/fuse/file.c:956] + └─ fuse_simple_request() [fs/fuse/file.c:932] + │ + │ Sends FUSE_READ to userspace daemon + │ Daemon may need to acquire cluster lock + │ + └─ Might return -EDEADLK if DLM detects possible + deadlock due to concurrant page invalidation + - For now -EAGAIN handled the same + + +───────────────────────────────────────────────────────────────────────────── +WHY FUSE NEEDS THIS CONVERSION +───────────────────────────────────────────────────────────────────────────── + +When fuse_simple_request() returns -EDEADLK: + - Userspace FUSE daemon encountered transient failure + - Could be: cluster lock contention, timeout, daemon busy + - The page might be stale/modified during the wait + - Page lock is STILL HELD at this point + +If -EDEADLK were returned directly: + ❌ VFS would see error and fail the read + ❌ Page would remain locked + ❌ No retry mechanism triggered + +By converting to AOP_TRUNCATED_PAGE: + ✓ Signals to VFS: "retry the entire page acquisition" + ✓ fuse_read_folio() unlocks page before returning (line 962) + ✓ Follows the OCFS2 pattern: unlock page, retry operation + ✓ Prevents potential deadlock with cluster lock operations + ✓ Uses VFS's built-in retry mechanism in mm/filemap.c + +───────────────────────────────────────────────────────────────────────────── +CODE: fuse_read_folio() - Ensures Page Unlock +───────────────────────────────────────────────────────────────────────────── + +Location: fs/fuse/file.c:947-964 + + static int fuse_read_folio(struct file *file, struct folio *folio) + { + struct inode *inode = folio->mapping->host; + int err; + + err = -EIO; + if (fuse_is_bad(inode)) + goto out; + + // LINE 956: Calls fuse_do_readfolio, may return AOP_TRUNCATED_PAGE + err = fuse_do_readfolio(file, folio, 0, folio_size(folio)); + if (!err) + folio_mark_uptodate(folio); + + fuse_invalidate_atime(inode); + out: + // LINE 962: ✓ CRITICAL - Always unlocks page before returning + folio_unlock(folio); + return err; // Returns AOP_TRUNCATED_PAGE if -EDEADLK occurred + } + +This matches OCFS2's pattern: + 1. Detect lock contention (-EAGAIN from daemon) + - Note: confusing that OCFS2 uses -EAGAIN instead of -EDEADLK + 2. Unlock the page (line 962) + 3. Return AOP_TRUNCATED_PAGE + 4. VFS retries the operation + +================================================================================= +THE CONTRACT: AOP_TRUNCATED_PAGE Semantics +================================================================================= + +From include/linux/fs.h: + "AOP_TRUNCATED_PAGE: The AOP method that was handed a locked page has + unlocked it and the page might have been truncated. The caller should + back up to acquiring a new page and trying again." + +Key requirements: + - The folio MUST be unlocked before returning AOP_TRUNCATED_PAGE + - Signals "retry from scratch" to the caller + - Caller will drop page reference and re-acquire + - Prevents livelock through filesystem's "reasonable precautions" + +Callers in mm/filemap.c handle it consistently: + - do_read_cache_folio(): "if (err == AOP_TRUNCATED_PAGE) goto repeat;" + - filemap_get_pages(): "if (err == AOP_TRUNCATED_PAGE) goto retry;" + - do_filemap_fault(): "if (error == AOP_TRUNCATED_PAGE) goto retry_find;" + +================================================================================= +WHY NOT JUST RETURN -EDEADLK (-EAGAIN)? +================================================================================= + +From OCFS2 comments (fs/ocfs2/dlmglue.c:2555-2560): + "If we can't get inode lock immediately, we should not return + directly here, since this will lead to a softlockup problem. + The method is to get a blocking lock and immediately unlock + before returning, this can avoid CPU resource waste due to + lots of retries, and benefits fairness in getting lock." + +Returning -EAGAIN directly would cause: + - VFS to immediately retry without unlocking the page + - Busy-waiting loop (soft lockup) + - CPU resource waste + - Potential starvation of lock waiters + +AOP_TRUNCATED_PAGE forces: + - Full retry cycle (unlock, re-acquire, re-read) + - Allows other threads to make progress + - Yields CPU properly + - Fair lock acquisition + +================================================================================= +COMPLETE EXECUTION TIMELINE - DEADLOCK AND RESOLUTION +================================================================================= + +Without AOP_TRUNCATED_PAGE (DEADLOCK SCENARIO): +──────────────────────────────────────────────────────────────────────────── + +Time Core-0 (Read Path) Core-1 (Reclaim/DLM) +──── ───────────────────────────── ────────────────────────────────── +T0 Enter filemap_read() [Idle] + +T1 Lock: invalidate_lock [Idle] + Lock: page lock + +T2 Call: ocfs2_read_folio() DLM downconvert starts + Lock: DLM inode lock + +T3 Want: DLM inode lock Want: page lock + ❌ BLOCKED (held by Core-1) ❌ BLOCKED (held by Core-0) + +T4 [DEADLOCK] [DEADLOCK] + Holding: page lock Holding: DLM lock + Waiting: DLM lock Waiting: page lock + +∞ System hangs System hangs + +With AOP_TRUNCATED_PAGE (DEADLOCK PREVENTION): +──────────────────────────────────────────────────────────────────────────── + +Time Core-0 (Read Path) Core-1 (Reclaim/DLM) +──── ───────────────────────────── ────────────────────────────────── +T0 Enter filemap_read() [Idle] + +T1 Lock: invalidate_lock [Idle] + Lock: page lock + +T2 Call: ocfs2_read_folio() DLM downconvert starts + Lock: DLM inode lock + +T3 Try: DLM lock (NONBLOCK) Want: page lock + Returns: -EDEADLK ❌ BLOCKED (held by Core-0) + +T4 ✓ Unlock: page lock ✓ Acquires: page lock + Return: AOP_TRUNCATED_PAGE Continues: reclaim work + +T5 Caller sees AOP_TRUNCATED_PAGE Completes: page eviction + goto retry (line 3971) Unlock: page lock + Unlock: DLM lock + +T6 Re-acquire: page from cache [Completes] + Re-try: read operation + +T7 Try: DLM lock (NONBLOCK) + ✓ SUCCESS (Core-1 released it) + Lock: DLM inode lock + +T8 Complete: read operation + Unlock: DLM lock + Unlock: page lock + +T9 Return: success to application + +Result: Both cores complete successfully, no deadlock + +================================================================================= +VFS RETRY MECHANISM - HOW AOP_TRUNCATED_PAGE TRIGGERS RETRY +================================================================================= + +Location: mm/filemap.c + +───────────────────────────────────────────────────────────────────────────── +Caller 1: do_read_cache_folio() - lines 3967-3973 +───────────────────────────────────────────────────────────────────────────── + + repeat: + folio = __filemap_get_folio(...); + + filler: + err = filemap_read_folio(file, filler, folio); // Calls read_folio + if (err) { + folio_put(folio); + if (err == AOP_TRUNCATED_PAGE) + goto repeat; // ← RETRY from beginning + return ERR_PTR(err); + } + +Effect: Full retry of page acquisition and read + +───────────────────────────────────────────────────────────────────────────── +Caller 2: filemap_get_pages() - lines 2638-2639 +───────────────────────────────────────────────────────────────────────────── + + retry: + ... + err = filemap_update_page(...); + if (err < 0) + goto err; + ... + err: + if (err < 0) + folio_put(folio); + if (likely(--fbatch->nr)) + return 0; + if (err == AOP_TRUNCATED_PAGE) + goto retry; // ← RETRY from beginning + return err; + +Effect: Re-attempts page update after releasing folio + +───────────────────────────────────────────────────────────────────────────── +Caller 3: do_filemap_fault() - lines 3542-3543 +───────────────────────────────────────────────────────────────────────────── + + retry_find: + ... + error = filemap_read_folio(file, mapping->a_ops->read_folio, folio); + ... + if (!error || error == AOP_TRUNCATED_PAGE) + goto retry_find; // ← RETRY page fault handling + +Effect: Re-handles the page fault from scratch + +================================================================================= +SUMMARY: Why -EDEADLK → AOP_TRUNCATED_PAGE is Essential +================================================================================= + +The conversion -EDEADLK → AOP_TRUNCATED_PAGE in fuse_read_folio() is necessary: + +1. **Prevent Deadlock**: Avoids page lock vs cluster lock ABBA deadlock + - Core-0 holds page lock, wants DLM lock + - Core-1 holds DLM lock, wants page lock + - Conversion forces Core-0 to release page lock first + +2. **Follow Lock Ordering**: Page lock must be released before cluster locks + - Kernel rule: Page lock is ALWAYS acquired last + - Documented in Documentation/filesystems/gfs2-glocks.rst + - Prevents lock inversion with memory reclaim + +3. **Enable Retry**: Uses VFS's built-in retry mechanism properly + - AOP_TRUNCATED_PAGE is understood by mm/filemap.c + - Triggers automatic retry loops at multiple call sites + - -EDEADLK alone would just fail the operation + +4. **Ensure Fairness**: Allows fair lock acquisition through retry cycle + - Other threads get chance to acquire locks + - Prevents starvation of waiters + - Better system responsiveness + +This pattern is the established solution in distributed filesystems (OCFS2, GFS2) +for handling deadlock between page cache and cluster coordination locks. + +================================================================================= +REFERENCES - Exact Code Locations +================================================================================= + +Core-0 (Read Path): + - Entry: filemap_read() [mm/filemap.c:2675] + - Page lock: folio_trylock() [mm/filemap.c:2466] + - FUSE read: fuse_do_readfolio() [fs/fuse/file.c:905] + - -EDEADLK conversion [fs/fuse/file.c:934-935] + - Page unlock: folio_unlock() [fs/fuse/file.c:962] + +Core-1 (Reclaim Path): + - Entry: shrink_folio_list() [mm/vmscan.c:1098] + - Page lock attempt: folio_trylock() [mm/vmscan.c:1129] + +OCFS2 Solution: + - Detection: ocfs2_inode_lock_with_folio() [fs/ocfs2/dlmglue.c:2547] + - Page unlock: folio_unlock() [fs/ocfs2/dlmglue.c:2554] + - Return: AOP_TRUNCATED_PAGE [fs/ocfs2/dlmglue.c:2564] + +VFS Retry: + - do_read_cache_folio() retry [mm/filemap.c:3971] + - filemap_get_pages() retry [mm/filemap.c:2639] + - do_filemap_fault() retry [mm/filemap.c:3543] + +Lock Ordering Documentation: + - GFS2 lock ordering rules [Documentation/filesystems/gfs2-glocks.rst:110-120] + - OCFS2 deadlock comments [fs/ocfs2/dlmglue.c:1627-1632] + +================================================================================= diff --git a/debian/scripts/misc/kconfig/__init__.py b/debian/scripts/misc/kconfig/__init__.py deleted file mode 100644 index e69de29bb2d1d6..00000000000000 diff --git a/fs/fuse/Makefile b/fs/fuse/Makefile index 22ad9538dfc4b8..2407870803000d 100644 --- a/fs/fuse/Makefile +++ b/fs/fuse/Makefile @@ -11,7 +11,7 @@ obj-$(CONFIG_CUSE) += cuse.o obj-$(CONFIG_VIRTIO_FS) += virtiofs.o fuse-y := trace.o # put trace.o first so we see ftrace errors sooner -fuse-y += dev.o dir.o file.o inode.o control.o xattr.o acl.o readdir.o ioctl.o +fuse-y += dev.o dir.o file.o inode.o control.o xattr.o acl.o readdir.o ioctl.o fuse_dlm_cache.o compound.o fuse-y += iomode.o fuse-$(CONFIG_FUSE_DAX) += dax.o fuse-$(CONFIG_FUSE_PASSTHROUGH) += passthrough.o backing.o diff --git a/fs/fuse/compound.c b/fs/fuse/compound.c new file mode 100644 index 00000000000000..5d84e3558a06f8 --- /dev/null +++ b/fs/fuse/compound.c @@ -0,0 +1,251 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * FUSE: Filesystem in Userspace + * Copyright (C) 2025 + * + * This file implements compound operations for FUSE, allowing multiple + * operations to be batched into a single request to reduce round trips + * between kernel and userspace. + */ + +#include "fuse_i.h" + +/* + * Compound request builder and state tracker and args pointer storage + */ +struct fuse_compound_req { + struct fuse_mount *fm; + struct fuse_compound_in compound_header; + struct fuse_compound_out result_header; + + /* Per-operation error codes */ + int op_errors[FUSE_MAX_COMPOUND_OPS]; + struct fuse_args *op_args[FUSE_MAX_COMPOUND_OPS]; +}; + +struct fuse_compound_req *fuse_compound_alloc(struct fuse_mount *fm, u32 flags) +{ + struct fuse_compound_req *compound; + + compound = kzalloc(sizeof(*compound), GFP_KERNEL); + if (!compound) + return ERR_PTR(-ENOMEM); + + compound->fm = fm; + compound->compound_header.flags = flags; + + return compound; +} + +int fuse_compound_add(struct fuse_compound_req *compound, + struct fuse_args *args) +{ + if (!compound || + compound->compound_header.count >= FUSE_MAX_COMPOUND_OPS) + return -EINVAL; + + if (args->in_pages) + return -EINVAL; + + compound->op_args[compound->compound_header.count] = args; + compound->compound_header.count++; + return 0; +} + +static void *fuse_copy_response_per_req(struct fuse_args *args, + char *resp) +{ + int i; + size_t copied = 0; + + for (i = 0; i < args->out_numargs; i++) { + struct fuse_arg current_arg = args->out_args[i]; + size_t arg_size = current_arg.size; + + if (current_arg.value && arg_size > 0) { + memcpy(current_arg.value, + (char *)resp + copied, arg_size); + copied += arg_size; + } + } + + return (char *)resp + copied; +} + +int fuse_compound_get_error(struct fuse_compound_req *compound, int op_idx) +{ + return compound->op_errors[op_idx]; +} + +static void *fuse_compound_parse_one_op(struct fuse_compound_req *compound, + int op_index, void *op_out_data, + void *response_end) +{ + struct fuse_out_header *op_hdr = op_out_data; + struct fuse_args *args = compound->op_args[op_index]; + + if (op_hdr->len < sizeof(struct fuse_out_header)) + return NULL; + + /* Check if the entire operation response fits in the buffer */ + if ((char *)op_out_data + op_hdr->len > (char *)response_end) + return NULL; + + if (op_hdr->error != 0) + compound->op_errors[op_index] = op_hdr->error; + + if (args && op_hdr->len > sizeof(struct fuse_out_header)) + return fuse_copy_response_per_req(args, op_out_data + + sizeof(struct fuse_out_header)); + + /* No response data, just advance past the header */ + return (char *)op_out_data + op_hdr->len; +} + +static int fuse_compound_parse_resp(struct fuse_compound_req *compound, + u32 count, void *response, + size_t response_size) +{ + void *op_out_data = response; + void *response_end = (char *)response + response_size; + int i; + + if (!response || response_size < sizeof(struct fuse_out_header)) + return -EIO; + + for (i = 0; i < count && i < compound->result_header.count; i++) { + op_out_data = fuse_compound_parse_one_op(compound, i, + op_out_data, + response_end); + if (!op_out_data) + return -EIO; + } + + return 0; +} + +ssize_t fuse_compound_send(struct fuse_compound_req *compound) +{ + struct fuse_args args = { + .opcode = FUSE_COMPOUND, + .nodeid = 0, + .in_numargs = 2, + .out_numargs = 2, + .out_argvar = true, + }; + size_t resp_buffer_size; + size_t actual_response_size; + size_t buffer_pos; + size_t total_expected_out_size; + void *buffer = NULL; + void *resp_payload; + ssize_t ret; + int i; + + if (!compound) { + pr_info_ratelimited("FUSE: compound request is NULL in %s\n", + __func__); + return -EINVAL; + } + + if (compound->compound_header.count == 0) { + pr_info_ratelimited("FUSE: compound request contains no operations\n"); + return -EINVAL; + } + + buffer_pos = 0; + total_expected_out_size = 0; + + for (i = 0; i < compound->compound_header.count; i++) { + struct fuse_args *op_args = compound->op_args[i]; + size_t needed_size = sizeof(struct fuse_in_header); + int j; + + for (j = 0; j < op_args->in_numargs; j++) + needed_size += op_args->in_args[j].size; + + buffer_pos += needed_size; + + for (j = 0; j < op_args->out_numargs; j++) + total_expected_out_size += op_args->out_args[j].size; + } + + buffer = kvmalloc(buffer_pos, GFP_KERNEL); + if (!buffer) + return -ENOMEM; + + buffer_pos = 0; + for (i = 0; i < compound->compound_header.count; i++) { + struct fuse_args *op_args = compound->op_args[i]; + struct fuse_in_header *hdr; + size_t needed_size = sizeof(struct fuse_in_header); + int j; + + for (j = 0; j < op_args->in_numargs; j++) + needed_size += op_args->in_args[j].size; + + hdr = (struct fuse_in_header *)(buffer + buffer_pos); + memset(hdr, 0, sizeof(*hdr)); + hdr->len = needed_size; + hdr->opcode = op_args->opcode; + hdr->nodeid = op_args->nodeid; + hdr->uid = from_kuid(compound->fm->fc->user_ns, + current_fsuid()); + hdr->gid = from_kgid(compound->fm->fc->user_ns, + current_fsgid()); + hdr->pid = pid_nr_ns(task_pid(current), + compound->fm->fc->pid_ns); + buffer_pos += sizeof(*hdr); + + for (j = 0; j < op_args->in_numargs; j++) { + memcpy(buffer + buffer_pos, op_args->in_args[j].value, + op_args->in_args[j].size); + buffer_pos += op_args->in_args[j].size; + } + } + + resp_buffer_size = total_expected_out_size + + (compound->compound_header.count * + sizeof(struct fuse_out_header)); + + resp_payload = kvmalloc(resp_buffer_size, GFP_KERNEL | __GFP_ZERO); + if (!resp_payload) { + ret = -ENOMEM; + goto out_free_buffer; + } + + compound->compound_header.result_size = total_expected_out_size; + + args.in_args[0].size = sizeof(compound->compound_header); + args.in_args[0].value = &compound->compound_header; + args.in_args[1].size = buffer_pos; + args.in_args[1].value = buffer; + + args.out_args[0].size = sizeof(compound->result_header); + args.out_args[0].value = &compound->result_header; + args.out_args[1].size = resp_buffer_size; + args.out_args[1].value = resp_payload; + + ret = fuse_simple_request(compound->fm, &args); + if (ret < 0) + goto out; + + actual_response_size = args.out_args[1].size; + + if (actual_response_size < sizeof(struct fuse_compound_out)) { + pr_info_ratelimited("FUSE: compound response too small (%zu bytes, minimum %zu bytes)\n", + actual_response_size, + sizeof(struct fuse_compound_out)); + ret = -EINVAL; + goto out; + } + + ret = fuse_compound_parse_resp(compound, compound->result_header.count, + (char *)resp_payload, + actual_response_size); +out: + kvfree(resp_payload); +out_free_buffer: + kvfree(buffer); + return ret; +} diff --git a/fs/fuse/dev.c b/fs/fuse/dev.c index 0b0241f47170d4..90117fb13ec8e0 100644 --- a/fs/fuse/dev.c +++ b/fs/fuse/dev.c @@ -23,7 +23,7 @@ #include #include #include -#include +#include #include "fuse_trace.h" @@ -418,6 +418,9 @@ static void fuse_send_one(struct fuse_iqueue *fiq, struct fuse_req *req) req->in.h.len = sizeof(struct fuse_in_header) + fuse_len_args(req->args->in_numargs, (struct fuse_arg *) req->args->in_args); + + /* enqueue, as it is send to "fiq->ops queue" */ + trace_fuse_request_enqueue(req); fiq->ops->send_req(fiq, req); } @@ -582,7 +585,8 @@ static void request_wait_answer(struct fuse_req *req) * Either request is already in userspace, or it was forced. * Wait it out. */ - wait_event(req->waitq, test_bit(FR_FINISHED, &req->flags)); + wait_event(req->waitq, + test_bit(FR_FINISHED, &req->flags)); } static void __fuse_request_send(struct fuse_req *req) @@ -660,6 +664,30 @@ static void fuse_args_to_req(struct fuse_req *req, struct fuse_args *args) __set_bit(FR_ASYNC, &req->flags); } +ssize_t fuse_compound_request(struct fuse_mount *fm, struct fuse_args *args) +{ + struct fuse_req *req; + ssize_t ret; + + req = fuse_get_req(&invalid_mnt_idmap, fm, false); + if (IS_ERR(req)) + return PTR_ERR(req); + + fuse_args_to_req(req, args); + + if (!args->noreply) + __set_bit(FR_ISREPLY, &req->flags); + + __fuse_request_send(req); + ret = req->out.h.error; + if (!ret && args->out_argvar) { + BUG_ON(args->out_numargs == 0); + ret = args->out_args[args->out_numargs - 1].size; + } + fuse_put_request(req); + return ret; +} + ssize_t __fuse_simple_request(struct mnt_idmap *idmap, struct fuse_mount *fm, struct fuse_args *args) @@ -732,6 +760,8 @@ static int fuse_request_queue_background(struct fuse_req *req) } __set_bit(FR_ISREPLY, &req->flags); + trace_fuse_request_bg_enqueue(req); + #ifdef CONFIG_FUSE_IO_URING if (fuse_uring_ready(fc)) return fuse_request_queue_background_uring(fc, req); @@ -800,38 +830,14 @@ static int fuse_simple_notify_reply(struct fuse_mount *fm, return 0; } -/* - * Lock the request. Up to the next unlock_request() there mustn't be - * anything that could cause a page-fault. If the request was already - * aborted bail out. - */ -static int lock_request(struct fuse_req *req) -{ - int err = 0; - if (req) { - spin_lock(&req->waitq.lock); - if (test_bit(FR_ABORTED, &req->flags)) - err = -ENOENT; - else - set_bit(FR_LOCKED, &req->flags); - spin_unlock(&req->waitq.lock); - } - return err; -} -/* - * Unlock request. If it was aborted while locked, caller is responsible - * for unlocking and ending the request. - */ -static int unlock_request(struct fuse_req *req) +static int check_req_aborted(struct fuse_req *req) { int err = 0; - if (req) { + if (req && test_bit(FR_ABORTED, &req->flags)) { spin_lock(&req->waitq.lock); if (test_bit(FR_ABORTED, &req->flags)) err = -ENOENT; - else - clear_bit(FR_LOCKED, &req->flags); spin_unlock(&req->waitq.lock); } return err; @@ -873,7 +879,7 @@ static int fuse_copy_fill(struct fuse_copy_state *cs) struct page *page; int err; - err = unlock_request(cs->req); + err = check_req_aborted(cs->req); if (err) return err; @@ -912,6 +918,15 @@ static int fuse_copy_fill(struct fuse_copy_state *cs) cs->pipebufs++; cs->nr_segs++; } + } else if (cs->ring.pages) { + cs->pg = cs->ring.pages[cs->ring.page_idx++]; + /* + * non stricly needed, just to avoid a uring exception in + * fuse_copy_finish + */ + get_page(cs->pg); + cs->len = PAGE_SIZE; + cs->offset = 0; } else { size_t off; err = iov_iter_get_pages2(cs->iter, &page, PAGE_SIZE, 1, &off); @@ -923,7 +938,7 @@ static int fuse_copy_fill(struct fuse_copy_state *cs) cs->pg = page; } - return lock_request(cs->req); + return 0; } /* Do as much copy to/from userspace buffer as we can */ @@ -984,9 +999,6 @@ static int fuse_try_move_folio(struct fuse_copy_state *cs, struct folio **foliop struct pipe_buffer *buf = cs->pipebufs; folio_get(oldfolio); - err = unlock_request(cs->req); - if (err) - goto out_put_old; fuse_copy_finish(cs); @@ -1072,9 +1084,7 @@ static int fuse_try_move_folio(struct fuse_copy_state *cs, struct folio **foliop cs->pg = buf->page; cs->offset = buf->offset; - err = lock_request(cs->req); - if (!err) - err = 1; + err = 1; goto out_put_old; } @@ -1083,17 +1093,11 @@ static int fuse_ref_folio(struct fuse_copy_state *cs, struct folio *folio, unsigned offset, unsigned count) { struct pipe_buffer *buf; - int err; if (cs->nr_segs >= cs->pipe->max_usage) return -EIO; folio_get(folio); - err = unlock_request(cs->req); - if (err) { - folio_put(folio); - return err; - } fuse_copy_finish(cs); @@ -1467,6 +1471,7 @@ static ssize_t fuse_dev_do_read(struct fuse_dev *fud, struct file *file, clear_bit(FR_PENDING, &req->flags); list_del_init(&req->list); spin_unlock(&fiq->lock); + trace_fuse_request_send(req); args = req->args; reqsize = req->in.h.len; @@ -2430,6 +2435,45 @@ static void end_polls(struct fuse_conn *fc) } } +/* + * Flush all pending requests and wait for them. Only call this function when + * it is no longer possible for other threads to add requests. + */ +void fuse_flush_requests(struct fuse_conn *fc, unsigned long timeout) +{ + unsigned long deadline; + + spin_lock(&fc->lock); + if (!fc->connected) { + spin_unlock(&fc->lock); + return; + } + + /* Push all the background requests to the queue. */ + spin_lock(&fc->bg_lock); + fc->blocked = 0; + fc->max_background = UINT_MAX; + flush_bg_queue(fc); + spin_unlock(&fc->bg_lock); + spin_unlock(&fc->lock); + + fuse_uring_flush_bg(fc); + + /* + * Wait 30s for all the events to complete or abort. Touch the + * watchdog once per second so that we don't trip the hangcheck timer + * while waiting for the fuse server. + */ + deadline = jiffies + timeout; + smp_mb(); + while (fc->connected && + (!timeout || time_before(jiffies, deadline)) && + wait_event_timeout(fc->blocked_waitq, + !fc->connected || atomic_read(&fc->num_waiting) == 0, + HZ) == 0) + touch_softlockup_watchdog(); +} + /* * Abort all requests. * @@ -2522,11 +2566,119 @@ void fuse_abort_conn(struct fuse_conn *fc) } EXPORT_SYMBOL_GPL(fuse_abort_conn); +static void fuse_debug_print_outstanding_reqs(struct fuse_conn *fc) +{ + struct fuse_dev *fud; + + pr_warn("FUSE: fuse_wait_aborted: num_waiting=%d (should be 0)\n", + atomic_read(&fc->num_waiting)); + +#ifdef CONFIG_FUSE_IO_URING + /* Print io_uring state if enabled */ + if (fc->ring) { + struct fuse_ring *ring = fc->ring; + + pr_warn("FUSE: io_uring enabled - queue_refs=%d ready=%d\n", + atomic_read(&ring->queue_refs), ring->ready); + } +#endif + + /* Print all outstanding requests - lockless for debug */ + list_for_each_entry(fud, &fc->devices, entry) { + struct fuse_pqueue *fpq = &fud->pq; + struct fuse_req *req; + int i; + + /* Print all requests on fpq->io */ + if (!list_empty(&fpq->io)) { + pr_warn("FUSE: Outstanding requests on fpq->io:\n"); + list_for_each_entry(req, &fpq->io, list) { +#ifdef CONFIG_FUSE_IO_URING + if (test_bit(FR_URING, &req->flags) && + req->ring_entry) { + struct fuse_ring_ent *ent = req->ring_entry; + + pr_warn(" req %p: opcode=%u unique=%llu flags=0x%lx FR_WAITING=%d FR_LOCKED=%d FR_FORCE=%d FR_ABORTED=%d FR_URING=%d ring_ent=%p state=%d\n", + req, req->in.h.opcode, + req->in.h.unique, req->flags, + test_bit(FR_WAITING, &req->flags), + test_bit(FR_LOCKED, &req->flags), + test_bit(FR_FORCE, &req->flags), + test_bit(FR_ABORTED, &req->flags), + test_bit(FR_URING, &req->flags), + ent, ent->state); + } else { +#endif + pr_warn(" req %p: opcode=%u unique=%llu flags=0x%lx FR_WAITING=%d FR_LOCKED=%d FR_FORCE=%d FR_ABORTED=%d FR_URING=%d\n", + req, req->in.h.opcode, + req->in.h.unique, req->flags, + test_bit(FR_WAITING, &req->flags), + test_bit(FR_LOCKED, &req->flags), + test_bit(FR_FORCE, &req->flags), + test_bit(FR_ABORTED, &req->flags), + test_bit(FR_URING, &req->flags)); +#ifdef CONFIG_FUSE_IO_URING + } +#endif + } + } + + /* Print all requests on fpq->processing */ + for (i = 0; i < FUSE_PQ_HASH_SIZE; i++) { + if (list_empty(&fpq->processing[i])) + continue; + + pr_warn("FUSE: Outstanding requests on fpq->processing[%d]:\n", + i); + list_for_each_entry(req, &fpq->processing[i], list) { +#ifdef CONFIG_FUSE_IO_URING + if (test_bit(FR_URING, &req->flags) && + req->ring_entry) { + struct fuse_ring_ent *ent = req->ring_entry; + + pr_warn(" req %p: opcode=%u unique=%llu flags=0x%lx FR_WAITING=%d FR_LOCKED=%d FR_FORCE=%d FR_ABORTED=%d FR_URING=%d ring_ent=%p state=%d\n", + req, req->in.h.opcode, + req->in.h.unique, req->flags, + test_bit(FR_WAITING, &req->flags), + test_bit(FR_LOCKED, &req->flags), + test_bit(FR_FORCE, &req->flags), + test_bit(FR_ABORTED, &req->flags), + test_bit(FR_URING, &req->flags), + ent, ent->state); + } else { +#endif + pr_warn(" req %p: opcode=%u unique=%llu flags=0x%lx FR_WAITING=%d FR_LOCKED=%d FR_FORCE=%d FR_ABORTED=%d FR_URING=%d\n", + req, req->in.h.opcode, + req->in.h.unique, req->flags, + test_bit(FR_WAITING, &req->flags), + test_bit(FR_LOCKED, &req->flags), + test_bit(FR_FORCE, &req->flags), + test_bit(FR_ABORTED, &req->flags), + test_bit(FR_URING, &req->flags)); +#ifdef CONFIG_FUSE_IO_URING + } +#endif + } + } + } +} + void fuse_wait_aborted(struct fuse_conn *fc) { + unsigned int timeout = 20; + /* matches implicit memory barrier in fuse_drop_waiting() */ smp_mb(); - wait_event(fc->blocked_waitq, atomic_read(&fc->num_waiting) == 0); + +wait: + wait_event_timeout(fc->blocked_waitq, atomic_read(&fc->num_waiting) == 0, HZ * timeout); + + /* Debug: print info if we're waiting */ + if (atomic_read(&fc->num_waiting) > 0) { + fuse_debug_print_outstanding_reqs(fc); + timeout *= 3; + goto wait; + } fuse_uring_wait_stopped_queues(fc); } diff --git a/fs/fuse/dev_uring.c b/fs/fuse/dev_uring.c index 3a38b61aac26f7..7fa79e68afd5c9 100644 --- a/fs/fuse/dev_uring.c +++ b/fs/fuse/dev_uring.c @@ -11,6 +11,7 @@ #include #include +#include static bool __read_mostly enable_uring; module_param(enable_uring, bool, 0644); @@ -18,6 +19,14 @@ MODULE_PARM_DESC(enable_uring, "Enable userspace communication through io-uring"); #define FUSE_URING_IOV_SEGS 2 /* header and payload */ +#define FUSE_RING_HEADER_PG 0 +#define FUSE_RING_PAYLOAD_PG 1 + +/* Threshold that determines if a better queue should be searched for */ +#define FUSE_URING_Q_THRESHOLD 2 + +/* Number of (re)tries to find a better queue */ +#define FUSE_URING_Q_TRIES 3 bool fuse_uring_enabled(void) @@ -48,7 +57,7 @@ static struct fuse_ring_ent *uring_cmd_to_ring_ent(struct io_uring_cmd *cmd) return pdu->ent; } -static void fuse_uring_flush_bg(struct fuse_ring_queue *queue) +static void fuse_uring_flush_queue_bg(struct fuse_ring_queue *queue) { struct fuse_ring *ring = queue->ring; struct fuse_conn *fc = ring->fc; @@ -86,14 +95,14 @@ static void fuse_uring_req_end(struct fuse_ring_ent *ent, struct fuse_req *req, lockdep_assert_not_held(&queue->lock); spin_lock(&queue->lock); ent->fuse_req = NULL; + queue->nr_reqs--; list_del_init(&req->list); if (test_bit(FR_BACKGROUND, &req->flags)) { queue->active_background--; spin_lock(&fc->bg_lock); - fuse_uring_flush_bg(queue); + fuse_uring_flush_queue_bg(queue); spin_unlock(&fc->bg_lock); } - spin_unlock(&queue->lock); if (error) @@ -113,32 +122,69 @@ static void fuse_uring_abort_end_queue_requests(struct fuse_ring_queue *queue) list_for_each_entry(req, &queue->fuse_req_queue, list) clear_bit(FR_PENDING, &req->flags); list_splice_init(&queue->fuse_req_queue, &req_list); + queue->nr_reqs = 0; spin_unlock(&queue->lock); /* must not hold queue lock to avoid order issues with fi->lock */ fuse_dev_end_requests(&req_list); } -void fuse_uring_abort_end_requests(struct fuse_ring *ring) +void fuse_uring_flush_bg(struct fuse_conn *fc) { int qid; struct fuse_ring_queue *queue; - struct fuse_conn *fc = ring->fc; + struct fuse_ring *ring = fc->ring; + + if (!ring) + return; - for (qid = 0; qid < ring->nr_queues; qid++) { + for (qid = 0; qid < ring->max_nr_queues; qid++) { queue = READ_ONCE(ring->queues[qid]); if (!queue) continue; - queue->stopped = true; - WARN_ON_ONCE(ring->fc->max_background != UINT_MAX); spin_lock(&queue->lock); spin_lock(&fc->bg_lock); - fuse_uring_flush_bg(queue); + queue->stopped = true; + fuse_uring_flush_queue_bg(queue); spin_unlock(&fc->bg_lock); spin_unlock(&queue->lock); - fuse_uring_abort_end_queue_requests(queue); + } +} + +/* + * Copy from memmap.c, should be exported + */ +static void io_pages_free(struct page ***pages, int npages) +{ + struct page **page_array = *pages; + + if (!page_array) + return; + + unpin_user_pages(page_array, npages); + kvfree(page_array); + *pages = NULL; +} + + +static void fuse_ring_destruct_q_map(struct fuse_queue_map *q_map) +{ + free_cpumask_var(q_map->registered_q_mask); + kfree(q_map->cpu_to_qid); +} + +static void fuse_uring_destruct_q_masks(struct fuse_ring *ring) +{ + int node; + + fuse_ring_destruct_q_map(&ring->q_map); + + if (ring->numa_q_map) { + for (node = 0; node < ring->nr_numa_nodes; node++) + fuse_ring_destruct_q_map(&ring->numa_q_map[node]); + kfree(ring->numa_q_map); } } @@ -166,7 +212,7 @@ bool fuse_uring_request_expired(struct fuse_conn *fc) if (!ring) return false; - for (qid = 0; qid < ring->nr_queues; qid++) { + for (qid = 0; qid < ring->max_nr_queues; qid++) { queue = READ_ONCE(ring->queues[qid]); if (!queue) continue; @@ -193,7 +239,7 @@ void fuse_uring_destruct(struct fuse_conn *fc) if (!ring) return; - for (qid = 0; qid < ring->nr_queues; qid++) { + for (qid = 0; qid < ring->max_nr_queues; qid++) { struct fuse_ring_queue *queue = ring->queues[qid]; struct fuse_ring_ent *ent, *next; @@ -208,6 +254,9 @@ void fuse_uring_destruct(struct fuse_conn *fc) list_for_each_entry_safe(ent, next, &queue->ent_released, list) { list_del_init(&ent->list); + io_pages_free(&ent->header_pages, ent->nr_header_pages); + io_pages_free(&ent->payload_pages, + ent->nr_payload_pages); kfree(ent); } @@ -216,11 +265,47 @@ void fuse_uring_destruct(struct fuse_conn *fc) ring->queues[qid] = NULL; } + fuse_uring_destruct_q_masks(ring); kfree(ring->queues); kfree(ring); fc->ring = NULL; } +static int fuse_uring_init_q_map(struct fuse_queue_map *q_map, size_t nr_cpu) +{ + if (!zalloc_cpumask_var(&q_map->registered_q_mask, GFP_KERNEL_ACCOUNT)) + return -ENOMEM; + + q_map->cpu_to_qid = kcalloc(nr_cpu, sizeof(*q_map->cpu_to_qid), + GFP_KERNEL_ACCOUNT); + if (!q_map->cpu_to_qid) + return -ENOMEM; + + return 0; +} + +static int fuse_uring_create_q_masks(struct fuse_ring *ring, size_t nr_queues) +{ + int err, node; + + err = fuse_uring_init_q_map(&ring->q_map, nr_queues); + if (err) + return err; + + ring->numa_q_map = kcalloc(ring->nr_numa_nodes, + sizeof(*ring->numa_q_map), + GFP_KERNEL_ACCOUNT); + if (!ring->numa_q_map) + return -ENOMEM; + for (node = 0; node < ring->nr_numa_nodes; node++) { + err = fuse_uring_init_q_map(&ring->numa_q_map[node], + nr_queues); + if (err) + return err; + } + return 0; +} + /* * Basic ring setup for this connection based on the provided configuration */ @@ -230,19 +315,26 @@ static struct fuse_ring *fuse_uring_create(struct fuse_conn *fc) size_t nr_queues = num_possible_cpus(); struct fuse_ring *res = NULL; size_t max_payload_size; + int err; ring = kzalloc_obj(*fc->ring, GFP_KERNEL_ACCOUNT); if (!ring) return NULL; - ring->queues = kzalloc_objs(struct fuse_ring_queue *, nr_queues, - GFP_KERNEL_ACCOUNT); + ring->nr_numa_nodes = num_online_nodes(); + + ring->queues = kcalloc(nr_queues, sizeof(struct fuse_ring_queue *), + GFP_KERNEL_ACCOUNT); if (!ring->queues) goto out_err; max_payload_size = max(FUSE_MIN_READ_BUFFER, fc->max_write); max_payload_size = max(max_payload_size, fc->max_pages * PAGE_SIZE); + err = fuse_uring_create_q_masks(ring, nr_queues); + if (err) + goto out_err; + spin_lock(&fc->lock); if (fc->ring) { /* race, another thread created the ring in the meantime */ @@ -253,7 +345,7 @@ static struct fuse_ring *fuse_uring_create(struct fuse_conn *fc) init_waitqueue_head(&ring->stop_waitq); - ring->nr_queues = nr_queues; + ring->max_nr_queues = nr_queues; ring->fc = fc; ring->max_payload_sz = max_payload_size; smp_store_release(&fc->ring, ring); @@ -262,17 +354,42 @@ static struct fuse_ring *fuse_uring_create(struct fuse_conn *fc) return ring; out_err: + fuse_uring_destruct_q_masks(ring); kfree(ring->queues); kfree(ring); return res; } +static void fuse_uring_cpu_qid_mapping(struct fuse_ring *ring, int qid, + struct fuse_queue_map *q_map, + int node) +{ + int cpu, qid_idx, mapping_count = 0; + size_t nr_queues; + + cpumask_set_cpu(qid, q_map->registered_q_mask); + nr_queues = cpumask_weight(q_map->registered_q_mask); + for (cpu = 0; cpu < ring->max_nr_queues; cpu++) { + if (node != -1 && cpu_to_node(cpu) != node) + continue; + + qid_idx = mapping_count % nr_queues; + q_map->cpu_to_qid[cpu] = cpumask_nth(qid_idx, + q_map->registered_q_mask); + mapping_count++; + pr_debug("%s node=%d qid=%d qid_idx=%d nr_queues=%zu %d->%d\n", + __func__, node, qid, qid_idx, nr_queues, cpu, + q_map->cpu_to_qid[cpu]); + } +} + static struct fuse_ring_queue *fuse_uring_create_queue(struct fuse_ring *ring, int qid) { struct fuse_conn *fc = ring->fc; struct fuse_ring_queue *queue; struct list_head *pq; + int node; queue = kzalloc_obj(*queue, GFP_KERNEL_ACCOUNT); if (!queue) @@ -310,6 +427,22 @@ static struct fuse_ring_queue *fuse_uring_create_queue(struct fuse_ring *ring, * write_once and lock as the caller mostly doesn't take the lock at all */ WRITE_ONCE(ring->queues[qid], queue); + + /* Static mapping from cpu to per numa queues */ + node = cpu_to_node(qid); + fuse_uring_cpu_qid_mapping(ring, qid, &ring->numa_q_map[node], node); + + /* + * smp_store_release, as the variable is read without fc->lock and + * we need to avoid compiler re-ordering of updating the nr_queues + * and setting ring->numa_queues[node].cpu_to_qid above + */ + smp_store_release (&ring->numa_q_map[node].nr_queues, + ring->numa_q_map[node].nr_queues + 1); + + /* global mapping */ + fuse_uring_cpu_qid_mapping(ring, qid, &ring->q_map, -1); + spin_unlock(&fc->lock); return queue; @@ -325,11 +458,11 @@ static void fuse_uring_stop_fuse_req_end(struct fuse_req *req) /* * Release a request/entry on connection tear down */ -static void fuse_uring_entry_teardown(struct fuse_ring_ent *ent) +static void fuse_uring_entry_teardown(struct fuse_ring_ent *ent, int issue_flags) { struct fuse_req *req; struct io_uring_cmd *cmd; - + ssize_t queue_refs; struct fuse_ring_queue *queue = ent->queue; spin_lock(&queue->lock); @@ -353,19 +486,20 @@ static void fuse_uring_entry_teardown(struct fuse_ring_ent *ent) spin_unlock(&queue->lock); if (cmd) - io_uring_cmd_done(cmd, -ENOTCONN, IO_URING_F_UNLOCKED); + io_uring_cmd_done(cmd, -ENOTCONN, issue_flags); if (req) fuse_uring_stop_fuse_req_end(req); + + queue_refs = atomic_dec_return(&queue->ring->queue_refs); + WARN_ON_ONCE(queue_refs < 0); } static void fuse_uring_stop_list_entries(struct list_head *head, struct fuse_ring_queue *queue, enum fuse_ring_req_state exp_state) { - struct fuse_ring *ring = queue->ring; struct fuse_ring_ent *ent, *next; - ssize_t queue_refs = SSIZE_MAX; LIST_HEAD(to_teardown); spin_lock(&queue->lock); @@ -382,11 +516,8 @@ static void fuse_uring_stop_list_entries(struct list_head *head, spin_unlock(&queue->lock); /* no queue lock to avoid lock order issues */ - list_for_each_entry_safe(ent, next, &to_teardown, list) { - fuse_uring_entry_teardown(ent); - queue_refs = atomic_dec_return(&ring->queue_refs); - WARN_ON_ONCE(queue_refs < 0); - } + list_for_each_entry_safe(ent, next, &to_teardown, list) + fuse_uring_entry_teardown(ent, IO_URING_F_UNLOCKED); } static void fuse_uring_teardown_entries(struct fuse_ring_queue *queue) @@ -405,7 +536,7 @@ static void fuse_uring_log_ent_state(struct fuse_ring *ring) int qid; struct fuse_ring_ent *ent; - for (qid = 0; qid < ring->nr_queues; qid++) { + for (qid = 0; qid < ring->max_nr_queues; qid++) { struct fuse_ring_queue *queue = ring->queues[qid]; if (!queue) @@ -424,6 +555,7 @@ static void fuse_uring_log_ent_state(struct fuse_ring *ring) pr_info(" ent-commit-queue ring=%p qid=%d ent=%p state=%d\n", ring, qid, ent, ent->state); } + spin_unlock(&queue->lock); } ring->stop_debug_log = 1; @@ -436,7 +568,7 @@ static void fuse_uring_async_stop_queues(struct work_struct *work) container_of(work, struct fuse_ring, async_teardown_work.work); /* XXX code dup */ - for (qid = 0; qid < ring->nr_queues; qid++) { + for (qid = 0; qid < ring->max_nr_queues; qid++) { struct fuse_ring_queue *queue = READ_ONCE(ring->queues[qid]); if (!queue) @@ -470,16 +602,25 @@ static void fuse_uring_async_stop_queues(struct work_struct *work) void fuse_uring_stop_queues(struct fuse_ring *ring) { int qid; + int node; - for (qid = 0; qid < ring->nr_queues; qid++) { + for (qid = 0; qid < ring->max_nr_queues; qid++) { struct fuse_ring_queue *queue = READ_ONCE(ring->queues[qid]); if (!queue) continue; + fuse_uring_abort_end_queue_requests(queue); fuse_uring_teardown_entries(queue); } + /* Reset all queue masks, we won't process any more IO */ + cpumask_clear(ring->q_map.registered_q_mask); + for (node = 0; node < ring->nr_numa_nodes; node++) { + if (ring->numa_q_map) + cpumask_clear(ring->numa_q_map[node].registered_q_mask); + } + if (atomic_read(&ring->queue_refs) > 0) { ring->teardown_time = jiffies; INIT_DELAYED_WORK(&ring->async_teardown_work, @@ -502,7 +643,7 @@ static void fuse_uring_cancel(struct io_uring_cmd *cmd, { struct fuse_ring_ent *ent = uring_cmd_to_ring_ent(cmd); struct fuse_ring_queue *queue; - bool need_cmd_done = false; + bool teardown = false; /* * direct access on ent - it must not be destructed as long as @@ -511,17 +652,14 @@ static void fuse_uring_cancel(struct io_uring_cmd *cmd, queue = ent->queue; spin_lock(&queue->lock); if (ent->state == FRRS_AVAILABLE) { - ent->state = FRRS_USERSPACE; - list_move_tail(&ent->list, &queue->ent_in_userspace); - need_cmd_done = true; - ent->cmd = NULL; + ent->state = FRRS_TEARDOWN; + list_del_init(&ent->list); + teardown = true; } spin_unlock(&queue->lock); - if (need_cmd_done) { - /* no queue lock to avoid lock order issues */ - io_uring_cmd_done(cmd, -ENOTCONN, issue_flags); - } + if (teardown) + fuse_uring_entry_teardown(ent, issue_flags); } static void fuse_uring_prepare_cancel(struct io_uring_cmd *cmd, int issue_flags, @@ -598,12 +736,69 @@ static int fuse_uring_copy_from_ring(struct fuse_ring *ring, fuse_copy_init(&cs, false, &iter); cs.is_uring = true; cs.req = req; + if (ent->payload_pages) + cs.ring.pages = ent->payload_pages; err = fuse_copy_out_args(&cs, args, ring_in_out.payload_sz); fuse_copy_finish(&cs); return err; } +/* + * Copy data from the req to the ring buffer + * In order to be able to write into the ring buffer from the application, + * i.e. to avoid io_uring_cmd_complete_in_task(), the header needs to be + * pinned as well. + */ +static int fuse_uring_args_to_ring_pages(struct fuse_ring *ring, + struct fuse_req *req, + struct fuse_ring_ent *ent, + struct fuse_uring_req_header *headers) +{ + struct fuse_copy_state cs; + struct fuse_args *args = req->args; + struct fuse_in_arg *in_args = args->in_args; + int num_args = args->in_numargs; + int err; + + struct fuse_uring_ent_in_out ent_in_out = { + .flags = 0, + .commit_id = req->in.h.unique, + }; + + fuse_copy_init(&cs, 1, NULL); + cs.is_uring = 1; + cs.req = req; + cs.ring.pages = ent->payload_pages; + + if (num_args > 0) { + /* + * Expectation is that the first argument is the per op header. + * Some op code have that as zero size. + */ + if (args->in_args[0].size > 0) { + memcpy(&headers->op_in, in_args->value, in_args->size); + } + in_args++; + num_args--; + } + + /* copy the payload */ + err = fuse_copy_args(&cs, num_args, args->in_pages, + (struct fuse_arg *)in_args, 0); + if (err) { + pr_info_ratelimited("%s fuse_copy_args failed\n", __func__); + goto copy_finish; + } + + ent_in_out.payload_sz = cs.ring.copied_sz; + memcpy(&headers->ring_ent_in_out, &ent_in_out, sizeof(ent_in_out)); + +copy_finish: + fuse_copy_finish(&cs); + return err; +} + /* * Copy data from the req to the ring buffer */ @@ -630,6 +825,8 @@ static int fuse_uring_args_to_ring(struct fuse_ring *ring, struct fuse_req *req, fuse_copy_init(&cs, true, &iter); cs.is_uring = true; cs.req = req; + if (ent->payload_pages) + cs.ring.pages = ent->payload_pages; if (num_args > 0) { /* @@ -655,12 +852,14 @@ static int fuse_uring_args_to_ring(struct fuse_ring *ring, struct fuse_req *req, fuse_copy_finish(&cs); if (err) { pr_info_ratelimited("%s fuse_copy_args failed\n", __func__); - return err; + goto copy_finish; } ent_in_out.payload_sz = cs.ring.copied_sz; err = copy_to_user(&ent->headers->ring_ent_in_out, &ent_in_out, sizeof(ent_in_out)); +copy_finish: + fuse_copy_finish(&cs); return err ? -EFAULT : 0; } @@ -670,6 +869,7 @@ static int fuse_uring_copy_to_ring(struct fuse_ring_ent *ent, struct fuse_ring_queue *queue = ent->queue; struct fuse_ring *ring = queue->ring; int err; + struct fuse_uring_req_header *headers = NULL; err = -EIO; if (WARN_ON(ent->state != FRRS_FUSE_REQ)) { @@ -682,22 +882,29 @@ static int fuse_uring_copy_to_ring(struct fuse_ring_ent *ent, if (WARN_ON(req->in.h.unique == 0)) return err; - /* copy the request */ - err = fuse_uring_args_to_ring(ring, req, ent); - if (unlikely(err)) { - pr_info_ratelimited("Copy to ring failed: %d\n", err); - return err; - } - /* copy fuse_in_header */ - err = copy_to_user(&ent->headers->in_out, &req->in.h, - sizeof(req->in.h)); - if (err) { - err = -EFAULT; - return err; + if (ent->header_pages) { + headers = kmap_local_page( + ent->header_pages[FUSE_RING_HEADER_PG]); + + memcpy(&headers->in_out, &req->in.h, sizeof(req->in.h)); + + err = fuse_uring_args_to_ring_pages(ring, req, ent, headers); + kunmap_local(headers); + } else { + /* copy the request */ + err = fuse_uring_args_to_ring(ring, req, ent); + if (unlikely(err)) { + pr_info_ratelimited("Copy to ring failed: %d\n", err); + return err; + } + err = copy_to_user(&ent->headers->in_out, &req->in.h, + sizeof(req->in.h)); + if (err) + err = -EFAULT; } - return 0; + return err; } static int fuse_uring_prepare_send(struct fuse_ring_ent *ent, @@ -894,7 +1101,7 @@ static int fuse_uring_commit_fetch(struct io_uring_cmd *cmd, int issue_flags, if (!ring) return err; - if (qid >= ring->nr_queues) + if (qid >= ring->max_nr_queues) return -EINVAL; queue = ring->queues[qid]; @@ -951,59 +1158,43 @@ static int fuse_uring_commit_fetch(struct io_uring_cmd *cmd, int issue_flags, return 0; } -static bool is_ring_ready(struct fuse_ring *ring, int current_qid) -{ - int qid; - struct fuse_ring_queue *queue; - bool ready = true; - - for (qid = 0; qid < ring->nr_queues && ready; qid++) { - if (current_qid == qid) - continue; - - queue = ring->queues[qid]; - if (!queue) { - ready = false; - break; - } - - spin_lock(&queue->lock); - if (list_empty(&queue->ent_avail_queue)) - ready = false; - spin_unlock(&queue->lock); - } - - return ready; -} - /* - * fuse_uring_req_fetch command handling + * Copy from memmap.c, should be exported there */ -static void fuse_uring_do_register(struct fuse_ring_ent *ent, - struct io_uring_cmd *cmd, - unsigned int issue_flags) +static struct page **io_pin_pages(unsigned long uaddr, unsigned long len, + int *npages) { - struct fuse_ring_queue *queue = ent->queue; - struct fuse_ring *ring = queue->ring; - struct fuse_conn *fc = ring->fc; - struct fuse_iqueue *fiq = &fc->iq; - - fuse_uring_prepare_cancel(cmd, issue_flags, ent); - - spin_lock(&queue->lock); - ent->cmd = cmd; - fuse_uring_ent_avail(ent, queue); - spin_unlock(&queue->lock); - - if (!ring->ready) { - bool ready = is_ring_ready(ring, queue->qid); + unsigned long start, end, nr_pages; + struct page **pages; + int ret; + + end = (uaddr + len + PAGE_SIZE - 1) >> PAGE_SHIFT; + start = uaddr >> PAGE_SHIFT; + nr_pages = end - start; + if (WARN_ON_ONCE(!nr_pages)) + return ERR_PTR(-EINVAL); + + pages = kvmalloc_array(nr_pages, sizeof(struct page *), GFP_KERNEL); + if (!pages) + return ERR_PTR(-ENOMEM); + + ret = pin_user_pages_fast(uaddr, nr_pages, FOLL_WRITE | FOLL_LONGTERM, + pages); + /* success, mapped all pages */ + if (ret == nr_pages) { + *npages = nr_pages; + return pages; + } - if (ready) { - WRITE_ONCE(fiq->ops, &fuse_io_uring_ops); - WRITE_ONCE(ring->ready, true); - wake_up_all(&fc->blocked_waitq); - } + /* partial map, or didn't map anything */ + if (ret >= 0) { + /* if we did partial map, release any pages we did get */ + if (ret) + unpin_user_pages(pages, ret); + ret = -EFAULT; } + kvfree(pages); + return ERR_PTR(ret); } /* @@ -1032,6 +1223,59 @@ static int fuse_uring_get_iovec_from_sqe(const struct io_uring_sqe *sqe, return 0; } +static int fuse_uring_pin_pages(struct fuse_ring_ent *ent) +{ + struct fuse_ring *ring = ent->queue->ring; + int err; + + /* + * This needs to do locked memory accounting, for now privileged servers + * only. + */ + if (!capable(CAP_SYS_ADMIN)) + return 0; + + /* Pin header pages */ + if (!PAGE_ALIGNED(ent->headers)) { + pr_info_ratelimited("ent->headers is not page-aligned: %p\n", + ent->headers); + return -EINVAL; + } + + ent->header_pages = io_pin_pages((unsigned long)ent->headers, + sizeof(struct fuse_uring_req_header), + &ent->nr_header_pages); + if (IS_ERR(ent->header_pages)) { + err = PTR_ERR(ent->header_pages); + pr_info_ratelimited("Failed to pin header pages, err=%d\n", + err); + ent->header_pages = NULL; + return err; + } + + if (ent->nr_header_pages != 1) { + pr_info_ratelimited("Header pages not pinned as one page\n"); + io_pages_free(&ent->header_pages, ent->nr_header_pages); + ent->header_pages = NULL; + return -EINVAL; + } + + /* Pin payload pages */ + ent->payload_pages = io_pin_pages((unsigned long)ent->payload, + ring->max_payload_sz, + &ent->nr_payload_pages); + if (IS_ERR(ent->payload_pages)) { + err = PTR_ERR(ent->payload_pages); + pr_info_ratelimited("Failed to pin payload pages, err=%d\n", + err); + io_pages_free(&ent->header_pages, ent->nr_header_pages); + ent->payload_pages = NULL; + return err; + } + + return 0; +} + static struct fuse_ring_ent * fuse_uring_create_ring_ent(struct io_uring_cmd *cmd, struct fuse_ring_queue *queue) @@ -1073,6 +1317,12 @@ fuse_uring_create_ring_ent(struct io_uring_cmd *cmd, ent->headers = iov[0].iov_base; ent->payload = iov[1].iov_base; + err = fuse_uring_pin_pages(ent); + if (err) { + kfree(ent); + return ERR_PTR(err); + } + atomic_inc(&ring->queue_refs); return ent; } @@ -1089,6 +1339,7 @@ static int fuse_uring_register(struct io_uring_cmd *cmd, struct fuse_ring *ring = smp_load_acquire(&fc->ring); struct fuse_ring_queue *queue; struct fuse_ring_ent *ent; + struct fuse_iqueue *fiq = &fc->iq; int err; unsigned int qid = READ_ONCE(cmd_req->qid); @@ -1099,7 +1350,7 @@ static int fuse_uring_register(struct io_uring_cmd *cmd, return err; } - if (qid >= ring->nr_queues) { + if (qid >= ring->max_nr_queues) { pr_info_ratelimited("fuse: Invalid ring qid %u\n", qid); return -EINVAL; } @@ -1120,7 +1371,19 @@ static int fuse_uring_register(struct io_uring_cmd *cmd, if (IS_ERR(ent)) return PTR_ERR(ent); - fuse_uring_do_register(ent, cmd, issue_flags); + fuse_uring_prepare_cancel(cmd, issue_flags, ent); + if (!ring->ready) { + WRITE_ONCE(fiq->ops, &fuse_io_uring_ops); + WRITE_ONCE(ring->ready, true); + wake_up_all(&fc->blocked_waitq); + } + + spin_lock(&queue->lock); + ent->cmd = cmd; + spin_unlock(&queue->lock); + + /* Marks the ring entry as ready */ + fuse_uring_next_fuse_req(ent, queue, issue_flags); return 0; } @@ -1207,6 +1470,7 @@ static void fuse_uring_send(struct fuse_ring_ent *ent, struct io_uring_cmd *cmd, ent->cmd = NULL; spin_unlock(&queue->lock); + trace_fuse_request_send(ent->fuse_req); io_uring_cmd_done(cmd, ret, issue_flags); } @@ -1236,30 +1500,107 @@ static void fuse_uring_send_in_task(struct io_tw_req tw_req, io_tw_token_t tw) fuse_uring_send(ent, cmd, err, issue_flags); } -static struct fuse_ring_queue *fuse_uring_task_to_queue(struct fuse_ring *ring) +static struct fuse_ring_queue *fuse_uring_select_queue(struct fuse_ring *ring, + bool background) { unsigned int qid; - struct fuse_ring_queue *queue; + int node, tries = 0; + unsigned int nr_queues; + unsigned int cpu = task_cpu(current); + struct fuse_ring_queue *queue, *primary_queue = NULL; - qid = task_cpu(current); + /* + * Background requests result in better performance on a different + * CPU, unless CPUs are already busy. + */ + if (background) + cpu++; - if (WARN_ONCE(qid >= ring->nr_queues, - "Core number (%u) exceeds nr queues (%zu)\n", qid, - ring->nr_queues)) - qid = 0; +retry: + cpu = cpu % ring->max_nr_queues; + + /* numa local registered queue bitmap */ + node = cpu_to_node(cpu); + if (WARN_ONCE(node >= ring->nr_numa_nodes, + "Node number (%d) exceeds nr nodes (%d)\n", + node, ring->nr_numa_nodes)) { + node = 0; + } - queue = ring->queues[qid]; - WARN_ONCE(!queue, "Missing queue for qid %d\n", qid); + nr_queues = READ_ONCE(ring->numa_q_map[node].nr_queues); + if (nr_queues) { + /* prefer the queue that corresponds to the current cpu */ + queue = READ_ONCE(ring->queues[cpu]); + if (queue) { + if (queue->nr_reqs <= FUSE_URING_Q_THRESHOLD) + return queue; + primary_queue = queue; + } - return queue; + qid = ring->numa_q_map[node].cpu_to_qid[cpu]; + if (WARN_ON_ONCE(qid >= ring->max_nr_queues)) + return NULL; + if (qid != cpu) { + queue = READ_ONCE(ring->queues[qid]); + + /* Might happen on teardown */ + if (unlikely(!queue)) + return NULL; + + if (queue->nr_reqs <= FUSE_URING_Q_THRESHOLD) + return queue; + } + + /* Retries help for load balancing */ + if (tries < FUSE_URING_Q_TRIES && tries + 1 < nr_queues) { + if (!primary_queue) + primary_queue = queue; + + /* Increase cpu, assuming it will map to a different qid*/ + cpu++; + tries++; + goto retry; + } + } + + /* Retries exceeded, take the primary target queue */ + if (primary_queue) + return primary_queue; + + /* global registered queue bitmap */ + qid = ring->q_map.cpu_to_qid[cpu]; + if (WARN_ON_ONCE(qid >= ring->max_nr_queues)) { + /* Might happen on teardown */ + return NULL; + } + return READ_ONCE(ring->queues[qid]); } -static void fuse_uring_dispatch_ent(struct fuse_ring_ent *ent) +static void fuse_uring_dispatch_ent(struct fuse_ring_ent *ent, bool bg) { struct io_uring_cmd *cmd = ent->cmd; - uring_cmd_set_ring_ent(cmd, ent); - io_uring_cmd_complete_in_task(cmd, fuse_uring_send_in_task); + /* + * Task needed when pages are not pinned as the application doing IO + * is not allowed to write into fuse-server pages. + * Additionally for IO through io-uring as issue flags are unknown then. + * backgrounds requests might hold spin-locks, that conflict with + * io_uring_cmd_done() mutex lock. + */ + if (!ent->header_pages || current->io_uring || bg) { + uring_cmd_set_ring_ent(cmd, ent); + io_uring_cmd_complete_in_task(cmd, fuse_uring_send_in_task); + } else { + int err = fuse_uring_prepare_send(ent, ent->fuse_req); + struct fuse_ring_queue *queue = ent->queue; + + if (err) { + fuse_uring_next_fuse_req(ent, queue, + IO_URING_F_UNLOCKED); + return; + } + fuse_uring_send(ent, cmd, 0, IO_URING_F_UNLOCKED); + } } /* queue a fuse request and send it if a ring entry is available */ @@ -1272,7 +1613,7 @@ void fuse_uring_queue_fuse_req(struct fuse_iqueue *fiq, struct fuse_req *req) int err; err = -EINVAL; - queue = fuse_uring_task_to_queue(ring); + queue = fuse_uring_select_queue(ring, false); if (!queue) goto err; @@ -1287,14 +1628,17 @@ void fuse_uring_queue_fuse_req(struct fuse_iqueue *fiq, struct fuse_req *req) req->ring_queue = queue; ent = list_first_entry_or_null(&queue->ent_avail_queue, struct fuse_ring_ent, list); + queue->nr_reqs++; + if (ent) fuse_uring_add_req_to_ring_ent(ent, req); else list_add_tail(&req->list, &queue->fuse_req_queue); + spin_unlock(&queue->lock); if (ent) - fuse_uring_dispatch_ent(ent); + fuse_uring_dispatch_ent(ent, false); return; @@ -1313,7 +1657,7 @@ bool fuse_uring_queue_bq_req(struct fuse_req *req) struct fuse_ring_queue *queue; struct fuse_ring_ent *ent = NULL; - queue = fuse_uring_task_to_queue(ring); + queue = fuse_uring_select_queue(ring, true); if (!queue) return false; @@ -1326,6 +1670,7 @@ bool fuse_uring_queue_bq_req(struct fuse_req *req) set_bit(FR_URING, &req->flags); req->ring_queue = queue; list_add_tail(&req->list, &queue->fuse_req_bg_queue); + queue->nr_reqs++; ent = list_first_entry_or_null(&queue->ent_avail_queue, struct fuse_ring_ent, list); @@ -1333,7 +1678,7 @@ bool fuse_uring_queue_bq_req(struct fuse_req *req) fc->num_background++; if (fc->num_background == fc->max_background) fc->blocked = 1; - fuse_uring_flush_bg(queue); + fuse_uring_flush_queue_bg(queue); spin_unlock(&fc->bg_lock); /* @@ -1347,7 +1692,7 @@ bool fuse_uring_queue_bq_req(struct fuse_req *req) fuse_uring_add_req_to_ring_ent(ent, req); spin_unlock(&queue->lock); - fuse_uring_dispatch_ent(ent); + fuse_uring_dispatch_ent(ent, true); } else { spin_unlock(&queue->lock); } @@ -1358,8 +1703,16 @@ bool fuse_uring_queue_bq_req(struct fuse_req *req) bool fuse_uring_remove_pending_req(struct fuse_req *req) { struct fuse_ring_queue *queue = req->ring_queue; + bool removed = fuse_remove_pending_req(req, &queue->lock); + + if (removed) { + /* Update counters after successful removal */ + spin_lock(&queue->lock); + queue->nr_reqs--; + spin_unlock(&queue->lock); + } - return fuse_remove_pending_req(req, &queue->lock); + return removed; } static const struct fuse_iqueue_ops fuse_io_uring_ops = { diff --git a/fs/fuse/dev_uring_i.h b/fs/fuse/dev_uring_i.h index 51a563922ce141..4518990e98bdd5 100644 --- a/fs/fuse/dev_uring_i.h +++ b/fs/fuse/dev_uring_i.h @@ -40,7 +40,11 @@ enum fuse_ring_req_state { struct fuse_ring_ent { /* userspace buffer */ struct fuse_uring_req_header __user *headers; + struct page **header_pages; + int nr_header_pages; void __user *payload; + struct page **payload_pages; + int nr_payload_pages; /* the ring queue that owns the request */ struct fuse_ring_queue *queue; @@ -94,6 +98,9 @@ struct fuse_ring_queue { /* background fuse requests */ struct list_head fuse_req_bg_queue; + /* number of requests queued or in userspace */ + unsigned int nr_reqs; + struct fuse_pqueue fpq; unsigned int active_background; @@ -101,6 +108,17 @@ struct fuse_ring_queue { bool stopped; }; +struct fuse_queue_map { + /* Tracks which queues are registered */ + cpumask_var_t registered_q_mask; + + /* number of registered queues */ + size_t nr_queues; + + /* cpu to qid mapping */ + int *cpu_to_qid; +}; + /** * Describes if uring is for communication and holds alls the data needed * for uring communication @@ -110,7 +128,10 @@ struct fuse_ring { struct fuse_conn *fc; /* number of ring queues */ - size_t nr_queues; + size_t max_nr_queues; + + /* number of numa nodes */ + int nr_numa_nodes; /* maximum payload/arg size */ size_t max_payload_sz; @@ -122,6 +143,12 @@ struct fuse_ring { */ unsigned int stop_debug_log : 1; + /* per numa node queue tracking */ + struct fuse_queue_map *numa_q_map; + + /* all queue tracking */ + struct fuse_queue_map q_map; + wait_queue_head_t stop_waitq; /* async tear down */ @@ -138,7 +165,7 @@ struct fuse_ring { bool fuse_uring_enabled(void); void fuse_uring_destruct(struct fuse_conn *fc); void fuse_uring_stop_queues(struct fuse_ring *ring); -void fuse_uring_abort_end_requests(struct fuse_ring *ring); +void fuse_uring_flush_bg(struct fuse_conn *fc); int fuse_uring_cmd(struct io_uring_cmd *cmd, unsigned int issue_flags); void fuse_uring_queue_fuse_req(struct fuse_iqueue *fiq, struct fuse_req *req); bool fuse_uring_queue_bq_req(struct fuse_req *req); @@ -152,10 +179,8 @@ static inline void fuse_uring_abort(struct fuse_conn *fc) if (ring == NULL) return; - if (atomic_read(&ring->queue_refs) > 0) { - fuse_uring_abort_end_requests(ring); - fuse_uring_stop_queues(ring); - } + fuse_uring_flush_bg(fc); + fuse_uring_stop_queues(ring); } static inline void fuse_uring_wait_stopped_queues(struct fuse_conn *fc) @@ -206,6 +231,14 @@ static inline bool fuse_uring_request_expired(struct fuse_conn *fc) return false; } +static inline bool fuse_uring_request_expired(struct fuse_conn *fc) +{ +} + +static inline void fuse_uring_flush_bg(struct fuse_conn *fc) +{ +} + #endif /* CONFIG_FUSE_IO_URING */ #endif /* _FS_FUSE_DEV_URING_I_H */ diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index 7ac6b232ef1232..05c081ca0ca59d 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -7,6 +7,7 @@ */ #include "fuse_i.h" +#include "fuse_dlm_cache.h" #include #include @@ -1492,14 +1493,7 @@ static int fuse_do_getattr(struct mnt_idmap *idmap, struct inode *inode, inarg.getattr_flags |= FUSE_GETATTR_FH; inarg.fh = ff->fh; } - args.opcode = FUSE_GETATTR; - args.nodeid = get_node_id(inode); - args.in_numargs = 1; - args.in_args[0].size = sizeof(inarg); - args.in_args[0].value = &inarg; - args.out_numargs = 1; - args.out_args[0].size = sizeof(outarg); - args.out_args[0].value = &outarg; + fuse_getattr_args_fill(&args, get_node_id(inode), &inarg, &outarg); err = fuse_simple_request(fm, &args); if (!err) { if (fuse_invalid_attr(&outarg.attr) || @@ -2116,6 +2110,12 @@ int fuse_flush_times(struct inode *inode, struct fuse_file *ff) inarg.valid |= FATTR_FH; inarg.fh = ff->fh; } + /* + * This is ->write_inode() flushing times the kernel owns locally, not + * a userspace utimes(); let the server tell the two apart. + */ + if (fm->fc->setattr_writeback) + inarg.valid |= FATTR_WRITEBACK; fuse_setattr_fill(fm->fc, &args, inode, &inarg, &outarg); return fuse_simple_request(fm, &args); @@ -2176,13 +2176,35 @@ int fuse_do_setattr(struct mnt_idmap *idmap, struct dentry *dentry, WARN_ON(!(attr->ia_valid & ATTR_SIZE)); WARN_ON(attr->ia_size != 0); if (fc->atomic_o_trunc) { + struct percpu_rw_semaphore *wb_sem = fi->wb_inval_rwsem; + /* * No need to send request to userspace, since actual * truncation has already been done by OPEN. But still * need to truncate page cache. + * + * Revoke and drop under the coherency gate write side, + * like the NOTIFY invalidate path: a gate reader that + * already re-validated its grant must not have the + * lock tree and the cache yanked mid-hold, or it + * would repopulate the truncated range trusting a + * grant that no longer exists. Waiting for gate + * readers here is safe: we hold i_rwsem exclusive, so + * no gate holder can be waiting on it (the write path + * takes i_rwsem before the gate, the read path never + * takes it). */ + if (wb_sem) + percpu_down_write(wb_sem); + if (fc->dlm && fc->writeback_cache) + fuse_dlm_cache_release_locks(fi); + spin_lock(&fi->lock); + fi->server_size = 0; i_size_write(inode, 0); + spin_unlock(&fi->lock); truncate_pagecache(inode, 0); + if (wb_sem) + percpu_up_write(wb_sem); goto out; } file = NULL; @@ -2273,6 +2295,13 @@ int fuse_do_setattr(struct mnt_idmap *idmap, struct dentry *dentry, /* see the comment in fuse_change_attributes() */ if (!is_wb || is_truncate) i_size_write(inode, outarg.attr.size); + /* + * A truncate settles the size on the server; only shrink the + * server-materialized bound: growing just exposes zeros, which the + * bound need not cover (see fuse_iomap_read_folio_range()). + */ + if (is_truncate && (loff_t) outarg.attr.size < fi->server_size) + fi->server_size = outarg.attr.size; if (is_truncate) { /* NOTE: this may release/reacquire fi->lock */ @@ -2286,8 +2315,23 @@ int fuse_do_setattr(struct mnt_idmap *idmap, struct dentry *dentry, */ if ((is_truncate || !is_wb) && S_ISREG(inode->i_mode) && oldsize != outarg.attr.size) { + struct percpu_rw_semaphore *wb_sem = fi->wb_inval_rwsem; + + /* + * Revoke and drop under the coherency gate write side; see + * the atomic-O_TRUNC branch above. i_rwsem is held + * exclusive here as well (setattr), so waiting out gate + * readers cannot deadlock. + */ + if (wb_sem) + percpu_down_write(wb_sem); + if (fc->dlm && fc->writeback_cache) + fuse_dlm_unlock_range(fi, outarg.attr.size & PAGE_MASK, -1); + truncate_pagecache(inode, outarg.attr.size); invalidate_inode_pages2(mapping); + if (wb_sem) + percpu_up_write(wb_sem); } clear_bit(FUSE_I_SIZE_UNSTABLE, &fi->state); diff --git a/fs/fuse/file.c b/fs/fuse/file.c index 676fd9856bfbf3..f019dccaeff6d2 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -7,6 +7,7 @@ */ #include "fuse_i.h" +#include "fuse_dlm_cache.h" #include #include @@ -23,6 +24,41 @@ #include #include +int sb_init_dio_done_wq(struct super_block *sb); + +/* + * Helper function to initialize fuse_args for OPEN/OPENDIR operations + */ +void fuse_open_args_fill(struct fuse_args *args, u64 nodeid, int opcode, + struct fuse_open_in *inarg, struct fuse_open_out *outarg) +{ + args->opcode = opcode; + args->nodeid = nodeid; + args->in_numargs = 1; + args->in_args[0].size = sizeof(*inarg); + args->in_args[0].value = inarg; + args->out_numargs = 1; + args->out_args[0].size = sizeof(*outarg); + args->out_args[0].value = outarg; +} + +/* + * Helper function to initialize fuse_args for GETATTR operations + */ +void fuse_getattr_args_fill(struct fuse_args *args, u64 nodeid, + struct fuse_getattr_in *inarg, + struct fuse_attr_out *outarg) +{ + args->opcode = FUSE_GETATTR; + args->nodeid = nodeid; + args->in_numargs = 1; + args->in_args[0].size = sizeof(*inarg); + args->in_args[0].value = inarg; + args->out_numargs = 1; + args->out_args[0].size = sizeof(*outarg); + args->out_args[0].value = outarg; +} + static int fuse_send_open(struct fuse_mount *fm, u64 nodeid, unsigned int open_flags, int opcode, struct fuse_open_out *outargp) @@ -40,14 +76,7 @@ static int fuse_send_open(struct fuse_mount *fm, u64 nodeid, inarg.open_flags |= FUSE_OPEN_KILL_SUIDGID; } - args.opcode = opcode; - args.nodeid = nodeid; - args.in_numargs = 1; - args.in_args[0].size = sizeof(inarg); - args.in_args[0].value = &inarg; - args.out_numargs = 1; - args.out_args[0].size = sizeof(*outargp); - args.out_args[0].value = outargp; + fuse_open_args_fill(&args, nodeid, opcode, &inarg, outargp); return fuse_simple_request(fm, &args); } @@ -126,8 +155,66 @@ static void fuse_file_put(struct fuse_file *ff, bool sync) } } +static int fuse_compound_open_getattr(struct fuse_mount *fm, u64 nodeid, + int flags, int opcode, + struct fuse_file *ff, + struct fuse_attr_out *outattrp, + struct fuse_open_out *outopenp) +{ + struct fuse_compound_req *compound; + struct fuse_args open_args = {}; + struct fuse_args getattr_args = {}; + struct fuse_open_in open_in = {}; + struct fuse_getattr_in getattr_in = {}; + int err; + + compound = fuse_compound_alloc(fm, 0); + if (IS_ERR(compound)) + return PTR_ERR(compound); + + open_in.flags = flags & ~(O_CREAT | O_EXCL | O_NOCTTY); + if (!fm->fc->atomic_o_trunc) + open_in.flags &= ~O_TRUNC; + + if (fm->fc->handle_killpriv_v2 && + (open_in.flags & O_TRUNC) && !capable(CAP_FSETID)) + open_in.open_flags |= FUSE_OPEN_KILL_SUIDGID; + + fuse_open_args_fill(&open_args, nodeid, opcode, &open_in, outopenp); + + err = fuse_compound_add(compound, &open_args); + if (err) + goto out; + + fuse_getattr_args_fill(&getattr_args, nodeid, &getattr_in, outattrp); + + err = fuse_compound_add(compound, &getattr_args); + if (err) + goto out; + + err = fuse_compound_send(compound); + if (err) + goto out; + + err = fuse_compound_get_error(compound, 0); + if (err) + goto out; + + err = fuse_compound_get_error(compound, 1); + if (err) + goto out; + + ff->fh = outopenp->fh; + ff->open_flags = outopenp->open_flags; + +out: + kfree(compound); + return err; +} + struct fuse_file *fuse_file_open(struct fuse_mount *fm, u64 nodeid, - unsigned int open_flags, bool isdir) + struct inode *inode, + unsigned int open_flags, bool isdir) { struct fuse_conn *fc = fm->fc; struct fuse_file *ff; @@ -153,23 +240,46 @@ struct fuse_file *fuse_file_open(struct fuse_mount *fm, u64 nodeid, if (open) { /* Store outarg for fuse_finish_open() */ struct fuse_open_out *outargp = &ff->args->open_outarg; - int err; + int err = -ENOSYS; + + if (inode && fc->compound_open_getattr) { + + struct fuse_attr_out attr_outarg; + + err = fuse_compound_open_getattr(fm, nodeid, open_flags, + opcode, ff, + &attr_outarg, outargp); + if (err == -ENOSYS) + fc->compound_open_getattr = 0; + if (!err) + fuse_change_attributes(inode, &attr_outarg.attr, + NULL, + ATTR_TIMEOUT(&attr_outarg), + fuse_get_attr_version(fc)); + } + if (err == -ENOSYS) { + err = fuse_send_open(fm, nodeid, open_flags, opcode, outargp); + if (!err) { + ff->fh = outargp->fh; + ff->open_flags = outargp->open_flags; + } + } - err = fuse_send_open(fm, nodeid, open_flags, opcode, outargp); - if (!err) { - ff->fh = outargp->fh; - ff->open_flags = outargp->open_flags; - } else if (err != -ENOSYS) { - fuse_file_free(ff); - return ERR_PTR(err); - } else { - if (isdir) { + if (err) { + if (err != -ENOSYS) { + /* err is not ENOSYS */ + fuse_file_free(ff); + return ERR_PTR(err); + } else { /* No release needed */ kfree(ff->args); ff->args = NULL; - fc->no_opendir = 1; - } else { - fc->no_open = 1; + + /* we don't have open */ + if (isdir) + fc->no_opendir = 1; + else + fc->no_open = 1; } } } @@ -185,11 +295,10 @@ struct fuse_file *fuse_file_open(struct fuse_mount *fm, u64 nodeid, int fuse_do_open(struct fuse_mount *fm, u64 nodeid, struct file *file, bool isdir) { - struct fuse_file *ff = fuse_file_open(fm, nodeid, file->f_flags, isdir); + struct fuse_file *ff = fuse_file_open(fm, nodeid, file_inode(file), file->f_flags, isdir); if (!IS_ERR(ff)) file->private_data = ff; - return PTR_ERR_OR_ZERO(ff); } EXPORT_SYMBOL_GPL(fuse_do_open); @@ -237,6 +346,7 @@ static void fuse_truncate_update_attr(struct inode *inode, struct file *file) spin_lock(&fi->lock); fi->attr_version = atomic64_inc_return(&fc->attr_version); + fi->server_size = 0; i_size_write(inode, 0); spin_unlock(&fi->lock); file_update_time(file); @@ -314,6 +424,18 @@ static void fuse_prepare_release(struct fuse_inode *fi, struct fuse_file *ff, if (likely(fi)) { spin_lock(&fi->lock); list_del(&ff->write_entry); + /* + * Leave forced direct IO mode once the last writer is gone: with + * no local writer left there is no cached-write contention with + * the remote modifier that triggered the switch. Restore + * FUSE_I_CACHE_IO_MODE for any frozen cached opens. + */ + if (test_bit(FUSE_I_FORCE_DIO, &fi->state) && + list_empty(&fi->write_files)) { + clear_bit(FUSE_I_FORCE_DIO, &fi->state); + if (fi->iocachectr > 0) + set_bit(FUSE_I_CACHE_IO_MODE, &fi->state); + } spin_unlock(&fi->lock); } spin_lock(&fc->lock); @@ -352,9 +474,24 @@ void fuse_file_release(struct inode *inode, struct fuse_file *ff, struct fuse_inode *fi = get_fuse_inode(inode); struct fuse_release_args *ra = &ff->args->release_args; int opcode = isdir ? FUSE_RELEASEDIR : FUSE_RELEASE; + bool was_force_dio = test_bit(FUSE_I_FORCE_DIO, &fi->state); fuse_prepare_release(fi, ff, open_flags, opcode, false); + /* + * If this release dropped the last writer, fuse_prepare_release() + * cleared the forced-direct-IO latch (under fi->lock). Drop any clean + * folios a read racing the latch may have repopulated so they cannot be + * served stale once caching mode resumes. No inode lock or + * wb_inval_rwsem: release may run on the fuse server thread (async fput + * from aio completion), where blocking on a contended inode lock could + * stall the connection. Writes were routed direct while latched, so + * only clean folios exist and this invalidate is server-free; the last + * writer is gone, so no forced-dio writer can race the drop. + */ + if (was_force_dio && !test_bit(FUSE_I_FORCE_DIO, &fi->state)) + invalidate_inode_pages2(inode->i_mapping); + if (ra && ff->flock) { ra->inarg.release_flags |= FUSE_RELEASE_FLOCK_UNLOCK; ra->inarg.lock_owner = fuse_lock_owner_id(ff->fm->fc, id); @@ -629,6 +766,19 @@ static ssize_t fuse_get_res_by_io(struct fuse_io_priv *io) return io->bytes < 0 ? io->size : io->bytes; } +static void fuse_aio_invalidate_worker(struct work_struct *work) +{ + struct fuse_io_priv *io = container_of(work, struct fuse_io_priv, work); + struct address_space *mapping = io->iocb->ki_filp->f_mapping; + ssize_t res = fuse_get_res_by_io(io); + pgoff_t start = io->offset >> PAGE_SHIFT; + pgoff_t end = (io->offset + res - 1) >> PAGE_SHIFT; + + invalidate_inode_pages2_range(mapping, start, end); + io->iocb->ki_complete(io->iocb, res); + kref_put(&io->refcnt, fuse_io_release); +} + /* * In case of short read, the caller sets 'pos' to the position of * actual end of fuse request in IO request. Otherwise, if bytes_requested @@ -661,10 +811,11 @@ static void fuse_aio_complete(struct fuse_io_priv *io, int err, ssize_t pos) spin_unlock(&io->lock); if (!left && !io->blocking) { + struct inode *inode = file_inode(io->iocb->ki_filp); + struct address_space *mapping = io->iocb->ki_filp->f_mapping; ssize_t res = fuse_get_res_by_io(io); if (res >= 0) { - struct inode *inode = file_inode(io->iocb->ki_filp); struct fuse_conn *fc = get_fuse_conn(inode); struct fuse_inode *fi = get_fuse_inode(inode); @@ -673,6 +824,17 @@ static void fuse_aio_complete(struct fuse_io_priv *io, int err, ssize_t pos) spin_unlock(&fi->lock); } + if (io->write && res > 0 && mapping->nrpages) { + /* + * As in generic_file_direct_write(), invalidate after the + * write, to invalidate read-ahead cache that may have competed + * with the write. + */ + INIT_WORK(&io->work, fuse_aio_invalidate_worker); + queue_work(inode->i_sb->s_dio_done_wq, &io->work); + return; + } + io->iocb->ki_complete(io->iocb, res); } @@ -835,8 +997,20 @@ static int fuse_do_readfolio(struct file *file, struct folio *folio, fuse_read_args_fill(&ia, file, pos, desc.length, FUSE_READ); res = fuse_simple_request(fm, &ia.ap.args); - if (res < 0) + if (res < 0) { + /* + * Please refer to Documentation/filesystems/fuse/fuse-AOP_TRUNCATED_PAGE-reason.txt + * why READ can return -EDEADLK from DLM subsystem. + * + * -EDEADLK: Preferred error code indicating DLM lock ordering violation + * (would cause deadlock with page lock) + * -EAGAIN: Legacy error code, maintained for backward compatibility + */ + if ((res == -EDEADLK || res == -EAGAIN) && fm->fc->dlm) + res = AOP_TRUNCATED_PAGE; return res; + } + /* * Short read means EOF. If file size is larger, truncate it */ @@ -989,9 +1163,78 @@ static int fuse_iomap_read_folio_range(const struct iomap_iter *iter, size_t len) { struct file *file = iter->private; + struct inode *inode = file_inode(file); + struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_inode *fi = get_fuse_inode(inode); size_t off = offset_in_folio(folio, pos); + bool hole; + int ret; + + /* + * Expanding writes claim their new i_size up front (see + * fuse_cache_write_iter()), which keeps iomap's own beyond-EOF + * zeroing in iomap_block_needs_zeroing() from ever firing for the + * write's own range: every block of a file expansion would be read + * from the server although it cannot contain data. Zero-fill + * locally instead when the server is known to hold no data in the + * range and we hold the DLM write lock covering it: + * + * - fi->server_size bounds the data materialized on the server + * (writeback and direct write acknowledgements, server + * attributes), + * - local data not yet acknowledged sits in uptodate blocks, which + * iomap never passes to this callback, + * - the page-granular DLM write lock excludes data written by + * other nodes, re-checked against the live lock tree so a + * revoked lock falls back to reading. + */ + if (fc->dlm) { + spin_lock(&fi->lock); + hole = pos >= fi->server_size; + spin_unlock(&fi->lock); + + if (hole && fuse_dlm_range_is_locked(fi, pos, pos + len - 1, + FUSE_PAGE_LOCK_WRITE)) { + folio_zero_range(folio, off, len); + return 0; + } + } + + ret = fuse_do_readfolio(file, folio, off, len); + + /* + * TEMPORARY WORKAROUND for iomap write deadlock: + * + * When FUSE server returns -EDEADLK (or legacy -EAGAIN) due to DLM + * lock contention, fuse_do_readfolio() converts it to AOP_TRUNCATED_PAGE + * and unlocks the folio (per AOP_TRUNCATED_PAGE contract). + * + * However, iomap doesn't understand AOP_TRUNCATED_PAGE. + * We need to: + * 1. Mark the retry flag (caller stored it in xarray) + * 2. Convert to -EAGAIN so iomap sees an error + * 3. Let fuse_cache_write_iter() detect and retry + * + * This breaks the ABBA deadlock: + * - Folio is unlocked (page invalidation can proceed) + * - Write will be retried at higher level + * + * Remove this when mainline iomap gains AOP_TRUNCATED_PAGE support. + */ + if (ret == AOP_TRUNCATED_PAGE) { + struct fuse_dlm_retry *retry; + unsigned long task_key = (unsigned long)current; + + retry = xa_load(&fc->dlm_retry_tasks, task_key); + if (retry) { + retry->retry_needed = true; + } + + /* Convert to -EAGAIN for iomap */ + ret = -EAGAIN; + } - return fuse_do_readfolio(file, folio, off, len); + return ret; } static void fuse_readpages_end(struct fuse_mount *fm, struct fuse_args *args, @@ -1084,10 +1327,23 @@ static void fuse_readahead(struct readahead_control *rac) iomap_readahead(&fuse_iomap_ops, &ctx, NULL); } +static ssize_t fuse_direct_read_iter(struct kiocb *iocb, struct iov_iter *to); + +/* + * Bound on re-requesting a revoked DLM grant before a cached read is + * served unlocked; see fuse_cache_read_iter(). + */ +#define FUSE_DLM_READ_RETRIES 3 + static ssize_t fuse_cache_read_iter(struct kiocb *iocb, struct iov_iter *to) { - struct inode *inode = iocb->ki_filp->f_mapping->host; + struct file *file = iocb->ki_filp; + struct inode *inode = file->f_mapping->host; struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_inode *fi = get_fuse_inode(inode); + struct percpu_rw_semaphore *wb_sem = fi->wb_inval_rwsem; + ssize_t res; + int lock_err = 0; /* * In auto invalidate mode, always update attributes on read. @@ -1102,7 +1358,69 @@ static ssize_t fuse_cache_read_iter(struct kiocb *iocb, struct iov_iter *to) return err; } - return generic_file_read_iter(iocb, to); + /* if we have dlm support acquire a read lock for the area + * we are reading from. */ + if (fc->writeback_cache && fc->dlm) + lock_err = fuse_get_dlm_lock(file, iocb->ki_pos, + iov_iter_count(to), + FUSE_PAGE_LOCK_READ); + + /* + * Fence the cache-serving read against a NOTIFY invalidate so we never + * hand back a folio the server has just superseded. The gate read side + * is per-CPU cheap; the NOTIFY holds the write side with priority. + * Re-check the forced-DIO latch under it: if a storm latched us while we + * waited on a pending writer, reroute to direct like the buffered write + * path, so we do not repopulate the cache the latch just dropped. + * wb_sem is NULL on non-writeback+dlm mounts (gate inactive). + */ + if (wb_sem) { + int tries = FUSE_DLM_READ_RETRIES; + +retry: + percpu_down_read(wb_sem); + if (fuse_inode_force_dio(inode)) { + percpu_up_read(wb_sem); + return fuse_direct_read_iter(iocb, to); + } + /* + * The DLM lock was requested before entering the gate, and + * the NOTIFY invalidate we may just have waited on revokes + * locks under the gate write side. Re-check the grant here + * and re-request with the gate dropped, so a + * FUSE_DLM_WB_LOCK round trip never parks a pending + * invalidate behind our own gate hold. Once the check + * passes the lock cannot go away for the rest of the gate + * hold. A failed or unrecorded request falls through + * unlocked, as before: the retry is taken even then (the + * latch must be re-checked under the re-entered gate), so + * lock_err has to stay sticky across it -- seeded by the + * pre-gate request above -- or a grant that failed would + * be re-requested forever. The retry is also bounded: a + * remote writer can revoke each successful grant before + * the gate is re-entered, and a reader-only inode has no + * force-DIO latch to end such a storm, so after + * FUSE_DLM_READ_RETRIES re-requests the read is served + * unlocked rather than looping without bound. + */ + if (!lock_err && fc->dlm && tries-- > 0 && + !fuse_dlm_lock_is_held(fi, iocb->ki_pos, + iov_iter_count(to), + FUSE_PAGE_LOCK_READ)) { + percpu_up_read(wb_sem); + lock_err = fuse_get_dlm_lock(file, iocb->ki_pos, + iov_iter_count(to), + FUSE_PAGE_LOCK_READ); + goto retry; + } + } + + res = generic_file_read_iter(iocb, to); + + if (wb_sem) + percpu_up_read(wb_sem); + + return res; } static void fuse_write_args_fill(struct fuse_io_args *ia, struct fuse_file *ff, @@ -1174,6 +1492,15 @@ bool fuse_write_update_attr(struct inode *inode, loff_t pos, ssize_t written) spin_lock(&fi->lock); fi->attr_version = atomic64_inc_return(&fc->attr_version); + if (written > 0 && S_ISREG(inode->i_mode)) { + /* + * The server acknowledged data up to @pos, keep the + * server-materialized bound in sync for the expansion + * zero-fill in fuse_iomap_read_folio_range(). + */ + if (pos > fi->server_size) + fi->server_size = pos; + } if (written > 0 && pos > inode->i_size) { i_size_write(inode, pos); ret = true; @@ -1384,13 +1711,6 @@ static ssize_t fuse_perform_write(struct kiocb *iocb, struct iov_iter *ii) return res; } -static bool fuse_io_past_eof(struct kiocb *iocb, struct iov_iter *iter) -{ - struct inode *inode = file_inode(iocb->ki_filp); - - return iocb->ki_pos + iov_iter_count(iter) > i_size_read(inode); -} - /* * @return true if an exclusive lock for direct IO writes is needed */ @@ -1400,9 +1720,15 @@ static bool fuse_dio_wr_exclusive_lock(struct kiocb *iocb, struct iov_iter *from struct fuse_file *ff = file->private_data; struct inode *inode = file_inode(iocb->ki_filp); struct fuse_inode *fi = get_fuse_inode(inode); + bool force_dio = test_bit(FUSE_I_FORCE_DIO, &fi->state); - /* Server side has to advise that it supports parallel dio writes. */ - if (!(ff->open_flags & FOPEN_PARALLEL_DIRECT_WRITES)) + /* + * Server side has to advise that it supports parallel dio writes. + * When the inode is latched into forced direct IO, parallel writes are + * used unconditionally: the page cache has been flushed and is bypassed + * for this inode. + */ + if (!force_dio && !(ff->open_flags & FOPEN_PARALLEL_DIRECT_WRITES)) return true; /* @@ -1413,18 +1739,14 @@ static bool fuse_dio_wr_exclusive_lock(struct kiocb *iocb, struct iov_iter *from return true; /* shared locks are not allowed with parallel page cache IO */ - if (test_bit(FUSE_I_CACHE_IO_MODE, &fi->state)) - return true; - - /* Parallel dio beyond EOF is not supported, at least for now. */ - if (fuse_io_past_eof(iocb, from)) + if (!force_dio && test_bit(FUSE_I_CACHE_IO_MODE, &fi->state)) return true; return false; } static void fuse_dio_lock(struct kiocb *iocb, struct iov_iter *from, - bool *exclusive) + bool *exclusive, bool *uncached) { struct inode *inode = file_inode(iocb->ki_filp); struct fuse_inode *fi = get_fuse_inode(inode); @@ -1438,19 +1760,20 @@ static void fuse_dio_lock(struct kiocb *iocb, struct iov_iter *from, * New parallal dio allowed only if inode is not in caching * mode and denies new opens in caching mode. This check * should be performed only after taking shared inode lock. - * Previous past eof check was without inode lock and might - * have raced, so check it again. */ - if (fuse_io_past_eof(iocb, from) || - fuse_inode_uncached_io_start(fi, NULL) != 0) { - inode_unlock_shared(inode); - inode_lock(inode); - *exclusive = true; + if (!test_bit(FUSE_I_FORCE_DIO, &fi->state)) { + if (fuse_inode_uncached_io_start(fi, NULL) != 0) { + inode_unlock_shared(inode); + inode_lock(inode); + *exclusive = true; + } else { + *uncached = true; + } } } } -static void fuse_dio_unlock(struct kiocb *iocb, bool exclusive) +static void fuse_dio_unlock(struct kiocb *iocb, bool exclusive, bool uncached) { struct inode *inode = file_inode(iocb->ki_filp); struct fuse_inode *fi = get_fuse_inode(inode); @@ -1458,8 +1781,8 @@ static void fuse_dio_unlock(struct kiocb *iocb, bool exclusive) if (exclusive) { inode_unlock(inode); } else { - /* Allow opens in caching mode after last parallel dio end */ - fuse_inode_uncached_io_end(fi); + if (uncached) + fuse_inode_uncached_io_end(fi); inode_unlock_shared(inode); } } @@ -1468,6 +1791,138 @@ static const struct iomap_write_ops fuse_iomap_write_ops = { .read_folio_range = fuse_iomap_read_folio_range, }; +static ssize_t fuse_writeback_write_iter(struct kiocb *iocb, + struct iov_iter *from, + struct file *file) +{ + struct fuse_conn *fc = get_fuse_conn(file_inode(file)); + ssize_t written, total_written = 0; + + /* + * TEMPORARY WORKAROUND for iomap write deadlock: + * + * Stack-allocate retry state and register it before calling + * iomap. If fuse_iomap_read_folio_range() encounters + * AOP_TRUNCATED_PAGE, it will mark retry_needed. + * + * Stack allocation ensures no memory leaks - the state is + * valid for the duration of this function call and is + * automatically cleaned up. + */ + struct fuse_dlm_retry retry_state = { + .retry_needed = false, + }; + unsigned long task_key = (unsigned long)current; + int xa_ret; + + xa_ret = xa_err(xa_store(&fc->dlm_retry_tasks, task_key, + &retry_state, GFP_KERNEL)); + if (xa_ret) + return xa_ret; + +retry: + /* + * Reset before each iomap call so retry_needed only reflects what + * happened in the most recent call. iomap may set retry_needed + * during an internal iteration that then recovers and completes + * the write fully; without the reset the flag would survive into + * the next iteration with iov_iter already drained, and iomap + * would re-enter with len==0 and livelock on a 0-length mapping. + */ + retry_state.retry_needed = false; + + /* + * Use iomap so that we can do granular uptodate reads + * and granular dirty tracking for large folios. + */ + written = iomap_file_buffered_write(iocb, from, &fuse_iomap_ops, + &fuse_iomap_write_ops, file); + + if (written > 0) + total_written += written; + + /* + * If DLM lock contention occurred (AOP_TRUNCATED_PAGE), + * retry the entire write operation. + * + * The folio has been unlocked by fuse_do_readfolio(), + * breaking the ABBA deadlock with page invalidation. + * + * Keep the entry in xarray and reuse it for the retry. + * + * Remove this when mainline iomap gains AOP_TRUNCATED_PAGE + * retry support. + */ + if (retry_state.retry_needed && iov_iter_count(from)) + goto retry; + + /* Remove from xarray now that we're done */ + xa_erase(&fc->dlm_retry_tasks, task_key); + + return written < 0 ? written : total_written; +} + +static ssize_t fuse_direct_write_iter(struct kiocb *iocb, struct iov_iter *from); + +/* + * @return true if an exclusive inode lock is needed for a cached (buffered) + * write. + * + * Buffered writes normally hold the inode rwsem exclusively, serialising all + * writers even on disjoint ranges. The DLM-serialised iomap writeback path is + * the exception: the DLM already excludes cluster-wide, and i_size is committed + * under fi->lock rather than the inode rwsem (see fuse_cache_write_iter()), so + * disjoint writers (MPI-IO / IOR) may share the lock. Mirrors + * fuse_dio_wr_exclusive_lock() for the direct path. + */ +static bool fuse_cache_wr_exclusive_lock(struct kiocb *iocb, bool writeback) +{ + struct inode *inode = file_inode(iocb->ki_filp); + struct fuse_conn *fc = get_fuse_conn(inode); + + /* Only the DLM-serialised iomap writeback path relaxes the lock. */ + if (!fc->dlm || !writeback) + return true; + + /* O_DIRECT writes fall back to generic_file_direct_write(). */ + if (iocb->ki_flags & IOCB_DIRECT) + return true; + + /* Append needs the eventual EOF - always needs an exclusive lock. */ + if (iocb->ki_flags & IOCB_APPEND) + return true; + + return false; +} + +static void fuse_cache_wr_unlock(struct inode *inode, bool exclusive) +{ + if (exclusive) + inode_unlock(inode); + else + inode_unlock_shared(inode); +} + +/* + * Request the DLM write lock covering a cached write. -ENOSYS cleared + * fc->dlm: the server has no DLM, proceed as a plain cached write. Any + * other failure means the cache would be dirtied without DLM coverage - + * the caller must fail the write instead. A granted-but-unrecorded + * lock (positive return) is covered cluster-wide; proceed, but flag it + * so the in-gate re-validation skips a check an invisible grant could + * never pass. + */ +static int fuse_cache_wr_dlm_lock(struct file *file, loff_t pos, size_t len, + bool *unrecorded) +{ + int err = fuse_get_dlm_lock(file, pos, len, FUSE_PAGE_LOCK_WRITE); + + if (err < 0 && err != -ENOSYS) + return err; + *unrecorded = err > 0; + return 0; +} + static ssize_t fuse_cache_write_iter(struct kiocb *iocb, struct iov_iter *from) { struct file *file = iocb->ki_filp; @@ -1477,32 +1932,168 @@ static ssize_t fuse_cache_write_iter(struct kiocb *iocb, struct iov_iter *from) struct inode *inode = mapping->host; ssize_t err, count; struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_inode *fi = get_fuse_inode(inode); + struct percpu_rw_semaphore *wb_sem = fi->wb_inval_rwsem; bool writeback = false; + bool wb_guard = false; + bool exclusive = true; + bool dlm_unrecorded = false; + loff_t dlm_pos = 0; + size_t dlm_len = 0; + + if (fuse_inode_force_dio(inode)) + return fuse_direct_write_iter(iocb, from); if (fc->writeback_cache) { - /* Update size (EOF optimization) and mode (SUID clearing) */ - err = fuse_update_attributes(mapping->host, file, - STATX_SIZE | STATX_MODE); + /* Update mode for SUID clearing, and also update size if the file + * is opened with O_APPEND mode. + */ + u32 request_mask = (file->f_flags & O_APPEND) ? + (STATX_SIZE | STATX_MODE) : STATX_MODE; + err = fuse_update_attributes(mapping->host, file, request_mask); if (err) return err; - if (!fc->handle_killpriv_v2 || + /* + * A write that drops suid/sgid stays off the writeback path, + * so it holds i_rwsem exclusive. + * + * With handle_killpriv_v2 that is because the server does the + * killing from the WRITE itself. Without it, fuse_setattr() + * has to ask the server, and fuse_do_setattr() freezes + * writepages around that SETATTR: fuse_set_nowrite() asserts + * BUG_ON(fi->writectr < 0), which assumes an exclusive + * i_rwsem, and the DLM-relaxed buffered write path below holds + * it only shared. Two writers can both see the bits set + * before either has cleared them, and the second one would + * then oops inside spin_lock(&fi->lock). + * + * Only the DLM path needs the detour: everywhere else the + * buffered write already holds i_rwsem exclusive, so the two + * writers cannot overlap in the first place. + * + * The bits are read without the inode lock here, so a server + * attribute update can still set them between this test and + * file_remove_privs(). That leaves the same race, but only + * for writers whose mode changed underneath them, rather than + * for every write to a suid file. + */ + if (!(fc->handle_killpriv_v2 || fc->dlm) || !setattr_should_drop_suidgid(idmap, file_inode(file))) writeback = true; } - inode_lock(inode); + exclusive = fuse_cache_wr_exclusive_lock(iocb, writeback); + + /* + * Request the DLM write lock before taking i_rwsem: the request is + * an unbounded cluster round trip, and holding the writer-priority + * rwsem across it would park a truncate -- and behind it every + * later writer -- for the duration. The grant-to-use window this + * leaves open is closed by the in-gate re-validation below. Only + * the append case must wait for the lock: its range depends on + * i_size, which is stable only under the exclusive inode lock. + */ + if (writeback && fc->dlm && !(iocb->ki_flags & IOCB_APPEND)) { + dlm_pos = iocb->ki_pos; + dlm_len = iov_iter_count(from); + + err = fuse_cache_wr_dlm_lock(file, dlm_pos, dlm_len, + &dlm_unrecorded); + if (err) + return err; + + /* + * The request above may have found that the server has no DLM + * at all, in which case it cleared fc->dlm. The relaxed shared + * lock was chosen just before, while fc->dlm still read 1, and + * it is only sound under DLM: the shared path claims the i_size + * extension up front, which stops iomap from zeroing beyond + * EOF, and the zero-fill that replaces it in + * fuse_iomap_read_folio_range() is itself gated on fc->dlm. + * Left as chosen, an expanding write would fall through to a + * READ of a range that cannot hold data -- which fails outright + * on a handle the client opened write-only. Re-decide now, + * while no lock is held yet. + */ + exclusive = fuse_cache_wr_exclusive_lock(iocb, writeback); + } + + if (exclusive) + inode_lock(inode); + else + inode_lock_shared(inode); + + /* note that this small code dup will save us a lot of headache later + * when appends are done concurrently without using parallel direct writes */ + if (writeback && fc->dlm && (iocb->ki_flags & IOCB_APPEND)) { + /* + * An append write lands at the current EOF no matter what + * ki_pos holds: generic_write_checks() rewrites ki_pos to + * i_size for IOCB_APPEND, and i_size is stable here because + * append writes hold the inode lock exclusive. Lock where + * the data will land. + */ + dlm_pos = i_size_read(inode); + dlm_len = iov_iter_count(from); + + err = fuse_cache_wr_dlm_lock(file, dlm_pos, dlm_len, + &dlm_unrecorded); + if (err) + goto out; + } err = count = generic_write_checks(iocb, from); if (err <= 0) goto out; - task_io_account_write(count); - + /* + * Kill suid/sgid and stamp the timestamps here, before the gate, + * instead of leaving them next to the write itself. kiocb_modified() + * -> file_remove_privs() is the one that reaches the server: without + * handle_killpriv[_v2] fuse_setattr() kills the bits by asking it (a + * FUSE_GETATTR to refresh the mode, then a FUSE_SETATTR, which for a + * writeback inode first flushes and freezes writepages), and + * security_inode_killpriv() can drop the capability xattr with another + * round trip. A server may have to invalidate this inode from inside + * such a handler; its NOTIFY_INVAL_INODE then blocks in + * percpu_down_write() draining a gate reader that is itself waiting for + * the reply. Nothing held under the gate may wait for the server. + * + * This also runs before the forced-DIO re-route below, so a re-routed + * write repeats it; there is nothing left to do the second time. + */ err = kiocb_modified(iocb); if (err) goto out; + wb_guard = !!wb_sem; + if (wb_guard) { +retry: + percpu_down_read(wb_sem); + if (fuse_inode_force_dio(inode)) { + percpu_up_read(wb_sem); + fuse_cache_wr_unlock(inode, exclusive); + return fuse_direct_write_iter(iocb, from); + } + if (writeback && fc->dlm && !dlm_unrecorded && + !fuse_dlm_lock_is_held(fi, dlm_pos, dlm_len, + FUSE_PAGE_LOCK_WRITE)) { + percpu_up_read(wb_sem); + err = fuse_cache_wr_dlm_lock(file, dlm_pos, dlm_len, + &dlm_unrecorded); + if (err) { + /* The gate is already dropped; funnel the + * failure through the one audited exit. */ + wb_guard = false; + goto out; + } + goto retry; + } + } + + task_io_account_write(count); + if (iocb->ki_flags & IOCB_DIRECT) { written = generic_file_direct_write(iocb, from); if (written < 0 || !iov_iter_count(from)) @@ -1510,19 +2101,77 @@ static ssize_t fuse_cache_write_iter(struct kiocb *iocb, struct iov_iter *from) written = direct_write_fallback(iocb, from, written, fuse_perform_write(iocb, from)); } else if (writeback) { + loff_t pos = iocb->ki_pos; + loff_t end = pos + count; + loff_t orig_size = 0; + bool extended = false; + + /* + * i_size is not protected by the shared lock in inode->i_rwsem. + * So if iomap_write_iter() grew EOF past i_size via its normal + * unlocked read-modify-write, two concurrent writers could race + * and one's update would get lost. + * To avoid this, claim the extension up front under fi->lock, + * so iomap sees pos + written <= i_size and never touches i_size + * itself. The update can then safely happen here, the same way + * fuse_write_update_attr() commits size on the direct io path. + * + * The lockless pre-check below avoids needlessly locking fi->lock + * if writes fall within the existing i_size. + * Operations that grow the file size take fi->lock, whereas a + * truncate holds the inode->i_rwsem exclusive. A stale read + * may over trigger this slow path, but it won’t miss an extension + * beyond i_size. + * + * The exclusive path keeps the classic behavior + * iomap owns the i_size update, serialized by the inode lock. + */ + if (!exclusive && end > i_size_read(inode)) { + spin_lock(&fi->lock); + orig_size = i_size_read(inode); + if (end > orig_size) { + i_size_write(inode, end); + extended = true; + } + spin_unlock(&fi->lock); + + /* Zero the tail of the folio straddling the old EOF. */ + if (extended && orig_size < pos) + pagecache_isize_extended(inode, orig_size, pos); + } + + written = fuse_writeback_write_iter(iocb, from, file); + /* - * Use iomap so that we can do granular uptodate reads - * and granular dirty tracking for large folios. + * Reconcile the speculative extension with what was actually + * written (short write, error, or nothing written all retract + * to the reached position). Only retract if no concurrent + * extender has pushed i_size past our claim; otherwise + * [reached, end) is a legitimate hole inside their extension and + * must remain. */ - written = iomap_file_buffered_write(iocb, from, - &fuse_iomap_ops, - &fuse_iomap_write_ops, - file); + if (extended) { + loff_t reached = written > 0 ? pos + written : orig_size; + + if (reached < end) { + spin_lock(&fi->lock); + if (i_size_read(inode) == end) + i_size_write(inode, reached); + spin_unlock(&fi->lock); + } + } + + if (written < 0) { + err = written; + goto out; + } } else { written = fuse_perform_write(iocb, from); } out: - inode_unlock(inode); + if (wb_guard) + percpu_up_read(wb_sem); + fuse_cache_wr_unlock(inode, exclusive); if (written > 0) written = generic_write_sync(iocb, written); @@ -1738,15 +2387,6 @@ ssize_t fuse_direct_io(struct fuse_io_priv *io, struct iov_iter *iter, if (res > 0) *ppos = pos; - if (res > 0 && write && fopen_direct_io) { - /* - * As in generic_file_direct_write(), invalidate after the - * write, to invalidate read-ahead cache that may have competed - * with the write. - */ - invalidate_inode_pages2_range(mapping, idx_from, idx_to); - } - return res > 0 ? res : err; } EXPORT_SYMBOL_GPL(fuse_direct_io); @@ -1765,14 +2405,16 @@ static ssize_t __fuse_direct_read(struct fuse_io_priv *io, return res; } -static ssize_t fuse_direct_IO(struct kiocb *iocb, struct iov_iter *iter); +static ssize_t __fuse_direct_IO(struct kiocb *iocb, struct iov_iter *iter, + bool exclusive); static ssize_t fuse_direct_read_iter(struct kiocb *iocb, struct iov_iter *to) { ssize_t res; if (!is_sync_kiocb(iocb)) { - res = fuse_direct_IO(iocb, to); + /* exclusive is unused on reads; rollback is write-only */ + res = __fuse_direct_IO(iocb, to, true); } else { struct fuse_io_priv io = FUSE_IO_PRIV_SYNC(iocb); @@ -1785,15 +2427,18 @@ static ssize_t fuse_direct_read_iter(struct kiocb *iocb, struct iov_iter *to) static ssize_t fuse_direct_write_iter(struct kiocb *iocb, struct iov_iter *from) { struct inode *inode = file_inode(iocb->ki_filp); + struct address_space *mapping = inode->i_mapping; + loff_t pos = iocb->ki_pos; + bool exclusive = false; + bool uncached = false; ssize_t res; - bool exclusive; - fuse_dio_lock(iocb, from, &exclusive); + fuse_dio_lock(iocb, from, &exclusive, &uncached); res = generic_write_checks(iocb, from); if (res > 0) { task_io_account_write(res); if (!is_sync_kiocb(iocb)) { - res = fuse_direct_IO(iocb, from); + res = __fuse_direct_IO(iocb, from, exclusive); } else { struct fuse_io_priv io = FUSE_IO_PRIV_SYNC(iocb); @@ -1801,8 +2446,18 @@ static ssize_t fuse_direct_write_iter(struct kiocb *iocb, struct iov_iter *from) FUSE_DIO_WRITE); fuse_write_update_attr(inode, iocb->ki_pos, res); } + if (res > 0 && mapping->nrpages) { + /* + * As in generic_file_direct_write(), invalidate after + * write, to invalidate read-ahead cache that may have + * with the write. + */ + invalidate_inode_pages2_range(mapping, + pos >> PAGE_SHIFT, + (pos + res - 1) >> PAGE_SHIFT); + } } - fuse_dio_unlock(iocb, exclusive); + fuse_dio_unlock(iocb, exclusive, uncached); return res; } @@ -1820,7 +2475,7 @@ static ssize_t fuse_file_read_iter(struct kiocb *iocb, struct iov_iter *to) return fuse_dax_read_iter(iocb, to); /* FOPEN_DIRECT_IO overrides FOPEN_PASSTHROUGH */ - if (ff->open_flags & FOPEN_DIRECT_IO) + if ((ff->open_flags & FOPEN_DIRECT_IO) || fuse_inode_force_dio(inode)) return fuse_direct_read_iter(iocb, to); else if (fuse_file_passthrough(ff)) return fuse_passthrough_read_iter(iocb, to); @@ -1841,7 +2496,7 @@ static ssize_t fuse_file_write_iter(struct kiocb *iocb, struct iov_iter *from) return fuse_dax_write_iter(iocb, from); /* FOPEN_DIRECT_IO overrides FOPEN_PASSTHROUGH */ - if (ff->open_flags & FOPEN_DIRECT_IO) + if ((ff->open_flags & FOPEN_DIRECT_IO) || fuse_inode_force_dio(inode)) return fuse_direct_write_iter(iocb, from); else if (fuse_file_passthrough(ff)) return fuse_passthrough_write_iter(iocb, from); @@ -1999,6 +2654,20 @@ static void fuse_writepage_end(struct fuse_mount *fm, struct fuse_args *args, if (!fc->writeback_cache) fuse_invalidate_attr_mask(inode, FUSE_STATX_MODIFY); spin_lock(&fi->lock); + if (!error) { + struct fuse_write_in *inarg = &wpa->ia.write.in; + + /* + * The server acknowledged this writeback, so data up to the + * end of the request is materialized on the server. Advance + * the bound before the folios end writeback below, i.e. + * before they can go clean and be reclaimed, so that + * fuse_iomap_read_folio_range() can never zero-fill a + * reclaimed range the server holds data in. + */ + if ((loff_t) (inarg->offset + inarg->size) > fi->server_size) + fi->server_size = inarg->offset + inarg->size; + } fi->writectr--; fuse_writepage_finish(wpa); spin_unlock(&fi->lock); @@ -2327,6 +2996,60 @@ static void fuse_vma_close(struct vm_area_struct *vma) mapping_set_error(vma->vm_file->f_mapping, err); } +/** + * Request a DLM lock from the FUSE server. + * + * This routine is similar to fuse_get_dlm_lock(), but it + * does not cache the DLM lock in the kernel. + */ +static int fuse_get_page_mkwrite_lock(struct file *file, loff_t offset, size_t length) +{ + struct fuse_file *ff = file->private_data; + struct inode *inode = file_inode(file); + struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = ff->fm; + + FUSE_ARGS(args); + struct fuse_dlm_lock_in inarg; + struct fuse_dlm_lock_out outarg; + int err; + + if (WARN_ON_ONCE((offset & ~PAGE_MASK) || (length & ~PAGE_MASK))) + return -EIO; + + memset(&inarg, 0, sizeof(inarg)); + inarg.fh = ff->fh; + + inarg.start = offset; + inarg.end = offset + length - 1; + inarg.type = FUSE_DLM_PAGE_MKWRITE; + + args.opcode = FUSE_DLM_WB_LOCK; + args.nodeid = get_node_id(inode); + args.in_numargs = 1; + args.in_args[0].size = sizeof(inarg); + args.in_args[0].value = &inarg; + args.out_numargs = 1; + args.out_args[0].size = sizeof(outarg); + args.out_args[0].value = &outarg; + err = fuse_simple_request(fm, &args); + if (err == -ENOSYS) { + fc->dlm = 0; + err = 0; + } + + if (!err && + fc->dlm && + (outarg.start > inarg.start || + outarg.end < inarg.end)) { + /* fuse server is seriously broken */ + pr_warn("fuse: dlm lock request for %llu:%llu bytes returned %llu:%llu bytes\n", + inarg.start, inarg.end, outarg.start, outarg.end); + fuse_abort_conn(fc); + err = -EINVAL; + } + return err; +} /* * Wait for writeback against this page to complete before allowing it * to be marked dirty again, and hence written back again, possibly @@ -2345,7 +3068,18 @@ static void fuse_vma_close(struct vm_area_struct *vma) static vm_fault_t fuse_page_mkwrite(struct vm_fault *vmf) { struct folio *folio = page_folio(vmf->page); - struct inode *inode = file_inode(vmf->vma->vm_file); + struct file *file = vmf->vma->vm_file; + struct inode *inode = file_inode(file); + struct fuse_mount *fm = get_fuse_mount(inode); + + if (fm->fc->dlm) { + loff_t pos = vmf->pgoff << PAGE_SHIFT; + size_t length = PAGE_SIZE; + int err = fuse_get_page_mkwrite_lock(file, pos, length); + if (err < 0) { + return vmf_error(err); + } + } file_update_time(vmf->vma->vm_file); folio_lock(folio); @@ -2386,6 +3120,29 @@ static int fuse_file_mmap(struct file *file, struct vm_area_struct *vma) else if (fuse_inode_backing(get_fuse_inode(inode))) return -ENODEV; + /* + * If the inode was latched into forced direct IO after a remote-modify + * notification, a mapping needs the page cache, so revert to caching + * mode. Revert without the inode lock or wb_inval_rwsem: ->mmap runs + * under mmap_lock and the buffered write path holds both across a fault + * on the user buffer (which takes mmap_lock), so taking either here + * would invert lock order (ABBA). Clearing the latch and dropping the + * cache is sufficient -- writers re-check the latch and route to cached + * IO once it is clear, and in-flight parallel dio drains itself. Cached + * opens frozen while latched are still counted in iocachectr, so restore + * FUSE_I_CACHE_IO_MODE for them. + */ + if (fuse_inode_force_dio(inode)) { + struct fuse_inode *fi = get_fuse_inode(inode); + + spin_lock(&fi->lock); + clear_bit(FUSE_I_FORCE_DIO, &fi->state); + if (fi->iocachectr > 0) + set_bit(FUSE_I_CACHE_IO_MODE, &fi->state); + spin_unlock(&fi->lock); + invalidate_inode_pages2(file->f_mapping); + } + /* * FOPEN_DIRECT_IO handling is special compared to O_DIRECT, * as does not allow MAP_SHARED mmap without FUSE_DIRECT_IO_ALLOW_MMAP. @@ -2814,7 +3571,7 @@ static inline loff_t fuse_round_up(struct fuse_conn *fc, loff_t off) } static ssize_t -fuse_direct_IO(struct kiocb *iocb, struct iov_iter *iter) +__fuse_direct_IO(struct kiocb *iocb, struct iov_iter *iter, bool exclusive) { DECLARE_COMPLETION_ONSTACK(wait); ssize_t ret = 0; @@ -2826,6 +3583,7 @@ fuse_direct_IO(struct kiocb *iocb, struct iov_iter *iter) size_t count = iov_iter_count(iter), shortened = 0; loff_t offset = iocb->ki_pos; struct fuse_io_priv *io; + bool async = ff->fm->fc->async_dio; pos = offset; inode = file->f_mapping->host; @@ -2834,6 +3592,12 @@ fuse_direct_IO(struct kiocb *iocb, struct iov_iter *iter) if ((iov_iter_rw(iter) == READ) && (offset >= i_size)) return 0; + if ((iov_iter_rw(iter) == WRITE) && async && !inode->i_sb->s_dio_done_wq) { + ret = sb_init_dio_done_wq(inode->i_sb); + if (ret < 0) + return ret; + } + io = kmalloc_obj(struct fuse_io_priv); if (!io) return -ENOMEM; @@ -2849,7 +3613,7 @@ fuse_direct_IO(struct kiocb *iocb, struct iov_iter *iter) * By default, we want to optimize all I/Os with async request * submission to the client filesystem if supported. */ - io->async = ff->fm->fc->async_dio; + io->async = async; io->iocb = iocb; io->blocking = is_sync_kiocb(iocb); @@ -2901,14 +3665,27 @@ fuse_direct_IO(struct kiocb *iocb, struct iov_iter *iter) if (iov_iter_rw(iter) == WRITE) { fuse_write_update_attr(inode, pos, ret); - /* For extending writes we already hold exclusive lock */ - if (ret < 0 && offset + count > i_size) + /* + * Whole-file rollback is only safe under an exclusive lock. + * Parallel writers commit i_size only on success (nothing to + * undo); the server owns failed-extend cleanup. + */ + if (exclusive && ret < 0 && offset + count > i_size) fuse_do_truncate(file); } return ret; } +static ssize_t fuse_direct_IO(struct kiocb *iocb, struct iov_iter *iter) +{ + /* + * Only reached via generic_file_direct_write/read() + * (caching-mode O_DIRECT), which holds the inode lock exclusively. + */ + return __fuse_direct_IO(iocb, iter, true); +} + static int fuse_writeback_range(struct inode *inode, loff_t start, loff_t end) { int err = filemap_write_and_wait_range(inode->i_mapping, start, LLONG_MAX); @@ -3206,11 +3983,49 @@ void fuse_init_file_inode(struct inode *inode, unsigned int flags) INIT_LIST_HEAD(&fi->write_files); INIT_LIST_HEAD(&fi->queued_writes); + fuse_dlm_cache_init(fi); fi->writectr = 0; fi->iocachectr = 0; + fi->server_size = 0; init_waitqueue_head(&fi->page_waitq); init_waitqueue_head(&fi->direct_io_waitq); + /* + * Coherency gate for the forced-direct-IO feature; only writeback+dlm + * regular files need it. A percpu_rw_semaphore embeds per-CPU state, + * so allocate it out of line and only when the mount can use it rather + * than paying it on every inode. On failure leave it NULL: the gate + * stays inactive (best-effort invalidate) and the inode is still usable. + */ + fi->wb_inval_rwsem = NULL; + if (fc->writeback_cache && fc->dlm) { + struct percpu_rw_semaphore *sem = kmalloc(sizeof(*sem), GFP_KERNEL); + + if (sem && percpu_init_rwsem(sem)) { + kfree(sem); + sem = NULL; + } + fi->wb_inval_rwsem = sem; + } + fi->notify_stamp = jiffies; + fi->notify_interval_ewma = FUSE_NOTIFY_EWMA_SEED << FUSE_NOTIFY_EWMA_SHIFT; if (IS_ENABLED(CONFIG_FUSE_DAX)) fuse_dax_inode_init(inode, flags); + + if (enable_large_folios) { + /* + * Readahead and writeback batch whole folios into a single + * request, capped at min(fc->max_pages, fc->max_read/PAGE_SIZE) + * pages. The page cache must therefore never build a folio + * larger than that, or fuse_readahead() trips WARN_ON(!pages) + * and then dereferences a NULL ap->folios[0] in + * fuse_send_readpages(). Bound the folio order to the request + * limit instead of MAX_PAGECACHE_ORDER. + */ + unsigned int max_pages = min(fc->max_pages, + fc->max_read >> PAGE_SHIFT); + + mapping_set_folio_order_range(inode->i_mapping, 0, + ilog2(max_pages ?: 1)); + } } diff --git a/fs/fuse/fuse_dev_i.h b/fs/fuse/fuse_dev_i.h index 134bf44aff0d39..4037fd7bdeee66 100644 --- a/fs/fuse/fuse_dev_i.h +++ b/fs/fuse/fuse_dev_i.h @@ -36,6 +36,8 @@ struct fuse_copy_state { bool is_uring:1; struct { unsigned int copied_sz; /* copied size into the user buffer */ + struct page **pages; + int page_idx; } ring; }; diff --git a/fs/fuse/fuse_dlm_cache.c b/fs/fuse/fuse_dlm_cache.c new file mode 100644 index 00000000000000..6531186d63b54b --- /dev/null +++ b/fs/fuse/fuse_dlm_cache.c @@ -0,0 +1,748 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * FUSE page lock cache implementation + */ +#include "fuse_i.h" +#include "fuse_dlm_cache.h" + +#include +#include +#include +#include + + +/* A range of pages with a lock */ +struct fuse_dlm_range { + /* Interval tree node */ + struct rb_node rb; + /* Start page offset (inclusive) */ + uint64_t start; + /* End page offset (inclusive) */ + uint64_t end; + /* Subtree end value for interval tree */ + uint64_t __subtree_end; + /* Lock mode */ + enum fuse_page_lock_mode mode; + /* Temporary list entry for operations */ + struct list_head list; +}; + +/* Lock modes for FUSE page cache */ +#define FUSE_PCACHE_LK_READ 1 /* Shared read lock */ +#define FUSE_PCACHE_LK_WRITE 2 /* Exclusive write lock */ + +/* Interval tree definitions for page ranges */ +static inline uint64_t fuse_dlm_range_start(struct fuse_dlm_range *range) +{ + return range->start; +} + +static inline uint64_t fuse_dlm_range_last(struct fuse_dlm_range *range) +{ + return range->end; +} + +INTERVAL_TREE_DEFINE(struct fuse_dlm_range, rb, uint64_t, __subtree_end, + fuse_dlm_range_start, fuse_dlm_range_last, static, + fuse_page_it); + +/** + * fuse_page_cache_init - Initialize a page cache lock manager + * @cache: The cache to initialize + * + * Initialize a page cache lock manager for a FUSE inode. + * + * Return: 0 on success, negative error code on failure + */ +int fuse_dlm_cache_init(struct fuse_inode *inode) +{ + struct fuse_dlm_cache *cache = &inode->dlm_locked_areas; + + if (!cache) + return -EINVAL; + + init_rwsem(&cache->lock); + cache->ranges = RB_ROOT_CACHED; + cache->revoke_gen = 0; + + return 0; +} + +/** + * fuse_page_cache_destroy - Clean up a page cache lock manager + * @cache: The cache to clean up + * + * Release all locks and free all resources associated with the cache. + */ +void fuse_dlm_cache_release_locks(struct fuse_inode *inode) +{ + struct fuse_dlm_cache *cache = &inode->dlm_locked_areas; + struct fuse_dlm_range *range; + struct rb_node *node; + + if (!cache) + return; + + /* Release all locks */ + down_write(&cache->lock); + WRITE_ONCE(cache->revoke_gen, cache->revoke_gen + 1); + while ((node = rb_first_cached(&cache->ranges)) != NULL) { + range = rb_entry(node, struct fuse_dlm_range, rb); + fuse_page_it_remove(range, &cache->ranges); + kfree(range); + } + up_write(&cache->lock); +} + +/** + * fuse_dlm_find_overlapping - Find a range that overlaps with [start, end] + * @cache: The page cache + * @start: Start page offset + * @end: End page offset + * + * Return: Pointer to the first overlapping range, or NULL if none found + */ +static struct fuse_dlm_range * +fuse_dlm_find_overlapping(struct fuse_dlm_cache *cache, uint64_t start, + uint64_t end) +{ + return fuse_page_it_iter_first(&cache->ranges, start, end); +} + +/** + * fuse_page_try_merge - Try to merge ranges within a specific region + * @cache: The page cache + * @start: Start page offset + * @end: End page offset + * + * Attempt to merge ranges within and adjacent to the specified region + * that have the same lock mode. + */ +static void fuse_dlm_try_merge(struct fuse_dlm_cache *cache, uint64_t start, + uint64_t end) +{ + struct fuse_dlm_range *range, *next; + uint64_t first = start ? start - 1 : start; + uint64_t last = end < U64_MAX ? end + 1 : end; + + if (!cache) + return; + + /* + * Find the first range that might need merging. Directly adjacent + * ranges can merge, hence the region is widened by one unit to each + * side (saturating at the type bounds). This must stay an + * interval-tree lookup: the tree holds every cached grant of the + * inode and strided writers grow it for the lifetime of the file, + * so seeding the merge by walking from the tree minimum would make + * every new grant cost a full scan. + */ + range = fuse_page_it_iter_first(&cache->ranges, first, last); + + /* Try to merge ranges in and around the specified region */ + while (range && range->start <= last) { + /* Get next range before we potentially modify the tree */ + next = NULL; + if (rb_next(&range->rb)) { + next = rb_entry(rb_next(&range->rb), + struct fuse_dlm_range, rb); + } + + /* Try to merge with next range if adjacent and same mode */ + if (next && range->mode == next->mode && + range->end + 1 == next->start) { + /* Merge ranges: re-insert so __subtree_end is updated */ + fuse_page_it_remove(next, &cache->ranges); + fuse_page_it_remove(range, &cache->ranges); + range->end = next->end; + fuse_page_it_insert(range, &cache->ranges); + kfree(next); + + /* Continue with the same range */ + continue; + } + + /* Move to next range */ + range = next; + } +} + +/** + * __fuse_dlm_lock_range - Lock a range of pages + * @cache: The page cache + * @start: Start page offset + * @end: End page offset + * @mode: Lock mode (read or write) + * @genp: If non-NULL, the revocation generation sampled before the grant + * was requested; recording fails with -EAGAIN if it has moved + * + * Add a locked range on the specified range of pages. + * If parts of the range are already locked, only add the remaining parts. + * For overlapping ranges, handle lock compatibility: + * - READ locks are compatible with existing READ locks + * - READ locks are compatible with existing WRITE locks (downgrade not needed) + * - WRITE locks need to upgrade existing READ locks + * + * Return: 0 on success, negative error code on failure + */ +static int __fuse_dlm_lock_range(struct fuse_inode *inode, uint64_t start, + uint64_t end, enum fuse_page_lock_mode mode, + const uint64_t *genp) +{ + struct fuse_dlm_cache *cache = &inode->dlm_locked_areas; + struct fuse_dlm_range *range, *new_range, *next; + int lock_mode; + bool covered_to_end = false; + int ret = 0; + LIST_HEAD(to_lock); + LIST_HEAD(to_upgrade); + uint64_t current_start = start; + + if (!cache || start > end) + return -EINVAL; + + /* Convert to lock mode */ + lock_mode = (mode == FUSE_PAGE_LOCK_READ) ? FUSE_PCACHE_LK_READ : + FUSE_PCACHE_LK_WRITE; + + down_write(&cache->lock); + + /* + * A revoke was processed after @genp was sampled; the grant this + * record carries may be the very one it targeted (a revoke of a + * not-yet-recorded grant removes nothing and would never be + * retried). Refuse, the caller re-requests. + */ + if (genp && cache->revoke_gen != *genp) { + up_write(&cache->lock); + return -EAGAIN; + } + + /* Find all ranges that overlap with [start, end] */ + range = fuse_page_it_iter_first(&cache->ranges, start, end); + while (range) { + /* Get next overlapping range before we potentially modify the tree */ + next = fuse_page_it_iter_next(range, start, end); + + /* Check lock compatibility */ + if (lock_mode == FUSE_PCACHE_LK_WRITE && + lock_mode != range->mode) { + /* we own the lock but have to update it. */ + list_add_tail(&range->list, &to_upgrade); + } + /* If WRITE lock already exists - nothing to do */ + + /* If there's a gap before this range, we need to add the missing range */ + if (current_start < range->start) { + new_range = kmalloc(sizeof(*new_range), GFP_KERNEL); + if (!new_range) { + ret = -ENOMEM; + goto out_free; + } + + new_range->start = current_start; + new_range->end = range->start - 1; + new_range->mode = lock_mode; + INIT_LIST_HEAD(&new_range->list); + + list_add_tail(&new_range->list, &to_lock); + } + + /* Move current_start past this range */ + if (range->end >= end) + covered_to_end = true; + else + current_start = max(current_start, range->end + 1); + + /* Move to next range */ + range = next; + } + + /* If there's a gap after the last range to the end, extend the range */ + if (!covered_to_end && current_start <= end) { + new_range = kmalloc(sizeof(*new_range), GFP_KERNEL); + if (!new_range) { + ret = -ENOMEM; + goto out_free; + } + + new_range->start = current_start; + new_range->end = end; + new_range->mode = lock_mode; + INIT_LIST_HEAD(&new_range->list); + + list_add_tail(&new_range->list, &to_lock); + } + + /* update locks, if any lock is in this list it has the wrong mode */ + list_for_each_entry(range, &to_upgrade, list) { + /* Update the lock mode */ + range->mode = lock_mode; + } + + /* Add all new ranges to the tree */ + list_for_each_entry(new_range, &to_lock, list) { + /* Add to interval tree */ + fuse_page_it_insert(new_range, &cache->ranges); + } + + /* Try to merge adjacent ranges with the same mode */ + fuse_dlm_try_merge(cache, start, end); + + up_write(&cache->lock); + return 0; + +out_free: + /* Free any ranges we allocated but didn't insert */ + while (!list_empty(&to_lock)) { + new_range = + list_first_entry(&to_lock, struct fuse_dlm_range, list); + list_del(&new_range->list); + kfree(new_range); + } + + /* Restore original lock modes for any partially upgraded locks */ + list_for_each_entry(range, &to_upgrade, list) { + if (lock_mode == FUSE_PCACHE_LK_WRITE) { + /* We upgraded this lock but failed later, downgrade it back */ + range->mode = FUSE_PCACHE_LK_READ; + } + } + + up_write(&cache->lock); + return ret; +} + +int fuse_dlm_lock_range(struct fuse_inode *inode, uint64_t start, + uint64_t end, enum fuse_page_lock_mode mode) +{ + return __fuse_dlm_lock_range(inode, start, end, mode, NULL); +} + +int fuse_dlm_lock_range_gen(struct fuse_inode *inode, uint64_t start, + uint64_t end, enum fuse_page_lock_mode mode, + uint64_t gen) +{ + return __fuse_dlm_lock_range(inode, start, end, mode, &gen); +} + +/** + * fuse_dlm_revoke_gen - sample the revocation generation + * @inode: the fuse inode + * + * Sampled before a FUSE_DLM_WB_LOCK request leaves the client. The + * reply and a NOTIFY revoke can be serviced on different threads, so a + * revoke may be processed between the reply arriving and its grant + * being recorded. fuse_dlm_lock_range_gen() re-checks the generation + * under the cache lock and refuses to record a grant such a revoke may + * have already killed. + */ +uint64_t fuse_dlm_revoke_gen(struct fuse_inode *inode) +{ + return READ_ONCE(inode->dlm_locked_areas.revoke_gen); +} + +/** + * fuse_dlm_punch_hole - Punch a hole in a locked range + * @cache: The page cache + * @start: Start page offset of the hole + * @end: End page offset of the hole + * + * Create a hole in a locked range by splitting it into two ranges. + * + * Return: 0 on success, negative error code on failure + */ +static int fuse_dlm_punch_hole(struct fuse_dlm_cache *cache, uint64_t start, + uint64_t end) +{ + struct fuse_dlm_range *range, *new_range; + int ret = 0; + + if (!cache || start > end) + return -EINVAL; + + /* Find a range that contains [start, end] */ + range = fuse_dlm_find_overlapping(cache, start, end); + if (!range) { + ret = -EINVAL; + goto out; + } + + /* If the hole is at the beginning of the range */ + if (start == range->start) { + fuse_page_it_remove(range, &cache->ranges); + range->start = end + 1; + fuse_page_it_insert(range, &cache->ranges); + goto out; + } + + /* If the hole is at the end of the range */ + if (end == range->end) { + fuse_page_it_remove(range, &cache->ranges); + range->end = start - 1; + fuse_page_it_insert(range, &cache->ranges); + goto out; + } + + /* The hole is in the middle, need to split */ + new_range = kmalloc(sizeof(*new_range), GFP_KERNEL); + if (!new_range) { + ret = -ENOMEM; + goto out; + } + + /* Copy properties from original range, keeping the original end */ + *new_range = *range; + INIT_LIST_HEAD(&new_range->list); + new_range->start = end + 1; + + /* + * Shorten the original end only while it is unlinked. range->end is + * used in calulating the interval tree's __subtree_end, so changes + * made while the node is still in the tree leaves every ancestor + * stale. Update range->end after fuse_page_it_remove() + */ + fuse_page_it_remove(range, &cache->ranges); + range->end = start - 1; + fuse_page_it_insert(range, &cache->ranges); + fuse_page_it_insert(new_range, &cache->ranges); + +out: + return ret; +} + +/** + * fuse_dlm_unlock_range - Unlock a range of pages + * @cache: The page cache + * @start: Start page offset + * @end: End page offset + * + * Release locks on the specified range of pages. An inverted range is + * rejected rather than silently removing nothing: the callers revoke + * coverage, and a revoke that quietly keeps the grant alive would let + * the re-validating IO paths trust a lock the server has taken away. + * To drop every grant use fuse_dlm_cache_release_locks() (there is no + * in-band sentinel range for it). + * + * Return: 0 on success, negative error code on failure + */ +int fuse_dlm_unlock_range(struct fuse_inode *inode, + uint64_t start, uint64_t end) +{ + struct fuse_dlm_cache *cache = &inode->dlm_locked_areas; + struct fuse_dlm_range *range, *next; + int ret = 0; + + if (!cache || start > end) + return -EINVAL; + + down_write(&cache->lock); + + /* + * Unconditional, even when nothing overlaps: the revoke racing + * with an in-flight grant finds an empty tree precisely because + * the grant is not recorded yet, and the bump is what makes the + * recording side notice (see fuse_dlm_lock_range_gen()). + */ + WRITE_ONCE(cache->revoke_gen, cache->revoke_gen + 1); + + /* Find all ranges that overlap with [start, end] */ + range = fuse_page_it_iter_first(&cache->ranges, start, end); + while (range) { + /* Get next overlapping range before we potentially modify the tree */ + next = fuse_page_it_iter_next(range, start, end); + + /* Check if we need to punch a hole */ + if (start > range->start && end < range->end) { + /* Punch a hole in the middle */ + ret = fuse_dlm_punch_hole(cache, start, end); + if (ret) + goto out; + /* After punching a hole, we're done */ + break; + } else if (start > range->start) { + /* Adjust the end of the range */ + fuse_page_it_remove(range, &cache->ranges); + range->end = start - 1; + fuse_page_it_insert(range, &cache->ranges); + } else if (end < range->end) { + /* Adjust the start of the range */ + fuse_page_it_remove(range, &cache->ranges); + range->start = end + 1; + fuse_page_it_insert(range, &cache->ranges); + } else { + /* Complete overlap, remove the range */ + fuse_page_it_remove(range, &cache->ranges); + kfree(range); + } + + range = next; + } + +out: + up_write(&cache->lock); + return ret; +} + +/** + * fuse_dlm_range_is_locked - Check if a page range is already locked + * @cache: The page cache + * @start: Start page offset + * @end: End page offset + * @mode: Lock mode to check for (or NULL to check for any lock) + * + * Check if the specified range of pages is already locked. + * The entire range must be locked for this to return true. + * + * Return: true if the entire range is locked, false otherwise + */ +bool fuse_dlm_range_is_locked(struct fuse_inode *inode, uint64_t start, + uint64_t end, enum fuse_page_lock_mode mode) +{ + struct fuse_dlm_cache *cache = &inode->dlm_locked_areas; + struct fuse_dlm_range *range; + int lock_mode = 0; + uint64_t current_start = start; + + if (!cache || start > end) + return false; + + /* Convert to lock mode if specified */ + if (mode == FUSE_PAGE_LOCK_READ) + lock_mode = FUSE_PCACHE_LK_READ; + else if (mode == FUSE_PAGE_LOCK_WRITE) + lock_mode = FUSE_PCACHE_LK_WRITE; + + down_read(&cache->lock); + + /* Find the first range that overlaps with [start, end] */ + range = fuse_dlm_find_overlapping(cache, start, end); + + /* Check if the entire range is covered */ + while (range && current_start <= end) { + /* + * The held lock must be at least as strong as the one + * requested. A WRITE lock (exclusive) satisfies a READ + * request, so only treat the range as uncovered when the + * held mode is weaker than what we ask for. This avoids + * re-requesting a READ lock for a range we already hold + * a WRITE lock on (e.g. read-after-write). + */ + if (lock_mode && range->mode < lock_mode) { + /* Held lock is weaker than requested */ + up_read(&cache->lock); + return false; + } + + /* Check if there's a gap before this range */ + if (current_start < range->start) { + /* Found a gap */ + up_read(&cache->lock); + return false; + } + + /* Covered through the end of the requested range? */ + if (range->end >= end) { + up_read(&cache->lock); + return true; + } + + /* Move current_start past this range */ + current_start = range->end + 1; + + /* Get next overlapping range */ + range = fuse_page_it_iter_next(range, start, end); + } + + /* Check if we covered the entire range */ + if (current_start <= end) { + /* There's a gap at the end */ + up_read(&cache->lock); + return false; + } + + up_read(&cache->lock); + return true; +} + +/** + * fuse_dlm_write_grant_exists - does the inode hold an exclusive grant anywhere + * @fi: the fuse inode + * + * Unlike fuse_dlm_range_is_locked(), which asks whether one range is fully + * covered, this asks whether any part of the file is held exclusively. A + * client that holds a write grant may be sitting on dirty page cache the + * server has not seen, so its mtime and ctime run ahead of anything the + * server can report. + * + * Return: true if at least one recorded range is held for write + */ +bool fuse_dlm_write_grant_exists(struct fuse_inode *fi) +{ + struct fuse_dlm_cache *cache = &fi->dlm_locked_areas; + struct fuse_dlm_range *range; + bool held = false; + + down_read(&cache->lock); + for (range = fuse_dlm_find_overlapping(cache, 0, U64_MAX); range; + range = fuse_page_it_iter_next(range, 0, U64_MAX)) { + if (range->mode == FUSE_PCACHE_LK_WRITE) { + held = true; + break; + } + } + up_read(&cache->lock); + + return held; +} + +/** + * fuse_dlm_lock_is_held - check that a byte range is covered by a granted lock + * @fi: the fuse inode + * @offset: byte offset into the file (need not be page-aligned) + * @length: length of the region in bytes (need not be page-aligned) + * @mode: FUSE_PAGE_LOCK_READ or FUSE_PAGE_LOCK_WRITE + * + * Re-validation helper for fuse_get_dlm_lock() callers: checks the same + * page-aligned range a fuse_get_dlm_lock() call with these arguments + * requests, against the live lock tree. + */ +bool fuse_dlm_lock_is_held(struct fuse_inode *fi, loff_t offset, + size_t length, enum fuse_page_lock_mode mode) +{ + uint64_t end = (offset + length - 1) | (PAGE_SIZE - 1); + + /* + * An empty range needs no coverage. Reporting it held keeps the + * re-validating IO paths from re-requesting a lock the tree can + * never show (the page-aligned end would invert below). + */ + if (!length) + return true; + + return fuse_dlm_range_is_locked(fi, offset & PAGE_MASK, end, mode); +} + +/** + * fuse_get_dlm_lock - request a dlm lock from the fuse server + * @file: the file being accessed + * @offset: byte offset into the file (need not be page-aligned) + * @length: length of the region in bytes (need not be page-aligned) + * @mode: FUSE_PAGE_LOCK_READ or FUSE_PAGE_LOCK_WRITE + * + * Return: 0 when the range is covered by a recorded grant on return, + * FUSE_DLM_GRANT_UNRECORDED when the server granted the lock but + * recording it failed (covered cluster-wide, invisible to + * fuse_dlm_lock_is_held()), a negative error code otherwise. Callers + * re-validating the grant must not re-request on a nonzero return or + * they would spin. + */ +int fuse_get_dlm_lock(struct file *file, loff_t offset, + size_t length, enum fuse_page_lock_mode mode) +{ + struct fuse_file *ff = file->private_data; + struct inode *inode = file_inode(file); + struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_inode *fi = get_fuse_inode(inode); + struct fuse_mount *fm = ff->fm; + + FUSE_ARGS(args); + struct fuse_dlm_lock_in inarg; + struct fuse_dlm_lock_out outarg; + uint64_t gen; + int err; + + /* An empty range needs no lock. */ + if (!length) + return 0; + +restart: + /* note that this can be run from different processes + * at the same time. It is intentionally not protected + * since a DLM implementation in the FUSE server should take care + * of any races in lock requests. + * The early exit uses the same helper the callers re-validate + * with, so this check and a later fuse_dlm_lock_is_held() can + * never disagree about what counts as covered. */ + if (fuse_dlm_lock_is_held(fi, offset, length, mode)) + return 0; /* we already have this area locked */ + + /* + * Sample the revocation generation before the request leaves. + * The reply and a NOTIFY revoke are serviced on different + * threads, so a revoke aimed at the grant this request returns + * can be processed before the grant is recorded below -- + * recording it anyway would resurrect a dead grant that no later + * NOTIFY will ever remove. + */ + gen = fuse_dlm_revoke_gen(fi); + + memset(&inarg, 0, sizeof(inarg)); + inarg.fh = ff->fh; + + /* note that the offset and length don't have to be page aligned + * here but since we only get here on writeback caching we will + * send out page aligned requests */ + inarg.start = offset & PAGE_MASK; + inarg.end = (offset + length - 1) | (PAGE_SIZE - 1); + inarg.type = (mode == FUSE_PAGE_LOCK_WRITE) ? + FUSE_DLM_LOCK_WRITE : FUSE_DLM_LOCK_READ; + + args.opcode = FUSE_DLM_WB_LOCK; + args.nodeid = get_node_id(inode); + args.in_numargs = 1; + args.in_args[0].size = sizeof(inarg); + args.in_args[0].value = &inarg; + args.out_numargs = 1; + args.out_args[0].size = sizeof(outarg); + args.out_args[0].value = &outarg; + err = fuse_simple_request(fm, &args); + if (err == -ENOSYS) { + /* fuse server does not support dlm, save the info */ + fc->dlm = 0; + return err; + } + + if (err) + return err; + + if (inarg.start < outarg.start || inarg.end > outarg.end) { + /* fuse server is seriously broken */ + pr_warn("fuse: dlm lock request for %llu:%llu returned %llu:%llu bytes\n", + inarg.start, inarg.end, outarg.start, outarg.end); + fuse_abort_conn(fc); + return -EIO; + } + + /* + * The server granted the lock; record it so + * fuse_dlm_lock_is_held() sees it. + */ + err = fuse_dlm_lock_range_gen(fi, outarg.start, outarg.end, mode, gen); + if (err == -EAGAIN) { + /* + * A revoke was processed while the request was in flight; + * the grant may already be dead, so re-request instead of + * recording it. Retry until a grant survives long enough to + * be recorded: giving up here would hand the caller an error + * for a range no one else holds, and the write path turns + * that into a failed write. Each pass makes a fresh server + * round trip, so a revoke storm throttles this loop rather + * than spinning it. + */ + goto restart; + } + + /* + * A failure to record (small-allocation -ENOMEM) does not undo + * the grant: coverage exists cluster-wide, only the local + * bookkeeping is missing. Report that as + * FUSE_DLM_GRANT_UNRECORDED so callers neither fail an IO that + * is actually covered nor keep re-requesting a grant that will + * not become visible. + */ + if (err) + return FUSE_DLM_GRANT_UNRECORDED; + + return 0; +} diff --git a/fs/fuse/fuse_dlm_cache.h b/fs/fuse/fuse_dlm_cache.h new file mode 100644 index 00000000000000..30fdbb26bd3daf --- /dev/null +++ b/fs/fuse/fuse_dlm_cache.h @@ -0,0 +1,81 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +/* + * FUSE page cache lock implementation + */ + +#ifndef _FS_FUSE_DLM_CACHE_H +#define _FS_FUSE_DLM_CACHE_H + +#include +#include +#include +#include + + +struct fuse_inode; + +/* Lock modes for page ranges */ +enum fuse_page_lock_mode { FUSE_PAGE_LOCK_READ, FUSE_PAGE_LOCK_WRITE }; + +/* + * fuse_get_dlm_lock() result: the server granted the lock but recording + * it locally failed, leaving the grant invisible to + * fuse_dlm_lock_is_held(). The IO is covered cluster-wide; the caller + * must proceed without re-validating (a re-request would spin) instead + * of failing the IO. + */ +#define FUSE_DLM_GRANT_UNRECORDED 1 + +/* Page cache lock manager */ +struct fuse_dlm_cache { + /* Lock protecting the tree */ + struct rw_semaphore lock; + /* Interval tree of locked ranges */ + struct rb_root_cached ranges; + /* + * Bumped under @lock by every revocation + * (fuse_dlm_unlock_range(), fuse_dlm_cache_release_locks()); + * lets fuse_get_dlm_lock() order recording a reply's grant + * against revokes processed while the reply was in flight. + */ + uint64_t revoke_gen; +}; + +/* Initialize a page cache lock manager */ +int fuse_dlm_cache_init(struct fuse_inode *inode); + +/* Clean up a page cache lock manager */ +void fuse_dlm_cache_release_locks(struct fuse_inode *inode); + +/* Lock a range of pages */ +int fuse_dlm_lock_range(struct fuse_inode *inode, uint64_t start, + uint64_t end, enum fuse_page_lock_mode mode); + +/* As above, but refuse (-EAGAIN) if a revoke ran since @gen was sampled */ +int fuse_dlm_lock_range_gen(struct fuse_inode *inode, uint64_t start, + uint64_t end, enum fuse_page_lock_mode mode, + uint64_t gen); + +/* Sample the revocation generation (see fuse_dlm_lock_range_gen()) */ +uint64_t fuse_dlm_revoke_gen(struct fuse_inode *inode); + +/* Unlock a range of pages */ +int fuse_dlm_unlock_range(struct fuse_inode *inode, uint64_t start, + uint64_t end); + +/* Check if a page range is already locked */ +bool fuse_dlm_range_is_locked(struct fuse_inode *inode, uint64_t start, + uint64_t end, enum fuse_page_lock_mode mode); + +/* Re-validate a fuse_get_dlm_lock() grant against the live lock tree */ +bool fuse_dlm_lock_is_held(struct fuse_inode *inode, loff_t offset, + size_t length, enum fuse_page_lock_mode mode); + +/* Is any part of the file held for write? */ +bool fuse_dlm_write_grant_exists(struct fuse_inode *inode); + +/* This is the interface to the filesystem */ +int fuse_get_dlm_lock(struct file *file, loff_t offset, + size_t length, enum fuse_page_lock_mode mode); + +#endif /* _FS_FUSE_DLM_CACHE_H */ diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 7f16049387d15e..c18ded6cd394a6 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -23,6 +23,7 @@ #include #include #include +#include #include #include #include @@ -31,6 +32,7 @@ #include #include #include +#include "fuse_dlm_cache.h" /** Default max number of pages that can be used in a single read request */ #define FUSE_DEFAULT_MAX_PAGES_PER_REQ 32 @@ -83,6 +85,7 @@ extern struct mutex fuse_mutex; /** Module parameters */ extern unsigned int max_user_bgreq; extern unsigned int max_user_congthresh; +extern bool enable_large_folios; /* One forget request */ struct fuse_forget_link { @@ -113,6 +116,31 @@ struct fuse_backing { struct rcu_head rcu; }; +/** + * data structure to save the information that we have + * requested dlm locks for the given area from the fuse server +*/ +struct dlm_locked_area +{ + struct list_head list; + loff_t offset; + size_t size; +}; + +/* + * Force-DIO switch trigger: an exponentially weighted moving average of the + * interval (in jiffies) between FUSE_NOTIFY_INVAL_INODE data invalidations for + * a file. When the average spacing falls below FUSE_NOTIFY_DIO_INTERVAL -- a + * remote writer streaming invalidations -- and the file is open for writing + * here, it is latched into direct IO. These are the source-level (not + * externally tunable) parameters of the heuristic: EWMA weight 1/2^SHIFT, + * seeded and capped at SEED so it takes a short burst rather than a single + * notify to trip. + */ +#define FUSE_NOTIFY_DIO_INTERVAL max_t(unsigned long, HZ / 10, 1) +#define FUSE_NOTIFY_EWMA_SHIFT 2 +#define FUSE_NOTIFY_EWMA_SEED (2 * FUSE_NOTIFY_DIO_INTERVAL) + /** FUSE inode */ struct fuse_inode { /** Inode data */ @@ -168,6 +196,51 @@ struct fuse_inode { /* waitq for direct-io completion */ wait_queue_head_t direct_io_waitq; + + /* dlm locked areas we have sent lock requests for */ + struct fuse_dlm_cache dlm_locked_areas; + + /* + * Server-materialized size: an upper bound for how far + * the server holds file data. Seeded from + * server-reported attributes, advanced when the server + * acknowledges data (writeback completion, + * fuse_write_update_attr()), lowered again on + * truncate. A read-modify-write of a block starting + * at or past this bound needs no READ request under a + * held DLM write lock: the server has no data there + * (see fuse_iomap_read_folio_range()). Protected by + * fi->lock. + */ + loff_t server_size; + + /* + * Serializes buffered-write page-cache dirtying against + * the forced-direct-IO latch transition driven by + * NOTIFY_INVAL_INODE (fuse_reverse_inval_inode()), which + * may be delivered by the same server thread that still + * owes a reply to an in-flight write holding the inode + * lock. The buffered writer holds this for read around + * the dirtying and re-checks the latch under it; the + * NOTIFY latch site takes it for write (trylock, never + * blocking) around its page-cache invalidate + latch set. + * Only regular files initialise it -- it shares storage + * with the readdir-cache union arm. + */ + struct percpu_rw_semaphore *wb_inval_rwsem; + + /* + * Rate of FUSE_NOTIFY_INVAL_INODE data invalidations + * for this whole file: notify_stamp is the jiffies of + * the last one, notify_interval_ewma the EWMA of the + * inter-arrival interval (jiffies, scaled by + * 2^FUSE_NOTIFY_EWMA_SHIFT). A rapid stream (short + * average interval) with a local writer latches the + * inode into direct IO. Protected by fi->lock; regular + * files only (shares the readdir-cache union arm). + */ + unsigned long notify_stamp; + unsigned int notify_interval_ewma; }; /* readdir cache (directory only) */ @@ -244,6 +317,14 @@ enum { * or the fuse server has an exclusive "lease" on distributed fs */ FUSE_I_EXCLUSIVE, + /* + * Latched into direct IO: a NOTIFY_INVAL_INODE arrived while the file + * was open for writing here, so another (remote) entity is modifying it + * concurrently. Reads and writes are routed direct (shared-lock + * parallel dio) until the last writer closes or the inode is mmapped. + * See fuse_reverse_inval_inode()/fuse_file_io_open(). + */ + FUSE_I_FORCE_DIO, }; struct fuse_conn; @@ -377,6 +458,7 @@ union fuse_file_args { /** The request IO state (for asynchronous processing) */ struct fuse_io_priv { struct kref refcnt; + struct work_struct work; int async; spinlock_t lock; unsigned reqs; @@ -634,6 +716,17 @@ struct fuse_sync_bucket { struct rcu_head rcu; }; +/** + * DLM retry tracking for iomap write deadlock workaround. + * + * Temporary workaround until mainline iomap gains AOP_TRUNCATED_PAGE + * retry support. Tracks tasks that need to retry write operations due + * to DLM lock contention (-EAGAIN from FUSE server). + */ +struct fuse_dlm_retry { + bool retry_needed; +}; + /** * A Fuse connection. * @@ -772,6 +865,15 @@ struct fuse_conn { */ unsigned handle_killpriv_v2:1; + /* invalidate inode entries when doing inode invalidation */ + unsigned inval_inode_entries:1; + + /* expire inode entries when doing inode invalidation */ + unsigned expire_inode_entries:1; + + /* mark writeback-initiated SETATTR requests with FATTR_WRITEBACK */ + unsigned setattr_writeback:1; + /* * The following bitfields are only for optimization purposes * and hence races in setting them will not cause malfunction @@ -909,6 +1011,9 @@ struct fuse_conn { /* Is statx not implemented by fs? */ unsigned int no_statx:1; + /* do we have support for dlm in the fs? */ + unsigned int dlm:1; + /** Passthrough support for read/write IO */ unsigned int passthrough:1; @@ -924,6 +1029,9 @@ struct fuse_conn { /* Use io_uring for communication */ unsigned int io_uring; + + /* Does the filesystem support compound operations? */ + unsigned int compound_open_getattr:1; /** Maximum stack depth for passthrough backing files */ int max_stack_depth; @@ -945,6 +1053,9 @@ struct fuse_conn { /** Version counter for attribute changes */ atomic64_t attr_version; + /** Waitqueue for attr_version initialization */ + wait_queue_head_t attr_version_waitq; + /** Version counter for evict inode */ atomic64_t evict_ctr; @@ -995,6 +1106,13 @@ struct fuse_conn { /* Request timeout (in jiffies). 0 = no timeout */ unsigned int req_timeout; } timeout; + + /** + * XArray tracking tasks that need DLM retry. + * Maps task pointer -> struct fuse_dlm_retry. + * Temporary workaround for iomap write deadlock. + */ + struct xarray dlm_retry_tasks; }; /* @@ -1179,6 +1297,14 @@ struct fuse_io_args { void fuse_read_args_fill(struct fuse_io_args *ia, struct file *file, loff_t pos, size_t count, int opcode); +/* + * Helper functions to initialize fuse_args for common operations + */ +void fuse_open_args_fill(struct fuse_args *args, u64 nodeid, int opcode, + struct fuse_open_in *inarg, struct fuse_open_out *outarg); +void fuse_getattr_args_fill(struct fuse_args *args, u64 nodeid, + struct fuse_getattr_in *inarg, + struct fuse_attr_out *outarg); struct fuse_file *fuse_file_alloc(struct fuse_mount *fm, bool release); void fuse_file_free(struct fuse_file *ff); @@ -1270,6 +1396,8 @@ static inline ssize_t fuse_simple_idmap_request(struct mnt_idmap *idmap, return __fuse_simple_request(idmap, fm, args); } +ssize_t fuse_compound_request(struct fuse_mount *fm, struct fuse_args *args); + int fuse_simple_background(struct fuse_mount *fm, struct fuse_args *args, gfp_t gfp_flags); @@ -1277,6 +1405,14 @@ int fuse_simple_background(struct fuse_mount *fm, struct fuse_args *args, * Assign a unique id to a fuse request */ void fuse_request_assign_unique(struct fuse_iqueue *fiq, struct fuse_req *req); +struct fuse_compound_req; + +struct fuse_compound_req *fuse_compound_alloc(struct fuse_mount *fm, uint32_t flags); +int fuse_compound_add(struct fuse_compound_req *compound, + struct fuse_args *args); +ssize_t fuse_compound_send(struct fuse_compound_req *compound); +int fuse_compound_get_error(struct fuse_compound_req * compound, + int op_idx); /** * End a finished request @@ -1295,6 +1431,12 @@ void fuse_dentry_tree_cleanup(void); void fuse_epoch_work(struct work_struct *work); +/** + * Flush all pending requests and wait for them. Takes an optional timeout + * in jiffies. + */ +void fuse_flush_requests(struct fuse_conn *fc, unsigned long timeout); + /** * Invalidate inode attributes */ @@ -1541,9 +1683,17 @@ void fuse_inode_uncached_io_end(struct fuse_inode *fi); int fuse_file_io_open(struct file *file, struct inode *inode); void fuse_file_io_release(struct fuse_file *ff, struct inode *inode); +/* Inode latched into forced direct IO after a remote-modify notification */ +static inline bool fuse_inode_force_dio(struct inode *inode) +{ + return test_bit(FUSE_I_FORCE_DIO, &get_fuse_inode(inode)->state); +} + /* file.c */ struct fuse_file *fuse_file_open(struct fuse_mount *fm, u64 nodeid, - unsigned int open_flags, bool isdir); + struct inode *inode, + unsigned int open_flags, + bool isdir); void fuse_file_release(struct inode *inode, struct fuse_file *ff, unsigned int open_flags, fl_owner_t id, bool isdir); diff --git a/fs/fuse/fuse_trace.h b/fs/fuse/fuse_trace.h index bbe9ddd8c71696..e81c93b9614627 100644 --- a/fs/fuse/fuse_trace.h +++ b/fs/fuse/fuse_trace.h @@ -58,6 +58,7 @@ EM( FUSE_SYNCFS, "FUSE_SYNCFS") \ EM( FUSE_TMPFILE, "FUSE_TMPFILE") \ EM( FUSE_STATX, "FUSE_STATX") \ + EM( FUSE_DLM_WB_LOCK, "FUSE_DLM_WB_LOCK") \ EMe(CUSE_INIT, "CUSE_INIT") /* @@ -77,30 +78,55 @@ OPCODES #define EM(a, b) {a, b}, #define EMe(a, b) {a, b} -TRACE_EVENT(fuse_request_send, +#define FUSE_REQ_TRACE_FIELDS \ + __field(dev_t, connection) \ + __field(uint64_t, unique) \ + __field(enum fuse_opcode, opcode) \ + __field(uint32_t, len) \ + +#define FUSE_REQ_TRACE_ASSIGN(req) \ + do { \ + __entry->connection = req->fm->fc->dev; \ + __entry->unique = req->in.h.unique; \ + __entry->opcode = req->in.h.opcode; \ + __entry->len = req->in.h.len; \ + } while (0) + + +TRACE_EVENT(fuse_request_enqueue, TP_PROTO(const struct fuse_req *req), + TP_ARGS(req), + TP_STRUCT__entry(FUSE_REQ_TRACE_FIELDS), + TP_fast_assign(FUSE_REQ_TRACE_ASSIGN(req)), + TP_printk("connection %u req %llu opcode %u (%s) len %u ", + __entry->connection, __entry->unique, __entry->opcode, + __print_symbolic(__entry->opcode, OPCODES), __entry->len) +); + +TRACE_EVENT(fuse_request_bg_enqueue, + TP_PROTO(const struct fuse_req *req), TP_ARGS(req), + TP_STRUCT__entry(FUSE_REQ_TRACE_FIELDS), + TP_fast_assign(FUSE_REQ_TRACE_ASSIGN(req)), - TP_STRUCT__entry( - __field(dev_t, connection) - __field(uint64_t, unique) - __field(enum fuse_opcode, opcode) - __field(uint32_t, len) - ), + TP_printk("connection %u req %llu opcode %u (%s) len %u ", + __entry->connection, __entry->unique, __entry->opcode, + __print_symbolic(__entry->opcode, OPCODES), __entry->len) +); - TP_fast_assign( - __entry->connection = req->fm->fc->dev; - __entry->unique = req->in.h.unique; - __entry->opcode = req->in.h.opcode; - __entry->len = req->in.h.len; - ), +TRACE_EVENT(fuse_request_send, + TP_PROTO(const struct fuse_req *req), + TP_ARGS(req), + TP_STRUCT__entry(FUSE_REQ_TRACE_FIELDS), + TP_fast_assign(FUSE_REQ_TRACE_ASSIGN(req)), TP_printk("connection %u req %llu opcode %u (%s) len %u ", __entry->connection, __entry->unique, __entry->opcode, __print_symbolic(__entry->opcode, OPCODES), __entry->len) ); + TRACE_EVENT(fuse_request_end, TP_PROTO(const struct fuse_req *req), diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index c795abe47a4f4a..431ae386872646 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -7,6 +7,7 @@ */ #include "fuse_i.h" +#include "fuse_dlm_cache.h" #include "fuse_dev_i.h" #include "dev_uring_i.h" @@ -32,6 +33,27 @@ MODULE_AUTHOR("Miklos Szeredi "); MODULE_DESCRIPTION("Filesystem in Userspace"); MODULE_LICENSE("GPL"); +static bool __read_mostly enable_compound; +module_param(enable_compound, bool, 0644); +MODULE_PARM_DESC(enable_uring, "Enable fuse compounds"); + +bool __read_mostly enable_large_folios = true; +module_param(enable_large_folios, bool, 0644); +MODULE_PARM_DESC(enable_large_folios, "Enable large folios support"); + +/* + * Gate for the notify-driven direct-IO latch (see + * fuse_reverse_inval_inode()): when a remote writer keeps invalidating a + * file that is also open for writing here, the inode is switched to + * direct IO until its last writer closes. Off by default -- it trades + * the writeback cache away for the duration, which only pays off on + * workloads that actually see such storms. + */ +static bool __read_mostly enable_notify_dio; +module_param(enable_notify_dio, bool, 0644); +MODULE_PARM_DESC(enable_notify_dio, + "Latch a contended inode to direct IO on an invalidation notify storm"); + static struct kmem_cache *fuse_inode_cachep; struct list_head fuse_conn_list; DEFINE_MUTEX(fuse_mutex); @@ -39,7 +61,7 @@ DECLARE_WAIT_QUEUE_HEAD(fuse_dev_waitq); static int set_global_limit(const char *val, const struct kernel_param *kp); -unsigned int fuse_max_pages_limit = 256; +unsigned int fuse_max_pages_limit = 4097; /* default is no timeout */ unsigned int fuse_default_req_timeout; unsigned int fuse_max_req_timeout; @@ -195,6 +217,24 @@ static void fuse_evict_inode(struct inode *inode) WARN_ON(fi->iocachectr != 0); WARN_ON(!list_empty(&fi->write_files)); WARN_ON(!list_empty(&fi->queued_writes)); + fuse_dlm_cache_release_locks(fi); + } + + /* + * Free the coherency gate here rather than in ->free_inode: that runs + * from an RCU callback, where percpu_free_rwsem() may sleep in + * rcu_sync_dtor() if the write side has not fully quiesced. No user + * can remain by eviction time: gate readers hold a file reference and + * a concurrent notify holds an inode reference. wb_inval_rwsem lives + * in the regular-file union arm and is only ever allocated for regular + * files, so gate on S_ISREG (but not fuse_is_bad() -- bad-marked + * regular files still own a gate); a directory's overlapping + * readdir-cache fields must not be misread. + */ + if (S_ISREG(inode->i_mode) && fi->wb_inval_rwsem) { + percpu_free_rwsem(fi->wb_inval_rwsem); + kfree(fi->wb_inval_rwsem); + fi->wb_inval_rwsem = NULL; } } @@ -246,6 +286,7 @@ void fuse_change_attributes_common(struct inode *inode, struct fuse_attr *attr, set_mask_bits(&fi->inval_mask, STATX_BASIC_STATS, 0); fi->attr_version = atomic64_inc_return(&fc->attr_version); + wake_up_all(&fc->attr_version_waitq); fi->i_time = attr_valid; inode->i_ino = fuse_squash_ino(attr->ino); @@ -319,12 +360,74 @@ u32 fuse_get_cache_mask(struct inode *inode) { struct fuse_conn *fc = get_fuse_conn(inode); - if (!fc->writeback_cache || !S_ISREG(inode->i_mode)) + if (!fc->writeback_cache || !S_ISREG(inode->i_mode) || fc->dlm) return 0; return STATX_MTIME | STATX_CTIME | STATX_SIZE; } +/* + * Which cached attributes survive a server reply. + * + * Without DLM this is fuse_get_cache_mask(): with the writeback cache on, + * writes update mtime and ctime and may extend i_size locally, the server + * knows about none of it, so the cached values win. + * + * With DLM the server is the authority (fuse_get_cache_mask() returns 0), + * because another node may have changed the file behind us and only the + * server can say so. That holds for the parts of the file we do not own. A + * write grant means no other node can touch the range until we are revoked, + * so anything the server reports about it is at best as new as what we have, + * and older if we still have unwritten data there. Keep the cached values + * for exactly what the grant covers: + * + * - size, when the server reports less than i_size and the tail it does not + * know about, [srv_size, i_size), is entirely under a write grant. Taking + * the server's answer would shrink i_size and have truncate_pagecache() + * throw the unwritten tail away. + * - mtime and ctime, while a write grant covers unwritten data: our writes + * have stamped them locally and the server's stamps predate them. Only + * while the cache is actually dirty, not for as long as the grant lives: + * a grant is held until it is revoked or the inode is evicted, and past + * the writeback the server's stamps are the newer ones. Keeping ours + * beyond that would hide a remote chown or chmod indefinitely. + * + * A remote truncate cannot slip through. It has to revoke the grant first, + * and the revoke launders the tail and drops the grant, so by the time the + * smaller size is reported neither check holds and the server's answer is + * applied as usual. A grant the server made but that could not be recorded + * (FUSE_DLM_GRANT_UNRECORDED) is invisible to the lock tree and falls back to + * trusting the server, as before. + * + * Must be called without fi->lock: the lock tree query sleeps. + */ +static u32 fuse_attr_cache_mask(struct inode *inode, struct fuse_attr *attr, + bool have_size) +{ + struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_inode *fi = get_fuse_inode(inode); + u32 cache_mask = fuse_get_cache_mask(inode); + loff_t size = i_size_read(inode); + + if (cache_mask || !fc->dlm || !fc->writeback_cache || + !S_ISREG(inode->i_mode)) + return cache_mask; + + if (!fuse_dlm_write_grant_exists(fi)) + return cache_mask; + + if (mapping_tagged(inode->i_mapping, PAGECACHE_TAG_DIRTY) || + mapping_tagged(inode->i_mapping, PAGECACHE_TAG_WRITEBACK)) + cache_mask |= STATX_MTIME | STATX_CTIME; + + if (have_size && size > (loff_t) attr->size && + fuse_dlm_lock_is_held(fi, attr->size, size - attr->size, + FUSE_PAGE_LOCK_WRITE)) + cache_mask |= STATX_SIZE; + + return cache_mask; +} + static void fuse_change_attributes_i(struct inode *inode, struct fuse_attr *attr, struct fuse_statx *sx, u64 attr_valid, u64 attr_version, u64 evict_ctr) @@ -334,14 +437,14 @@ static void fuse_change_attributes_i(struct inode *inode, struct fuse_attr *attr u32 cache_mask; loff_t oldsize; struct timespec64 old_mtime; + bool have_size = !sx || (sx->mask & STATX_SIZE); + u64 srv_size; + + cache_mask = fuse_attr_cache_mask(inode, attr, have_size); spin_lock(&fi->lock); - /* - * In case of writeback_cache enabled, writes update mtime, ctime and - * may update i_size. In these cases trust the cached value in the - * inode. - */ - cache_mask = fuse_get_cache_mask(inode); + srv_size = attr->size; + if (cache_mask & STATX_SIZE) attr->size = i_size_read(inode); @@ -360,6 +463,19 @@ static void fuse_change_attributes_i(struct inode *inode, struct fuse_attr *attr return; } + /* + * srv_size is the size the server reported before the writeback + * cache_mask above replaced attr->size with the local value. It + * bounds how far the server can hold data, letting the iomap write + * path zero-fill expansion read-modify-writes instead of sending + * READ requests, see fuse_iomap_read_folio_range(). Only ever grow + * it here: stale attributes were rejected above and truncation + * lowers it directly. + */ + if (have_size && S_ISREG(inode->i_mode) && + (loff_t) srv_size > fi->server_size) + fi->server_size = srv_size; + old_mtime = inode_get_mtime(inode); fuse_change_attributes_common(inode, attr, sx, attr_valid, cache_mask, evict_ctr); @@ -374,7 +490,17 @@ static void fuse_change_attributes_i(struct inode *inode, struct fuse_attr *attr i_size_write(inode, attr->size); spin_unlock(&fi->lock); - if (!cache_mask && S_ISREG(inode->i_mode)) { + /* + * Only do page cache invalidation when the size was not served from + * the cache (writeback_cache disabled, or no grant covering the tail) + * AND the relevant attributes (SIZE/MTIME) were actually returned by + * the server. This has to key off STATX_SIZE alone: i_size_write() + * above took the server's size for any mask without that bit, and the + * cache has to be truncated to match it. The mtime branch neutralises + * itself when STATX_MTIME is set, since attr->mtime then holds the + * value old_mtime was read from. + */ + if (!(cache_mask & STATX_SIZE) && S_ISREG(inode->i_mode)) { bool inval = false; if (oldsize != attr->size) { @@ -553,9 +679,131 @@ struct inode *fuse_ilookup(struct fuse_conn *fc, u64 nodeid, return NULL; } +static void fuse_prune_aliases(struct inode *inode) +{ + struct dentry *dentry; + + spin_lock(&inode->i_lock); + hlist_for_each_entry(dentry, &inode->i_dentry, d_u.d_alias) { + fuse_invalidate_entry_cache(dentry); + } + spin_unlock(&inode->i_lock); + + d_prune_aliases(inode); +} + +static void fuse_invalidate_inode_entry(struct inode *inode) +{ + struct dentry *dentry; + + if (S_ISDIR(inode->i_mode)) { + /* For directories, use d_invalidate to handle children and submounts */ + dentry = d_find_alias(inode); + if (dentry) { + d_invalidate(dentry); + fuse_invalidate_entry_cache(dentry); + dput(dentry); + } + } else { + /* For regular files, just unhash the dentry */ + spin_lock(&inode->i_lock); + hlist_for_each_entry(dentry, &inode->i_dentry, d_u.d_alias) { + spin_lock(&dentry->d_lock); + if (!d_unhashed(dentry)) + __d_drop(dentry); + spin_unlock(&dentry->d_lock); + fuse_invalidate_entry_cache(dentry); + } + spin_unlock(&inode->i_lock); + } +} + +/* + * Fold one FUSE_NOTIFY_INVAL_INODE data invalidation into the per-inode + * moving average of the notification inter-arrival interval and report whether + * the file is now "hot" -- notifications are arriving fast enough (short + * average interval) that a remote writer is repeatedly invalidating it. The + * average is an EWMA (weight 1/2^FUSE_NOTIFY_EWMA_SHIFT); the sample is clamped + * to FUSE_NOTIFY_EWMA_SEED so a notify after a long idle only cools the average + * and cannot overflow the accumulator. Must be called under fi->lock; called + * for every data invalidation so the average stays current even while no local + * writer is open. + */ +static bool fuse_notify_inval_hot(struct fuse_inode *fi) +{ + unsigned long now = jiffies; + unsigned long sample; + unsigned int avg; + + sample = min_t(unsigned long, now - fi->notify_stamp, + FUSE_NOTIFY_EWMA_SEED); + fi->notify_stamp = now; + + /* E += sample - (E >> SHIFT); avg = E >> SHIFT */ + fi->notify_interval_ewma += sample - + (fi->notify_interval_ewma >> FUSE_NOTIFY_EWMA_SHIFT); + avg = fi->notify_interval_ewma >> FUSE_NOTIFY_EWMA_SHIFT; + + return avg < FUSE_NOTIFY_DIO_INTERVAL; +} + +/* + * Revoke the DLM grants backing an invalidated byte range. Grants are + * recorded page-aligned, so widen the revoke to page boundaries: dropping + * more than the server invalidated only costs a re-request, dropping less + * would leave a stale grant that fuse_dlm_lock_is_held() keeps trusting. + * len <= 0 means "invalidate to EOF" (see fuse_notify_inval_inode()) and + * revokes through U64_MAX -- it must not become an inverted range, which + * fuse_dlm_unlock_range() rejects without removing anything. + */ +static void fuse_dlm_revoke_inval_range(struct fuse_inode *fi, loff_t offset, + loff_t len) +{ + uint64_t start = (uint64_t)offset & PAGE_MASK; + uint64_t end = len <= 0 ? U64_MAX : + (((uint64_t)offset + len - 1) | (PAGE_SIZE - 1)); + + fuse_dlm_unlock_range(fi, start, end); +} + +/* + * Drop a page-cache range on behalf of a NOTIFY invalidate. + * + * invalidate_inode_pages2_range() waits out folios under writeback and + * launders dirty ones, both of which need a FUSE_WRITE reply. While + * writepages are frozen (fuse_set_nowrite(): truncate, O_TRUNC open, fsync, + * pre-SETATTR flush) no reply can arrive, because fuse_flush_writepages() + * parks the request on fi->queued_writes until fuse_release_nowrite(). A + * server that revokes from inside the handler it is revoking for then + * deadlocks against its own reply. fuse_do_setattr() states the same rule + * for its own invalidate. + * + * So while frozen use invalidate_mapping_pages(), which skips dirty and + * under-writeback folios and never blocks. The stale clean folios still + * go, and the freezes that span a request drop the cache themselves once + * they complete: fuse_do_setattr() invalidates the mapping after releasing + * the freeze, the O_TRUNC open path calls truncate_pagecache(). + */ +static void fuse_notify_invalidate_range(struct inode *inode, pgoff_t start, + pgoff_t end) +{ + struct fuse_inode *fi = get_fuse_inode(inode); + bool frozen; + + spin_lock(&fi->lock); + frozen = fi->writectr < 0; + spin_unlock(&fi->lock); + + if (frozen) + invalidate_mapping_pages(inode->i_mapping, start, end); + else + invalidate_inode_pages2_range(inode->i_mapping, start, end); +} + int fuse_reverse_inval_inode(struct fuse_conn *fc, u64 nodeid, loff_t offset, loff_t len) { + struct percpu_rw_semaphore *wb_sem = NULL; struct fuse_inode *fi; struct inode *inode; pgoff_t pg_start; @@ -566,20 +814,143 @@ int fuse_reverse_inval_inode(struct fuse_conn *fc, u64 nodeid, return -ENOENT; fi = get_fuse_inode(inode); + spin_lock(&fi->lock); + while (fi->attr_version == 0) { + spin_unlock(&fi->lock); + wait_event(fc->attr_version_waitq, READ_ONCE(fi->attr_version) != 0); + spin_lock(&fi->lock); + } + fi->attr_version = atomic64_inc_return(&fc->attr_version); spin_unlock(&fi->lock); + + if (fc->inval_inode_entries) + fuse_invalidate_inode_entry(inode); + else if (fc->expire_inode_entries) + fuse_prune_aliases(inode); fuse_invalidate_attr(inode); forget_all_cached_acls(inode); + security_inode_invalidate_secctx(inode); if (offset >= 0) { pg_start = offset >> PAGE_SHIFT; if (len <= 0) pg_end = -1; else pg_end = (offset + len - 1) >> PAGE_SHIFT; - invalidate_inode_pages2_range(inode->i_mapping, - pg_start, pg_end); + + /* + * A data invalidation means another (remote) entity is modifying + * the file. Two things happen here: + * + * 1. Coherency. Drop the affected page-cache range so no local + * read returns a folio the remote modify has superseded. This + * runs under the write side of the per-inode coherency gate + * (wb_inval_rwsem), which fences cache-serving buffered reads + * and buffered writes out for the whole invalidate. Unlike the + * old best-effort trylock this BLOCKS -- the notify has + * priority: percpu_down_write() parks new gate readers, drains + * in-flight ones, then invalidates. A blocking writer here is + * safe only under a server that services request replies on + * threads other than the one delivering this notify: the write + * side waits for gate readers to drain, and a cache-miss read + * holds the read side across its FUSE_READ round-trip. redfs' + * dlm server provides that contract; a server that cannot must + * not enable writeback+dlm. + * + * 2. Latch. Keep a moving average (fuse_notify_inval_hot(), under + * fi->lock, updated for every data invalidation) of how fast + * these arrive; when they come in a rapid stream -- a remote + * writer repeatedly invalidating -- and the inode is also open + * for writing here, latch it into direct IO until the last + * writer closes or it is mmapped. When latched, drop the whole + * mapping rather than just the notified range, or dirty folios + * outside it would be invisible to the forced direct reads + * (stale read / lost write). Latching is opt-in via the + * enable_notify_dio module parameter and off by default; the + * average is kept up to date either way, so enabling it at + * runtime takes effect on the next storm rather than after a + * warm-up. Clearing it at runtime stops new latches but lets + * already-latched inodes run out on the usual exits (last + * writer closes, or mmap). + * + * The gate (and the average) exist only for writeback+dlm regular + * files; elsewhere wb_sem is NULL and the invalidate runs + * unserialized (best-effort), as before. An mmapped inode + * keeps the gate -- fuse_cache_read_iter() and + * fuse_cache_write_iter() enter it unconditionally and rely + * on the revoke staying fenced -- but is never latched: + * a mapping needs the page cache, and fuse_file_mmap() + * reverts any latch it races with. + */ + if (S_ISREG(inode->i_mode) && fc->writeback_cache && + fc->dlm && !FUSE_IS_DAX(inode) && + !fuse_inode_backing(fi)) + wb_sem = fi->wb_inval_rwsem; + + if (wb_sem) { + bool hot, has_writer, latched = false; + + spin_lock(&fi->lock); + hot = fuse_notify_inval_hot(fi); + has_writer = !list_empty(&fi->write_files); + spin_unlock(&fi->lock); + + /* + * Priority write side: park new gate readers, + * drain in-flight ones, then invalidate. Blocks + * (unlike the old trylock) -- see the contract in + * the comment above. + */ + percpu_down_write(wb_sem); + + /* + * Revoke the DLM lock range under the gate write + * side, atomically with the page drop: gate readers + * re-validate their grant right after entering, and + * a grant that passed that check must stay visible + * for their whole gate hold. + */ + if (fc->dlm && fc->writeback_cache) + fuse_dlm_revoke_inval_range(fi, offset, len); + + if (enable_notify_dio && hot && has_writer && + !mapping_mapped(inode->i_mapping) && + !fuse_inode_force_dio(inode)) { + spin_lock(&fi->lock); + if (!list_empty(&fi->write_files)) { + set_bit(FUSE_I_FORCE_DIO, &fi->state); + latched = true; + } + spin_unlock(&fi->lock); + } + + /* + * Latched: drop the whole mapping (dirty folios + * outside the notified range would be invisible to + * the forced direct reads). Otherwise just the + * notified range. + */ + if (fuse_inode_force_dio(inode)) + fuse_notify_invalidate_range(inode, 0, -1); + else + fuse_notify_invalidate_range(inode, pg_start, + pg_end); + + percpu_up_write(wb_sem); + + if (latched) + pr_info_ratelimited("FUSE: inode %llu latched to direct IO on invalidation notify storm\n", + nodeid); + } else { + /* No gate on this inode (DAX, backing, non-regular, + * or the gate allocation failed): drop the lock + * range unserialized (best-effort), as before. */ + if (fc->dlm && fc->writeback_cache) + fuse_dlm_revoke_inval_range(fi, offset, len); + fuse_notify_invalidate_range(inode, pg_start, pg_end); + } } iput(inode); return 0; @@ -979,6 +1350,7 @@ void fuse_conn_init(struct fuse_conn *fc, struct fuse_mount *fm, atomic_set(&fc->epoch, 1); INIT_WORK(&fc->epoch_work, fuse_epoch_work); init_waitqueue_head(&fc->blocked_waitq); + init_waitqueue_head(&fc->attr_version_waitq); fuse_iqueue_init(&fc->iq, fiq_ops, fiq_priv); INIT_LIST_HEAD(&fc->bg_queue); INIT_LIST_HEAD(&fc->entry); @@ -991,6 +1363,12 @@ void fuse_conn_init(struct fuse_conn *fc, struct fuse_mount *fm, fc->blocked = 0; fc->initialized = 0; fc->connected = 1; + fc->dlm = 1; + + /* module option for now */ + fc->compound_open_getattr = enable_compound; + + xa_init(&fc->dlm_retry_tasks); atomic64_set(&fc->attr_version, 1); atomic64_set(&fc->evict_ctr, 1); get_random_bytes(&fc->scramble_key, sizeof(fc->scramble_key)); @@ -1043,6 +1421,7 @@ void fuse_conn_put(struct fuse_conn *fc) } if (IS_ENABLED(CONFIG_FUSE_PASSTHROUGH)) fuse_backing_files_free(fc); + xa_destroy(&fc->dlm_retry_tasks); call_rcu(&fc->rcu, delayed_release); } EXPORT_SYMBOL_GPL(fuse_conn_put); @@ -1456,6 +1835,12 @@ static void process_init_reply(struct fuse_mount *fm, struct fuse_args *args, if (flags & FUSE_REQUEST_TIMEOUT) timeout = arg->request_timeout; + if (flags & FUSE_INVAL_INODE_ENTRY) + fc->inval_inode_entries = 1; + if (flags & FUSE_EXPIRE_INODE_ENTRY) + fc->expire_inode_entries = 1; + if (flags & FUSE_SETATTR_WRITEBACK) + fc->setattr_writeback = 1; } else { ra_pages = fc->max_read / PAGE_SIZE; fc->no_lock = 1; @@ -1464,7 +1849,10 @@ static void process_init_reply(struct fuse_mount *fm, struct fuse_args *args, init_server_timeout(fc, timeout); - fm->sb->s_bdi->ra_pages = + if (CAP_SYS_ADMIN) + fm->sb->s_bdi->ra_pages = ra_pages; + else + fm->sb->s_bdi->ra_pages = min(fm->sb->s_bdi->ra_pages, ra_pages); fc->minor = arg->minor; fc->max_write = arg->minor < 5 ? 4096 : arg->max_write; @@ -1492,8 +1880,7 @@ static struct fuse_init_args *fuse_new_init(struct fuse_mount *fm) ia->in.major = FUSE_KERNEL_VERSION; ia->in.minor = FUSE_KERNEL_MINOR_VERSION; ia->in.max_readahead = fm->sb->s_bdi->ra_pages * PAGE_SIZE; - flags = - FUSE_ASYNC_READ | FUSE_POSIX_LOCKS | FUSE_ATOMIC_O_TRUNC | + flags = FUSE_ASYNC_READ | FUSE_POSIX_LOCKS | FUSE_ATOMIC_O_TRUNC | FUSE_EXPORT_SUPPORT | FUSE_BIG_WRITES | FUSE_DONT_MASK | FUSE_SPLICE_WRITE | FUSE_SPLICE_MOVE | FUSE_SPLICE_READ | FUSE_FLOCK_LOCKS | FUSE_HAS_IOCTL_DIR | FUSE_AUTO_INVAL_DATA | @@ -1505,8 +1892,10 @@ static struct fuse_init_args *fuse_new_init(struct fuse_mount *fm) FUSE_HANDLE_KILLPRIV_V2 | FUSE_SETXATTR_EXT | FUSE_INIT_EXT | FUSE_SECURITY_CTX | FUSE_CREATE_SUPP_GROUP | FUSE_HAS_EXPIRE_ONLY | FUSE_DIRECT_IO_ALLOW_MMAP | - FUSE_NO_EXPORT_SUPPORT | FUSE_HAS_RESEND | FUSE_ALLOW_IDMAP | - FUSE_REQUEST_TIMEOUT; + FUSE_NO_EXPORT_SUPPORT | FUSE_INVAL_INODE_ENTRY | + FUSE_EXPIRE_INODE_ENTRY | FUSE_URING_REDUCED_Q | + FUSE_EXPIRE_INODE_ENTRY | + FUSE_REQUEST_TIMEOUT | FUSE_SETATTR_WRITEBACK; #ifdef CONFIG_FUSE_DAX if (fm->fc->dax) flags |= FUSE_MAP_ALIGNMENT; @@ -1594,7 +1983,7 @@ static int fuse_bdi_init(struct fuse_conn *fc, struct super_block *sb) if (err) return err; - sb->s_bdi->capabilities |= BDI_CAP_STRICTLIMIT; + sb->s_bdi->capabilities &= ~BDI_CAP_STRICTLIMIT; /* * For a single fuse filesystem use max 1% of dirty + @@ -2087,6 +2476,7 @@ void fuse_conn_destroy(struct fuse_mount *fm) { struct fuse_conn *fc = fm->fc; + fuse_flush_requests(fc, 30 * HZ); if (fc->destroy) fuse_send_destroy(fm); diff --git a/fs/fuse/ioctl.c b/fs/fuse/ioctl.c index fdc175e93f7474..07a02e47b2c3a6 100644 --- a/fs/fuse/ioctl.c +++ b/fs/fuse/ioctl.c @@ -494,7 +494,7 @@ static struct fuse_file *fuse_priv_ioctl_prepare(struct inode *inode) if (!S_ISREG(inode->i_mode) && !isdir) return ERR_PTR(-ENOTTY); - return fuse_file_open(fm, get_node_id(inode), O_RDONLY, isdir); + return fuse_file_open(fm, get_node_id(inode), NULL, O_RDONLY, isdir); } static void fuse_priv_ioctl_cleanup(struct inode *inode, struct fuse_file *ff) diff --git a/fs/fuse/iomode.c b/fs/fuse/iomode.c index 3728933188f307..1076d954de94eb 100644 --- a/fs/fuse/iomode.c +++ b/fs/fuse/iomode.c @@ -232,6 +232,16 @@ int fuse_file_io_open(struct file *file, struct inode *inode) !(ff->open_flags & FOPEN_PASSTHROUGH)) return 0; + /* + * The inode was latched into direct IO after a remote-modify + * notification arrived while it was open for writing here. Open this + * file uncached as well so its IO is routed direct and it does not + * re-enter caching mode. + */ + if (test_bit(FUSE_I_FORCE_DIO, &fi->state) && + !(ff->open_flags & FOPEN_PASSTHROUGH)) + return 0; + if (ff->open_flags & FOPEN_PASSTHROUGH) err = fuse_file_passthrough_open(inode, file); else diff --git a/include/uapi/linux/fuse.h b/include/uapi/linux/fuse.h index c13e1f9a2f12bd..ebe2735c186e08 100644 --- a/include/uapi/linux/fuse.h +++ b/include/uapi/linux/fuse.h @@ -371,6 +371,17 @@ struct fuse_file_lock { #define FATTR_LOCKOWNER (1 << 9) #define FATTR_CTIME (1 << 10) #define FATTR_KILL_SUIDGID (1 << 11) +/* + * Not an attribute selector: marks the request as a kernel-initiated + * writeback of locally owned attributes rather than a userspace-initiated + * change. Only sent if the server negotiated FUSE_SETATTR_WRITEBACK. + * + * The bit is deliberately far above the sequentially allocated FATTR_* + * range: libfuse mirrors these bits into its own FUSE_SET_ATTR_* space, + * which has its own allocations from bit 12 upwards, and only a value that + * is free on both sides can be passed through without translation. + */ +#define FATTR_WRITEBACK (1 << 30) /** * Flags returned by the OPEN request @@ -448,6 +459,12 @@ struct fuse_file_lock { * FUSE_OVER_IO_URING: Indicate that client supports io-uring * FUSE_REQUEST_TIMEOUT: kernel supports timing out requests. * init_out.request_timeout contains the timeout (in secs) + * FUSE_INVAL_INODE_ENTRY: invalidate inode aliases when doing inode invalidation + * FUSE_EXPIRE_INODE_ENTRY: expire inode aliases when doing inode invalidation + * FUSE_URING_REDUCED_Q: Client (kernel) supports less queues - Server is free + * to register between 1 and nr-core io-uring queues + * FUSE_SETATTR_WRITEBACK: kernel marks writeback-initiated SETATTR requests + * with FATTR_WRITEBACK */ #define FUSE_ASYNC_READ (1 << 0) #define FUSE_POSIX_LOCKS (1 << 1) @@ -495,6 +512,11 @@ struct fuse_file_lock { #define FUSE_ALLOW_IDMAP (1ULL << 40) #define FUSE_OVER_IO_URING (1ULL << 41) #define FUSE_REQUEST_TIMEOUT (1ULL << 42) +#define FUSE_ALIGN_PG_ORDER (1ULL << 50) +#define FUSE_SETATTR_WRITEBACK (1ULL << 58) +#define FUSE_URING_REDUCED_Q (1ULL << 59) +#define FUSE_INVAL_INODE_ENTRY (1ULL << 60) +#define FUSE_EXPIRE_INODE_ENTRY (1ULL << 61) /** * CUSE INIT request/reply flags @@ -662,7 +684,17 @@ enum fuse_opcode { FUSE_SYNCFS = 50, FUSE_TMPFILE = 51, FUSE_STATX = 52, - FUSE_COPY_FILE_RANGE_64 = 53, + FUSE_COPY_FILE_RANGE_64 = 53, + + /* Operations which have not been merged into upstream */ + FUSE_DLM_WB_LOCK = 100, + + /* A compound request works like multiple simple requests. + * This is a special case for calls that can be combined atomic on the + * fuse server. If the server actually does atomically execute the command is + * left to the fuse server implementation. + */ + FUSE_COMPOUND = 101, /* CUSE specific operations */ CUSE_INIT = 4096, @@ -1245,6 +1277,74 @@ struct fuse_supp_groups { uint32_t groups[]; }; +/** + * Type of the dlm lock requested + */ +enum fuse_dlm_lock_type { + FUSE_DLM_LOCK_NONE = 0, + FUSE_DLM_LOCK_READ = 1, + FUSE_DLM_LOCK_WRITE = 2, + FUSE_DLM_PAGE_MKWRITE = 3, +}; + +/** + * struct fuse_dlm_lock_in - Lock request + * @fh: file handle + * @offset: offset into the file + * @size: size of the locked region + * @type: type of lock + */ +struct fuse_dlm_lock_in { + uint64_t fh; + uint64_t start; + uint64_t end; + uint32_t type; + uint32_t reserved; +}; + + +/** + * struct fuse_dlm_lock_out - Lock response + * @locksize: how many bytes where locked by the call + * (most of the time we want to lock more than is requested + * to reduce number of calls) + */ +struct fuse_dlm_lock_out { + uint64_t start; + uint64_t end; + uint64_t reserved; +}; + +/* + * Compound request header + * + * This header is followed by the fuse requests + */ +struct fuse_compound_in { + uint32_t count; /* Number of operations */ + uint32_t flags; /* Compound flags */ + + /* Total size of all results. + * This is needed for preallocating the whole result for all + * commands in this compound. + */ + uint32_t result_size; + uint64_t reserved; +}; + +/* + * Compound response header + * + * This header is followed by complete fuse responses + */ +struct fuse_compound_out { + uint32_t count; /* Number of results */ + uint32_t flags; /* Result flags */ + uint64_t reserved; +}; + +#define FUSE_MAX_COMPOUND_OPS 16 /* Maximum operations per compound */ + /** * Size of the ring buffer header */ diff --git a/ubuntu/igh-ecat/master/Makefile b/ubuntu/igh-ecat/master/Makefile index 8aef742ebb2f74..4f8fc539c6ab6c 100644 --- a/ubuntu/igh-ecat/master/Makefile +++ b/ubuntu/igh-ecat/master/Makefile @@ -1,5 +1,5 @@ ccflags-y := -I$(src)/../ \ - -Wmaybe-uninitialized + -Wuninitialized obj-$(CONFIG_IGH_ECAT) += ec_master.o