From d09f85ed8db24c7a56a5735ccc03277d88d5ab99 Mon Sep 17 00:00:00 2001 From: Pranjal Shrivastava Date: Fri, 14 Aug 2026 14:32:51 +0000 Subject: nfs: make nfs_page pin-aware Modernizing the NFS Direct I/O path to use iov_iter_extract_pages() introduces page pinning (GUP) instead of standard page referencing. To handle this correctly, nfs_page must track whether it holds a pin or a standard reference. Introduce a new flag, PG_PINNED, to struct nfs_page. Update the creation path (nfs_page_create_from_page and nfs_page_create_from_folio) to accept a pinned bool and set the flag accordingly. If the page is pinned, we skip the existing reference increment (get_page/folio_get) as the pin itself acts as a reference. Update nfs_clear_request() & nfs_direct_release_pages() to use unpin_user_page() or unpin_user_folio() instead of only refcount decrement (put_page) when PG_PINNED flag is set. Finally, ensure subrequests inherit the pinning status from their parent request. Signed-off-by: Pranjal Shrivastava Reviewed-by: Christoph Hellwig Signed-off-by: Anna Schumaker --- include/linux/nfs_page.h | 3 +++ 1 file changed, 3 insertions(+) (limited to 'include/linux') diff --git a/include/linux/nfs_page.h b/include/linux/nfs_page.h index 4b9a35dbc062..fd7aafe7cb54 100644 --- a/include/linux/nfs_page.h +++ b/include/linux/nfs_page.h @@ -38,6 +38,7 @@ enum { PG_REMOVE, /* page group sync bit in write path */ PG_CONTENDED1, /* Is someone waiting for a lock? */ PG_CONTENDED2, /* Is someone waiting for a lock? */ + PG_PINNED, /* page is pinned by GUP */ }; struct nfs_inode; @@ -125,11 +126,13 @@ struct nfs_pageio_descriptor { extern struct nfs_page *nfs_page_create_from_page(struct nfs_open_context *ctx, struct page *page, + bool pinned, unsigned int pgbase, loff_t offset, unsigned int count); extern struct nfs_page *nfs_page_create_from_folio(struct nfs_open_context *ctx, struct folio *folio, + bool pinned, unsigned int offset, unsigned int count); extern void nfs_release_request(struct nfs_page *); -- cgit v1.2.3 From 6b387ac65185e5824a9b8c47c2a2701422d452e7 Mon Sep 17 00:00:00 2001 From: Pranjal Shrivastava Date: Fri, 14 Aug 2026 14:32:52 +0000 Subject: nfs: track number of pinned pages in nfs_page Track the number of pinned pages in nfs_page to handle unpinning correctly, ensuring that only primary requests perform the final unpinning operation, preventing subrequests from incorrectly performing unpinning on behalf of their parent requests. Add wb_nr_pinned to struct nfs_page to store the count of pinned pages owned by the request. Update request creation and cleanup helpers to initialize and use wb_nr_pinned for primary requests. Use the nfs_page_array_len() helper to calculate the number of pages spanned by a request's offset and length. Signed-off-by: Pranjal Shrivastava Reviewed-by: Christoph Hellwig Signed-off-by: Anna Schumaker --- fs/nfs/pagelist.c | 13 +++++++++---- include/linux/nfs_page.h | 1 + 2 files changed, 10 insertions(+), 4 deletions(-) (limited to 'include/linux') diff --git a/fs/nfs/pagelist.c b/fs/nfs/pagelist.c index a562cfe2a126..b9ccf2a87e3c 100644 --- a/fs/nfs/pagelist.c +++ b/fs/nfs/pagelist.c @@ -457,6 +457,8 @@ struct nfs_page *nfs_page_create_from_page(struct nfs_open_context *ctx, offset_in_page(offset), count); if (!IS_ERR(ret)) { nfs_page_assign_page(ret, page, pinned); + if (pinned) + ret->wb_nr_pinned = 1; nfs_page_group_init(ret, NULL); } nfs_put_lock_context(l_ctx); @@ -489,6 +491,9 @@ struct nfs_page *nfs_page_create_from_folio(struct nfs_open_context *ctx, ret = nfs_page_create(l_ctx, offset, folio->index, offset, count); if (!IS_ERR(ret)) { nfs_page_assign_folio(ret, folio, pinned); + if (pinned) + ret->wb_nr_pinned = nfs_page_array_len(offset_in_page(offset), + count); nfs_page_group_init(ret, NULL); } nfs_put_lock_context(l_ctx); @@ -567,8 +572,8 @@ static void nfs_clear_request(struct nfs_page *req) if (folio != NULL) { if (test_and_clear_bit(PG_PINNED, &req->wb_flags)) { - if (req == req->wb_head) - unpin_user_folio(folio, 1); + if (req->wb_nr_pinned > 0) + unpin_user_folio(folio, req->wb_nr_pinned); } else { folio_put(folio); } @@ -576,8 +581,8 @@ static void nfs_clear_request(struct nfs_page *req) clear_bit(PG_FOLIO, &req->wb_flags); } else if (page != NULL) { if (test_and_clear_bit(PG_PINNED, &req->wb_flags)) { - if (req == req->wb_head) - unpin_user_page(page); + if (req->wb_nr_pinned > 0) + unpin_user_pages(&page, req->wb_nr_pinned); } else { put_page(page); } diff --git a/include/linux/nfs_page.h b/include/linux/nfs_page.h index fd7aafe7cb54..080fa3e23580 100644 --- a/include/linux/nfs_page.h +++ b/include/linux/nfs_page.h @@ -59,6 +59,7 @@ struct nfs_page { struct nfs_page *wb_this_page; /* list of reqs for this page */ struct nfs_page *wb_head; /* head pointer for req list */ unsigned short wb_nio; /* Number of I/O attempts */ + unsigned int wb_nr_pinned; /* Number of pinned pages */ }; struct nfs_pgio_mirror; -- cgit v1.2.3 From 77dc84f2a230154f37f542552e98a42a711965c5 Mon Sep 17 00:00:00 2001 From: Pranjal Shrivastava Date: Fri, 14 Aug 2026 14:32:53 +0000 Subject: nfs: introduce nfs_release_request_list helper Introduce a centralized helper, nfs_release_request_list, to handle the bulk release of nfs_page requests from a list. This serves as a preparatory step for two upcoming improvements: 1. Pin-Aware Cleanup: As we migrate to iov_iter_extract_* API, requests will hold pins (GUP) instead of standard references. The helper ensures that the correct unpinning logic gets applied consistently across all requests in a list. 2. Folio Support: In subsequent patches where nfs_page structures will cover multi-page folios, this helper provides a clean infrastructure to unlock these larger units of I/O in bulk during completion, similar to the pattern in bio_release_pages. Additionally, refactor nfs_read_sync_pgio_error() to utilize this new helper. Signed-off-by: Pranjal Shrivastava Reviewed-by: Shivaji Kant Reviewed-by: Christoph Hellwig Signed-off-by: Anna Schumaker --- fs/nfs/direct.c | 8 +------- fs/nfs/pagelist.c | 17 +++++++++++++++++ include/linux/nfs_page.h | 4 ++-- 3 files changed, 20 insertions(+), 9 deletions(-) (limited to 'include/linux') diff --git a/fs/nfs/direct.c b/fs/nfs/direct.c index 6a7f487fd43e..8b094f8ef087 100644 --- a/fs/nfs/direct.c +++ b/fs/nfs/direct.c @@ -294,13 +294,7 @@ out_put: static void nfs_read_sync_pgio_error(struct list_head *head, int error) { - struct nfs_page *req; - - while (!list_empty(head)) { - req = nfs_list_entry(head->next); - nfs_list_remove_request(req); - nfs_release_request(req); - } + nfs_release_request_list(head); } static void nfs_direct_pgio_init(struct nfs_pgio_header *hdr) diff --git a/fs/nfs/pagelist.c b/fs/nfs/pagelist.c index b9ccf2a87e3c..71f0ce2bc4ea 100644 --- a/fs/nfs/pagelist.c +++ b/fs/nfs/pagelist.c @@ -628,6 +628,23 @@ void nfs_release_request(struct nfs_page *req) } EXPORT_SYMBOL_GPL(nfs_release_request); +/* + * nfs_release_request_list - Release a list of NFS read/write requests + * @head: list of requests to release + * + * Removes each request from the list and drops it's refcount. + */ +void nfs_release_request_list(struct list_head *head) +{ + while (!list_empty(head)) { + struct nfs_page *req = nfs_list_entry(head->next); + + nfs_list_remove_request(req); + nfs_release_request(req); + } +} +EXPORT_SYMBOL_GPL(nfs_release_request_list); + /* * nfs_generic_pg_test - determine if requests can be coalesced * @desc: pointer to descriptor diff --git a/include/linux/nfs_page.h b/include/linux/nfs_page.h index 080fa3e23580..c38e4b380be5 100644 --- a/include/linux/nfs_page.h +++ b/include/linux/nfs_page.h @@ -136,8 +136,8 @@ extern struct nfs_page *nfs_page_create_from_folio(struct nfs_open_context *ctx, bool pinned, unsigned int offset, unsigned int count); -extern void nfs_release_request(struct nfs_page *); - +void nfs_release_request(struct nfs_page *req); +void nfs_release_request_list(struct list_head *head); extern void nfs_pageio_init(struct nfs_pageio_descriptor *desc, struct inode *inode, -- cgit v1.2.3 From ee4d94ca9f8e7ad1d4bbad4e0c9a70490f30a395 Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Fri, 4 Sep 2026 08:56:06 -0400 Subject: NFSv4.1/pnfs: remember when a server needs a larger LAYOUTGET reply buffer When a LAYOUTGET only succeeds after escalating the reply buffer, every layout fetched from that server is likely to need the larger buffer: remember the escalated size on the nfs_server and use it as the floor for subsequent LAYOUTGET reply buffers, skipping the doomed attempt at the layout driver's default size. This also lets the LAYOUTGET attached to OPEN benefit: the lgopen path is best-effort with no retry of its own, so without the learned size it would fail with NFS4ERR_TOOSMALL at every open against a server handing out wide layouts, and layouts would only ever be acquired by the I/O path's retry. The field is a hint: reads and writes are racy by design, the value only ever grows toward the session's maximum response size, and a stale-low read merely costs one escalation round trip. Assisted-by: Claude:claude-opus-4-8 Signed-off-by: Benjamin Coddington Reviewed-by: Jeff Layton Signed-off-by: Anna Schumaker --- fs/nfs/pnfs.c | 15 ++++++++++++--- include/linux/nfs_fs_sb.h | 4 ++++ 2 files changed, 16 insertions(+), 3 deletions(-) (limited to 'include/linux') diff --git a/fs/nfs/pnfs.c b/fs/nfs/pnfs.c index 0c938d17b6c6..3e8d1a1fd827 100644 --- a/fs/nfs/pnfs.c +++ b/fs/nfs/pnfs.c @@ -1188,10 +1188,12 @@ pnfs_alloc_init_layoutget_args(struct inode *ino, } /* - * A previous LAYOUTGET for this layout did not fit the reply - * buffer: raise the layout driver's default up to the session's - * maximum response size. + * A previous LAYOUTGET on this layout or on this server did not + * fit the reply buffer: raise the layout driver's default up to + * the session's maximum response size. */ + if (!min_reply_sz) + min_reply_sz = READ_ONCE(server->lg_reply_sz); if (min_reply_sz) { size_t npages = (min_reply_sz + PAGE_SIZE - 1) >> PAGE_SHIFT; if (npages > max_pages) @@ -2387,6 +2389,13 @@ lookup_again: goto lookup_again; } } else { + /* + * A LAYOUTGET that only succeeded with an escalated reply + * buffer: remember the size so that future LAYOUTGETs to + * this server skip the attempt at the driver's default. + */ + if (reply_sz) + WRITE_ONCE(server->lg_reply_sz, reply_sz); pnfs_layout_clear_fail_bit(lo, pnfs_iomode_to_fail_bit(iomode)); } diff --git a/include/linux/nfs_fs_sb.h b/include/linux/nfs_fs_sb.h index 34d294774f8c..3e6bae7e5221 100644 --- a/include/linux/nfs_fs_sb.h +++ b/include/linux/nfs_fs_sb.h @@ -248,6 +248,10 @@ struct nfs_server { that are supported on this filesystem */ struct pnfs_layoutdriver_type *pnfs_curr_ld; /* Active layout driver */ + unsigned int lg_reply_sz; /* Learned LAYOUTGET reply + buffer size, when the layout + driver's default has proved + too small */ struct rpc_wait_queue roc_rpcwaitq; /* the following fields are protected by nfs_client->cl_lock */ -- cgit v1.2.3 From 6fcb69e2fdee94b033b26ea0bc9d0217f4bfd5bb Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:11 -0400 Subject: NFSv4/flexfiles: Carry the device node reference across each I/O Rather than dropping the device node reference when DS setup completes, transfer it to the in-flight I/O: carry it on nfs_pgio_header and nfs_commit_data as ds_dev, and release it when the header or commit data is released, alongside the lseg reference. Convert the completion paths to use the carried node instead of re-reading the mirror's pinned pointer: DS error tracking, marking the deviceid available/unavailable, and deleting the deviceid on connection errors now act on the node the I/O was actually sent to. Once a CHANGE notification can re-point the pinned pointer mid-flight, this keeps error attribution on the old device rather than its replacement (and removes a NULL dereference had the pointer been reset to NULL). FF_LAYOUT_DEVID_NODE() is now unused; remove it. Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/flexfilelayout/flexfilelayout.c | 82 ++++++++++++++----------------- fs/nfs/flexfilelayout/flexfilelayout.h | 15 +----- fs/nfs/flexfilelayout/flexfilelayoutdev.c | 11 +++-- fs/nfs/pnfs.c | 2 + fs/nfs/pnfs.h | 8 +++ fs/nfs/pnfs_nfs.c | 1 + include/linux/nfs_xdr.h | 2 + 7 files changed, 58 insertions(+), 63 deletions(-) (limited to 'include/linux') diff --git a/fs/nfs/flexfilelayout/flexfilelayout.c b/fs/nfs/flexfilelayout/flexfilelayout.c index 7a908faa4131..928e691af9e7 100644 --- a/fs/nfs/flexfilelayout/flexfilelayout.c +++ b/fs/nfs/flexfilelayout/flexfilelayout.c @@ -857,24 +857,6 @@ nfs4_ff_layout_stat_io_end_write(struct rpc_task *task, spin_unlock(&mirror->lock); } -static void -ff_layout_mark_ds_unreachable(struct pnfs_layout_segment *lseg, u32 idx, u32 dss_id) -{ - struct nfs4_deviceid_node *devid = FF_LAYOUT_DEVID_NODE(lseg, idx, dss_id); - - if (devid) - nfs4_mark_deviceid_unavailable(devid); -} - -static void -ff_layout_mark_ds_reachable(struct pnfs_layout_segment *lseg, u32 idx, u32 dss_id) -{ - struct nfs4_deviceid_node *devid = FF_LAYOUT_DEVID_NODE(lseg, idx, dss_id); - - if (devid) - nfs4_mark_deviceid_available(devid); -} - static struct nfs4_ff_layout_ds * ff_layout_choose_ds_for_read(struct pnfs_layout_segment *lseg, u32 start_idx, u32 *best_idx, @@ -1333,11 +1315,10 @@ static int ff_layout_async_handle_error_v4(struct rpc_task *task, struct nfs4_state *state, struct nfs_client *clp, struct pnfs_layout_segment *lseg, - u32 idx, u32 dss_id) + struct nfs4_deviceid_node *devid) { struct pnfs_layout_hdr *lo = lseg->pls_layout; struct inode *inode = lo->plh_inode; - struct nfs4_deviceid_node *devid = FF_LAYOUT_DEVID_NODE(lseg, idx, dss_id); struct nfs4_slot_table *tbl = nfs4_has_session(clp) ? &clp->cl_session->fc_slot_table : clp->cl_slot_tbl; @@ -1410,8 +1391,9 @@ static int ff_layout_async_handle_error_v4(struct rpc_task *task, case -ENODEV: dprintk("%s DS connection error %d\n", __func__, task->tk_status); - nfs4_delete_deviceid(devid->ld, devid->nfs_client, - &devid->deviceid); + if (devid) + nfs4_delete_deviceid(devid->ld, devid->nfs_client, + &devid->deviceid); rpc_wake_up(&tbl->slot_tbl_waitq); break; default: @@ -1435,9 +1417,8 @@ static int ff_layout_async_handle_error_v3(struct rpc_task *task, u32 op_status, struct nfs_client *clp, struct pnfs_layout_segment *lseg, - u32 idx, u32 dss_id) + struct nfs4_deviceid_node *devid) { - struct nfs4_deviceid_node *devid = FF_LAYOUT_DEVID_NODE(lseg, idx, dss_id); switch (op_status) { case NFS_OK: @@ -1483,8 +1464,9 @@ static int ff_layout_async_handle_error_v3(struct rpc_task *task, default: dprintk("%s DS connection error %d\n", __func__, task->tk_status); - nfs4_delete_deviceid(devid->ld, devid->nfs_client, - &devid->deviceid); + if (devid) + nfs4_delete_deviceid(devid->ld, devid->nfs_client, + &devid->deviceid); } out_reset_to_pnfs: /* FIXME: Need to prevent infinite looping here. */ @@ -1501,12 +1483,13 @@ static int ff_layout_async_handle_error(struct rpc_task *task, struct nfs4_state *state, struct nfs_client *clp, struct pnfs_layout_segment *lseg, - u32 idx, u32 dss_id) + struct nfs4_deviceid_node *devid) { int vers = clp->cl_nfs_mod->rpc_vers->number; if (task->tk_status >= 0) { - ff_layout_mark_ds_reachable(lseg, idx, dss_id); + if (devid) + nfs4_mark_deviceid_available(devid); return 0; } @@ -1517,10 +1500,10 @@ static int ff_layout_async_handle_error(struct rpc_task *task, switch (vers) { case 3: return ff_layout_async_handle_error_v3(task, op_status, clp, - lseg, idx, dss_id); + lseg, devid); case 4: return ff_layout_async_handle_error_v4(task, op_status, state, - clp, lseg, idx, dss_id); + clp, lseg, devid); default: /* should never happen */ WARN_ON_ONCE(1); @@ -1529,6 +1512,7 @@ static int ff_layout_async_handle_error(struct rpc_task *task, } static void ff_layout_io_track_ds_error(struct pnfs_layout_segment *lseg, + struct nfs4_deviceid_node *devid, u32 idx, u32 dss_id, u64 offset, u64 length, u32 *op_status, int opnum, int error) { @@ -1578,8 +1562,8 @@ static void ff_layout_io_track_ds_error(struct pnfs_layout_segment *lseg, mirror = FF_LAYOUT_COMP(lseg, idx); err = ff_layout_track_ds_error(FF_LAYOUT_FROM_HDR(lseg->pls_layout), - mirror, dss_id, offset, length, status, opnum, - nfs_io_gfp_mask()); + mirror, devid, dss_id, offset, length, + status, opnum, nfs_io_gfp_mask()); /* * I/O we cancelled ourselves to return a recalled or revoked layout @@ -1596,7 +1580,8 @@ static void ff_layout_io_track_ds_error(struct pnfs_layout_segment *lseg, case NFS4ERR_PERM: break; case NFS4ERR_NXIO: - ff_layout_mark_ds_unreachable(lseg, idx, dss_id); + if (devid) + nfs4_mark_deviceid_unavailable(devid); /* * Don't return the layout if this is a read and we still * have layouts to try @@ -1625,7 +1610,7 @@ static int ff_layout_read_done_cb(struct rpc_task *task, int err; if (task->tk_status < 0) { - ff_layout_io_track_ds_error(hdr->lseg, + ff_layout_io_track_ds_error(hdr->lseg, hdr->ds_dev, hdr->pgio_mirror_idx, dss_id, hdr->args.offset, hdr->args.count, &hdr->res.op_status, OP_READ, @@ -1636,8 +1621,7 @@ static int ff_layout_read_done_cb(struct rpc_task *task, err = ff_layout_async_handle_error(task, hdr->res.op_status, hdr->args.context->state, hdr->ds_clp, hdr->lseg, - hdr->pgio_mirror_idx, - dss_id); + hdr->ds_dev); trace_nfs4_pnfs_read(hdr, err); clear_bit(NFS_IOHDR_RESEND_PNFS, &hdr->flags); @@ -1830,7 +1814,7 @@ static int ff_layout_write_done_cb(struct rpc_task *task, int err; if (task->tk_status < 0) { - ff_layout_io_track_ds_error(hdr->lseg, + ff_layout_io_track_ds_error(hdr->lseg, hdr->ds_dev, hdr->pgio_mirror_idx, dss_id, hdr->args.offset, hdr->args.count, &hdr->res.op_status, OP_WRITE, @@ -1841,8 +1825,7 @@ static int ff_layout_write_done_cb(struct rpc_task *task, err = ff_layout_async_handle_error(task, hdr->res.op_status, hdr->args.context->state, hdr->ds_clp, hdr->lseg, - hdr->pgio_mirror_idx, - dss_id); + hdr->ds_dev); trace_nfs4_pnfs_write(hdr, err); clear_bit(NFS_IOHDR_RESEND_PNFS, &hdr->flags); @@ -1884,7 +1867,7 @@ static int ff_layout_commit_done_cb(struct rpc_task *task, u32 dss_id = calc_dss_id_from_commit(data->lseg, data->ds_commit_index); if (task->tk_status < 0) { - ff_layout_io_track_ds_error(data->lseg, idx, dss_id, + ff_layout_io_track_ds_error(data->lseg, data->ds_dev, idx, dss_id, data->args.offset, data->args.count, &data->res.op_status, OP_COMMIT, task->tk_status); @@ -1892,8 +1875,8 @@ static int ff_layout_commit_done_cb(struct rpc_task *task, } err = ff_layout_async_handle_error(task, data->res.op_status, - NULL, data->ds_clp, data->lseg, idx, - dss_id); + NULL, data->ds_clp, data->lseg, + data->ds_dev); trace_nfs4_pnfs_commit_ds(data, err); switch (err) { @@ -2248,13 +2231,16 @@ ff_layout_read_pagelist(struct nfs_pgio_header *hdr) ff_layout_read_record_layoutstats_start(&hdr->task, hdr); } + /* Transfer the device node reference to the I/O; put on release */ + pnfs_put_ds_dev(hdr->ds_dev); + hdr->ds_dev = &mirror_ds->id_node; + /* Perform an asynchronous read to ds */ nfs_initiate_pgio(ds_clnt, hdr, ds_cred, ds->ds_clp->rpc_ops, vers == 3 ? &ff_layout_read_call_ops_v3 : &ff_layout_read_call_ops_v4, 0, RPC_TASK_SOFTCONN, localio); put_cred(ds_cred); - nfs4_ff_layout_put_deviceid(mirror_ds); return PNFS_ATTEMPTED; out_failed: @@ -2347,13 +2333,16 @@ ff_layout_write_pagelist(struct nfs_pgio_header *hdr, int sync) ff_layout_write_record_layoutstats_start(&hdr->task, hdr); } + /* Transfer the device node reference to the I/O; put on release */ + pnfs_put_ds_dev(hdr->ds_dev); + hdr->ds_dev = &mirror_ds->id_node; + /* Perform an asynchronous write */ nfs_initiate_pgio(ds_clnt, hdr, ds_cred, ds->ds_clp->rpc_ops, vers == 3 ? &ff_layout_write_call_ops_v3 : &ff_layout_write_call_ops_v4, sync, RPC_TASK_SOFTCONN, localio); put_cred(ds_cred); - nfs4_ff_layout_put_deviceid(mirror_ds); return PNFS_ATTEMPTED; out_failed: @@ -2441,12 +2430,15 @@ static int ff_layout_initiate_commit(struct nfs_commit_data *data, int how) ff_layout_commit_record_layoutstats_start(&data->task, data); } + /* Transfer the device node reference to the commit; put on release */ + pnfs_put_ds_dev(data->ds_dev); + data->ds_dev = &mirror_ds->id_node; + ret = nfs_initiate_commit(ds_clnt, data, ds->ds_clp->rpc_ops, vers == 3 ? &ff_layout_commit_call_ops_v3 : &ff_layout_commit_call_ops_v4, how, RPC_TASK_SOFTCONN, localio); put_cred(ds_cred); - nfs4_ff_layout_put_deviceid(mirror_ds); return ret; out_err: nfs4_ff_layout_put_deviceid(mirror_ds); diff --git a/fs/nfs/flexfilelayout/flexfilelayout.h b/fs/nfs/flexfilelayout/flexfilelayout.h index f9e491a0347e..7eb47ce09442 100644 --- a/fs/nfs/flexfilelayout/flexfilelayout.h +++ b/fs/nfs/flexfilelayout/flexfilelayout.h @@ -162,20 +162,6 @@ FF_LAYOUT_COMP(struct pnfs_layout_segment *lseg, u32 idx) return NULL; } -static inline struct nfs4_deviceid_node * -FF_LAYOUT_DEVID_NODE(struct pnfs_layout_segment *lseg, u32 idx, u32 dss_id) -{ - struct nfs4_ff_layout_mirror *mirror = FF_LAYOUT_COMP(lseg, idx); - - if (mirror != NULL) { - struct nfs4_ff_layout_ds *mirror_ds = mirror->dss[dss_id].mirror_ds; - - if (!IS_ERR_OR_NULL(mirror_ds)) - return &mirror_ds->id_node; - } - return NULL; -} - static inline u32 FF_LAYOUT_MIRROR_COUNT(struct pnfs_layout_segment *lseg) { @@ -232,6 +218,7 @@ void nfs4_ff_layout_put_deviceid(struct nfs4_ff_layout_ds *mirror_ds); void nfs4_ff_layout_free_deviceid(struct nfs4_ff_layout_ds *mirror_ds); int ff_layout_track_ds_error(struct nfs4_flexfile_layout *flo, struct nfs4_ff_layout_mirror *mirror, + const struct nfs4_deviceid_node *devid, u32 dss_id, u64 offset, u64 length, int status, enum nfs_opnum4 opnum, gfp_t gfp_flags); void ff_layout_send_layouterror(struct pnfs_layout_segment *lseg); diff --git a/fs/nfs/flexfilelayout/flexfilelayoutdev.c b/fs/nfs/flexfilelayout/flexfilelayoutdev.c index 3712a47b20d0..ebcfba447879 100644 --- a/fs/nfs/flexfilelayout/flexfilelayoutdev.c +++ b/fs/nfs/flexfilelayout/flexfilelayoutdev.c @@ -243,6 +243,7 @@ ff_layout_add_ds_error_locked(struct nfs4_flexfile_layout *flo, int ff_layout_track_ds_error(struct nfs4_flexfile_layout *flo, struct nfs4_ff_layout_mirror *mirror, + const struct nfs4_deviceid_node *devid, u32 dss_id, u64 offset, u64 length, int status, enum nfs_opnum4 opnum, gfp_t gfp_flags) { @@ -251,7 +252,7 @@ int ff_layout_track_ds_error(struct nfs4_flexfile_layout *flo, if (status == 0) return 0; - if (IS_ERR_OR_NULL(mirror->dss[dss_id].mirror_ds)) + if (devid == NULL) return -EINVAL; dserr = kmalloc_obj(*dserr, gfp_flags); @@ -264,8 +265,7 @@ int ff_layout_track_ds_error(struct nfs4_flexfile_layout *flo, dserr->status = status; dserr->opnum = opnum; nfs4_stateid_copy(&dserr->stateid, &mirror->dss[dss_id].stateid); - memcpy(&dserr->deviceid, &mirror->dss[dss_id].mirror_ds->id_node.deviceid, - NFS4_DEVICEID4_SIZE); + memcpy(&dserr->deviceid, &devid->deviceid, NFS4_DEVICEID4_SIZE); spin_lock(&flo->generic_hdr.plh_inode->i_lock); ff_layout_add_ds_error_locked(flo, dserr); @@ -422,7 +422,10 @@ nfs4_ff_layout_prepare_ds(struct pnfs_layout_segment *lseg, } noconnect: ff_layout_track_ds_error(FF_LAYOUT_FROM_HDR(lseg->pls_layout), - mirror, dss_id, lseg->pls_range.offset, + mirror, + IS_ERR_OR_NULL(mirror_ds) ? + NULL : &mirror_ds->id_node, + dss_id, lseg->pls_range.offset, lseg->pls_range.length, NFS4ERR_NXIO, OP_ILLEGAL, GFP_NOIO); ff_layout_send_layouterror(lseg); diff --git a/fs/nfs/pnfs.c b/fs/nfs/pnfs.c index 3e8d1a1fd827..a74ffdda5bcd 100644 --- a/fs/nfs/pnfs.c +++ b/fs/nfs/pnfs.c @@ -3145,6 +3145,7 @@ pnfs_do_write(struct nfs_pageio_descriptor *desc, static void pnfs_writehdr_free(struct nfs_pgio_header *hdr) { + pnfs_put_ds_dev(hdr->ds_dev); pnfs_put_lseg(hdr->lseg); nfs_pgio_header_free(hdr); } @@ -3290,6 +3291,7 @@ pnfs_do_read(struct nfs_pageio_descriptor *desc, struct nfs_pgio_header *hdr) static void pnfs_readhdr_free(struct nfs_pgio_header *hdr) { + pnfs_put_ds_dev(hdr->ds_dev); pnfs_put_lseg(hdr->lseg); nfs_pgio_header_free(hdr); } diff --git a/fs/nfs/pnfs.h b/fs/nfs/pnfs.h index 5725ea06efc2..a2c70573dda5 100644 --- a/fs/nfs/pnfs.h +++ b/fs/nfs/pnfs.h @@ -385,6 +385,14 @@ void nfs4_init_deviceid_node(struct nfs4_deviceid_node *, struct nfs_server *, const struct nfs4_deviceid *); bool nfs4_put_deviceid_node(struct nfs4_deviceid_node *); void nfs4_mark_deviceid_available(struct nfs4_deviceid_node *node); + +/* Put the device node reference carried by an in-flight I/O, if any */ +static inline void pnfs_put_ds_dev(struct nfs4_deviceid_node *dev) +{ + if (dev) + nfs4_put_deviceid_node(dev); +} + void nfs4_mark_deviceid_unavailable(struct nfs4_deviceid_node *node); bool nfs4_test_deviceid_unavailable(struct nfs4_deviceid_node *node); void nfs4_deviceid_purge_client(const struct nfs_client *); diff --git a/fs/nfs/pnfs_nfs.c b/fs/nfs/pnfs_nfs.c index 5bd014b981bf..e0e3fc7414e6 100644 --- a/fs/nfs/pnfs_nfs.c +++ b/fs/nfs/pnfs_nfs.c @@ -55,6 +55,7 @@ void pnfs_generic_commit_release(void *calldata) struct nfs_commit_data *data = calldata; data->completion_ops->completion(data); + pnfs_put_ds_dev(data->ds_dev); pnfs_put_lseg(data->lseg); nfs_put_client(data->ds_clp); nfs_commitdata_release(data); diff --git a/include/linux/nfs_xdr.h b/include/linux/nfs_xdr.h index 7ed8fdb930d6..c0e29b4dfa62 100644 --- a/include/linux/nfs_xdr.h +++ b/include/linux/nfs_xdr.h @@ -1693,6 +1693,7 @@ struct nfs_pgio_header { struct nfs_client *ds_clp; /* pNFS data server */ u32 ds_commit_idx; /* ds index if ds_clp is set */ u32 pgio_mirror_idx;/* mirror index in pgio layer */ + struct nfs4_deviceid_node *ds_dev; /* device node ref held across the I/O */ }; struct nfs_mds_commit_info { @@ -1731,6 +1732,7 @@ struct nfs_commit_data { struct nfs_open_context *context; struct pnfs_layout_segment *lseg; struct nfs_client *ds_clp; /* pNFS data server */ + struct nfs4_deviceid_node *ds_dev; /* device node ref held across the commit */ int ds_commit_index; loff_t lwb; const struct rpc_call_ops *mds_ops; -- cgit v1.2.3 From b7af4bf5690761fcb3eacd92d99bee798f81386d Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:18 -0400 Subject: pNFS: Discard a GETDEVICEINFO reply that raced a CHANGE notification RFC 8881 Section 18.40.4: a GETDEVICEINFO reply in flight while the server changes the device mapping may carry the pre-change mapping; if it is inserted into the cache after the CHANGE notification unhashed the stale entry, the client re-caches stale data. Track a change epoch, bumped when a CHANGE notification is processed before the stale entry is unhashed. nfs4_find_get_deviceid() snapshots the epoch before issuing GETDEVICEINFO and, serialized against the unhash by nfs4_deviceid_lock at insert time, discards the reply and refetches if the epoch moved. A stale insert that instead precedes the unhash is removed by the unhash itself, so the cache does not retain the pre-change entry either way; a reference already handed to a caller in that ordering is dropped by the re-resolve walk instead. The refetch is bounded. The epoch is bumped once per CHANGE entry -- that is, at a rate the server chooses -- so an unbounded retry would let a server drive GETDEVICEINFO traffic without limit, and each discarded node can carry a DS client teardown and reconnect with it. After NFS4_DEVICEID_FETCH_RETRIES attempts the reply is accepted. That is safe because discarding is an optimisation rather than a correctness requirement: it avoids caching a mapping already known to be superseded, but before this patch the client cached whatever the reply carried, so the bounded case is no worse than the previous behaviour and a mapping that really is stale is corrected by the notification that follows. The epoch lives on the nfs_client, so a CHANGE delivered on one server's callback channel does not force an unrelated server's in-flight lookup to discard its reply and refetch. Mounts that share an nfs_client do share the counter; the deviceid cache is keyed per client ID, so that is the granularity the race is defined at. Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/callback_proc.c | 4 ++++ fs/nfs/pnfs.h | 1 + fs/nfs/pnfs_dev.c | 25 +++++++++++++++++++++++++ include/linux/nfs_fs_sb.h | 2 ++ 4 files changed, 32 insertions(+) (limited to 'include/linux') diff --git a/fs/nfs/callback_proc.c b/fs/nfs/callback_proc.c index 01372d8548e1..ea8c558b07b6 100644 --- a/fs/nfs/callback_proc.c +++ b/fs/nfs/callback_proc.c @@ -395,7 +395,11 @@ __be32 nfs4_callback_devicenotify(void *argp, void *resp, * Unhash the cached device first so re-resolution cannot * re-pin the stale node, then re-point any references * pinned under live layouts (RFC 8881 Section 12.2.10). + * The epoch bump lets an in-flight GETDEVICEINFO detect + * that its reply may predate the change. */ + if (dev->cbd_notify_type == NOTIFY_DEVICEID4_CHANGE) + nfs4_deviceid_bump_change_epoch(cps->clp); nfs4_delete_deviceid(ld, cps->clp, &dev->cbd_dev_id); if (dev->cbd_notify_type == NOTIFY_DEVICEID4_CHANGE) pnfs_layout_reresolve_deviceid_byclid(cps->clp, ld, diff --git a/fs/nfs/pnfs.h b/fs/nfs/pnfs.h index bb1ca7ba0221..9a5b8070f595 100644 --- a/fs/nfs/pnfs.h +++ b/fs/nfs/pnfs.h @@ -406,6 +406,7 @@ nfs4_find_get_deviceid(struct nfs_server *server, const struct nfs4_deviceid *id, const struct cred *cred, gfp_t gfp_mask); void nfs4_delete_deviceid(const struct pnfs_layoutdriver_type *, const struct nfs_client *, const struct nfs4_deviceid *); +void nfs4_deviceid_bump_change_epoch(struct nfs_client *clp); void nfs4_init_deviceid_node(struct nfs4_deviceid_node *, struct nfs_server *, const struct nfs4_deviceid *); bool nfs4_put_deviceid_node(struct nfs4_deviceid_node *); diff --git a/fs/nfs/pnfs_dev.c b/fs/nfs/pnfs_dev.c index 274abdd6d5f3..a3b28409539a 100644 --- a/fs/nfs/pnfs_dev.c +++ b/fs/nfs/pnfs_dev.c @@ -181,6 +181,21 @@ __nfs4_find_get_deviceid(struct nfs_server *server, return d; } +/* + * Bumped before the stale entry is unhashed, so an insert serialised + * after the unhash by nfs4_deviceid_lock observes the new epoch. + */ +void +nfs4_deviceid_bump_change_epoch(struct nfs_client *clp) +{ + atomic_inc(&clp->cl_deviceid_change_epoch); +} + +/* Discarding a raced reply is an optimisation, not a correctness + * requirement, and the epoch moves at the server's rate: bound it. + */ +#define NFS4_DEVICEID_FETCH_RETRIES 3 + struct nfs4_deviceid_node * nfs4_find_get_deviceid(struct nfs_server *server, const struct nfs4_deviceid *id, const struct cred *cred, @@ -188,11 +203,14 @@ nfs4_find_get_deviceid(struct nfs_server *server, { long hash = nfs4_deviceid_hash(id); struct nfs4_deviceid_node *d, *new; + int epoch, tries = 0; +retry: d = __nfs4_find_get_deviceid(server, id, hash); if (d) goto found; + epoch = atomic_read(&server->nfs_client->cl_deviceid_change_epoch); new = nfs4_get_device_info(server, id, cred, gfp_mask); if (!new) { trace_nfs4_find_deviceid(server, id, -ENOENT); @@ -200,6 +218,13 @@ nfs4_find_get_deviceid(struct nfs_server *server, } spin_lock(&nfs4_deviceid_lock); + if (atomic_read(&server->nfs_client->cl_deviceid_change_epoch) != epoch && + ++tries <= NFS4_DEVICEID_FETCH_RETRIES) { + /* a mapping changed while we fetched; ours may be stale */ + spin_unlock(&nfs4_deviceid_lock); + server->pnfs_curr_ld->free_deviceid_node(new); + goto retry; + } d = __nfs4_find_get_deviceid(server, id, hash); if (d) { spin_unlock(&nfs4_deviceid_lock); diff --git a/include/linux/nfs_fs_sb.h b/include/linux/nfs_fs_sb.h index 3e6bae7e5221..482daf1fa4b0 100644 --- a/include/linux/nfs_fs_sb.h +++ b/include/linux/nfs_fs_sb.h @@ -74,6 +74,8 @@ struct nfs_client { u64 cl_clientid; /* constant */ nfs4_verifier cl_confirm; /* Clientid verifier */ unsigned long cl_state; + /* bumped on each CB_NOTIFY_DEVICEID CHANGE for this client */ + atomic_t cl_deviceid_change_epoch; spinlock_t cl_lock; -- cgit v1.2.3 From 4fd02d0776169a4ccdc6d49ce2d2a37eb2a1dc6b Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:20 -0400 Subject: NFSv4/pnfs: Recover revoked layouts on a deleted deviceID RFC 8881 Section 20.12 lets a server send CB_NOTIFY_DEVICEID DELETE for a deviceID once it has revoked every layout referring to it. Revocation is not announced, so the client can still be holding what it believes are live layouts on that deviceID. Section 18.40.4 resolves that: TEST_STATEID each referring layout and recover the ones that come back revoked -- mark the layout stateid invalid, free the lsegs, FREE_STATEID to acknowledge. The callback thread cannot issue fore-channel RPCs, so suspects are queued on the nfs_client (dedup'd, holding a layoutdriver reference) and resolved by a new state-manager step keyed on NFS4CLNT_DEVICEID_DELETE. The worker re-collects the referring layouts, so layouts returned or recalled in the meantime are skipped. Drop the cached device once the collected layouts account for the delete: every one of them was revoked here. Section 18.48.3 defines TEST_STATEID's answers, and NFS4ERR_OLD_STATEID says the layout exists and was not revoked -- only that it moved on after this stateid was snapshotted -- so it counts against the delete as NFS4_OK does. Any other answer leaves the revocation unresolved and keeps the device cached, as does a layout the server still considers valid (verifying that one with GETDEVICEINFO comes next). A layout counts as revoked only if it was invalidated here; a stateid that no longer matches its layout is a stale snapshot. Invalidating one is paired with nfs_commit_inode(), since pnfs_clear_lseg_state() drops only the VALID and LAYOUTCOMMIT references, and an lseg still held by a commit bucket would keep the layout -- and the device nodes this recovery is trying to release -- alive. If the walk collects no referring layouts, the device is unreferenced and the delete is carried out directly. If the collection could not be completed, recovery leaves the device cached for the next notification. Nothing enqueues suspects yet, so no behavior change. Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/nfs4_fs.h | 2 ++ fs/nfs/nfs4client.c | 2 ++ fs/nfs/nfs4proc.c | 83 +++++++++++++++++++++++++++++++++++++++++++++++ fs/nfs/nfs4state.c | 3 ++ fs/nfs/pnfs.c | 62 +++++++++++++++++++++++++++++++++++ fs/nfs/pnfs.h | 19 +++++++++++ include/linux/nfs_fs_sb.h | 2 ++ 7 files changed, 173 insertions(+) (limited to 'include/linux') diff --git a/fs/nfs/nfs4_fs.h b/fs/nfs/nfs4_fs.h index b48e5b87cb2a..d642aca0adc3 100644 --- a/fs/nfs/nfs4_fs.h +++ b/fs/nfs/nfs4_fs.h @@ -52,6 +52,7 @@ enum nfs4_client_state { NFS4CLNT_RECALL_ANY_LAYOUT_READ, NFS4CLNT_RECALL_ANY_LAYOUT_RW, NFS4CLNT_DELEGRETURN_DELAYED, + NFS4CLNT_DEVICEID_DELETE, }; #define NFS4_RENEW_TIMEOUT 0x01 @@ -493,6 +494,7 @@ int nfs41_discover_server_trunking(struct nfs_client *clp, struct nfs_client **, const struct cred *); extern void nfs4_schedule_session_recovery(struct nfs4_session *, int); extern void nfs41_notify_server(struct nfs_client *); +extern void nfs4_deviceid_delete_recover_run(struct nfs_client *clp); bool nfs4_check_serverowner_major_id(struct nfs41_server_owner *o1, struct nfs41_server_owner *o2); diff --git a/fs/nfs/nfs4client.c b/fs/nfs/nfs4client.c index b661f446ea49..e6a589913666 100644 --- a/fs/nfs/nfs4client.c +++ b/fs/nfs/nfs4client.c @@ -217,6 +217,7 @@ struct nfs_client *nfs4_alloc_client(const struct nfs_client_initdata *cl_init) clp->cl_last_renewal = jiffies; init_waitqueue_head(&clp->cl_lock_waitq); INIT_LIST_HEAD(&clp->pending_cb_stateids); + INIT_LIST_HEAD(&clp->cl_deviceid_deletes); if (cl_init->minorversion != 0) __set_bit(NFS_CS_INFINITE_SLOTS, &clp->cl_flags); @@ -286,6 +287,7 @@ static void nfs4_shutdown_client(struct nfs_client *clp) nfs4_kill_renewd(clp); clp->cl_mvops->shutdown_client(clp); nfs4_destroy_callback(clp); + pnfs_deviceid_delete_queue_free(clp); if (__test_and_clear_bit(NFS_CS_IDMAP, &clp->cl_res_state)) nfs_idmap_delete(clp); diff --git a/fs/nfs/nfs4proc.c b/fs/nfs/nfs4proc.c index e41c792a2725..2192875c168e 100644 --- a/fs/nfs/nfs4proc.c +++ b/fs/nfs/nfs4proc.c @@ -10506,6 +10506,89 @@ out_put_clp: return ret; } +/* + * A DELETE for a deviceID we still hold layouts on implies the server + * revoked them: run the RFC 8881 Section 18.40.4 recovery. + */ +static void nfs4_deviceid_delete_recover(struct nfs_client *clp, + const struct pnfs_layoutdriver_type *ld, + const struct nfs4_deviceid *id) +{ + LIST_HEAD(layouts); + struct nfs4_deviceid_ref *ref; + bool revoked = false; + bool referenced = false; + bool inconclusive = false; + + if (pnfs_layout_collect_deviceid_refs(clp, ld, id, &layouts)) { + /* Only a partial list -- an allocation failed, or an inode is + * being evicted. Leave the device cached and recover on a + * later notification. + */ + pnfs_layout_put_deviceid_refs(&layouts); + return; + } + + if (list_empty(&layouts)) { + nfs4_delete_deviceid(ld, clp, id); + return; + } + + list_for_each_entry(ref, &layouts, node) { + struct pnfs_layout_hdr *lo = ref->lo; + struct inode *inode = ref->inode; + bool invalidated = false; + LIST_HEAD(head); + int status; + + status = nfs41_test_stateid(NFS_SERVER(inode), &ref->stateid, + ref->cred); + switch (status) { + case NFS_OK: + case -NFS4ERR_OLD_STATEID: + referenced = true; + break; + case -NFS4ERR_ADMIN_REVOKED: + case -NFS4ERR_DELEG_REVOKED: + case -NFS4ERR_EXPIRED: + case -NFS4ERR_BAD_STATEID: + spin_lock(&inode->i_lock); + if (pnfs_layout_is_valid(lo) && + nfs4_stateid_match_other(&ref->stateid, + &lo->plh_stateid)) { + pnfs_mark_layout_stateid_invalid(lo, &head); + revoked = true; + invalidated = true; + } + spin_unlock(&inode->i_lock); + pnfs_free_lseg_list(&head); + if (invalidated) + nfs_commit_inode(inode, 0); + nfs41_free_stateid(NFS_SERVER(inode), &ref->stateid, + ref->cred, true); + break; + default: + inconclusive = true; + break; + } + } + pnfs_layout_put_deviceid_refs(&layouts); + + if (revoked && !referenced && !inconclusive) + nfs4_delete_deviceid(ld, clp, id); +} + +void nfs4_deviceid_delete_recover_run(struct nfs_client *clp) +{ + struct nfs4_deviceid_delete *dd; + + while ((dd = pnfs_deviceid_delete_dequeue(clp)) != NULL) { + nfs4_deviceid_delete_recover(clp, dd->ld, &dd->id); + pnfs_put_layoutdriver(dd->ld); + kfree(dd); + } +} + static void nfs41_free_lock_state(struct nfs_server *server, struct nfs4_lock_state *lsp) { diff --git a/fs/nfs/nfs4state.c b/fs/nfs/nfs4state.c index a5dec0473e22..1faf9dafd331 100644 --- a/fs/nfs/nfs4state.c +++ b/fs/nfs/nfs4state.c @@ -2669,6 +2669,9 @@ static void nfs4_state_manager(struct nfs_client *clp) set_bit(NFS4CLNT_RUN_MANAGER, &clp->cl_state); } nfs4_layoutreturn_any_run(clp); + if (test_and_clear_bit(NFS4CLNT_DEVICEID_DELETE, + &clp->cl_state)) + nfs4_deviceid_delete_recover_run(clp); clear_bit(NFS4CLNT_RECALL_RUNNING, &clp->cl_state); } diff --git a/fs/nfs/pnfs.c b/fs/nfs/pnfs.c index fbe4b58b5b72..79f671355e5b 100644 --- a/fs/nfs/pnfs.c +++ b/fs/nfs/pnfs.c @@ -3143,6 +3143,68 @@ pnfs_layout_put_deviceid_refs(struct list_head *result) } } +/* + * Queue @id for the state manager's Section 18.40.4 recovery, + * dropping duplicates of an already-queued suspect. + */ +void pnfs_deviceid_delete_mark(struct nfs_client *clp, + const struct pnfs_layoutdriver_type *ld, + const struct nfs4_deviceid *id) +{ + struct nfs4_deviceid_delete *dd, *new; + + new = kzalloc_obj(*new, GFP_KERNEL); + if (!new) + return; /* lost notification; recovery waits for the next */ + new->ld = pnfs_find_layoutdriver(ld->id); + if (!new->ld) { + kfree(new); + return; + } + memcpy(&new->id, id, sizeof(new->id)); + + spin_lock(&clp->cl_lock); + list_for_each_entry(dd, &clp->cl_deviceid_deletes, list) { + if (dd->ld == new->ld && + !memcmp(&dd->id, &new->id, sizeof(dd->id))) { + spin_unlock(&clp->cl_lock); + pnfs_put_layoutdriver(new->ld); + kfree(new); + return; + } + } + list_add_tail(&new->list, &clp->cl_deviceid_deletes); + spin_unlock(&clp->cl_lock); + + set_bit(NFS4CLNT_DEVICEID_DELETE, &clp->cl_state); + nfs4_schedule_state_manager(clp); +} + +struct nfs4_deviceid_delete *pnfs_deviceid_delete_dequeue( + struct nfs_client *clp) +{ + struct nfs4_deviceid_delete *dd = NULL; + + spin_lock(&clp->cl_lock); + if (!list_empty(&clp->cl_deviceid_deletes)) { + dd = list_first_entry(&clp->cl_deviceid_deletes, + struct nfs4_deviceid_delete, list); + list_del(&dd->list); + } + spin_unlock(&clp->cl_lock); + return dd; +} + +void pnfs_deviceid_delete_queue_free(struct nfs_client *clp) +{ + struct nfs4_deviceid_delete *dd; + + while ((dd = pnfs_deviceid_delete_dequeue(clp)) != NULL) { + pnfs_put_layoutdriver(dd->ld); + kfree(dd); + } +} + /* Check if we have we have a valid layout but if there isn't an intersection * between the request and the pgio->pg_lseg, put this pgio->pg_lseg away. */ diff --git a/fs/nfs/pnfs.h b/fs/nfs/pnfs.h index 8dd892d875e3..18ea4e8e0d85 100644 --- a/fs/nfs/pnfs.h +++ b/fs/nfs/pnfs.h @@ -400,6 +400,25 @@ int pnfs_layout_collect_deviceid_refs(struct nfs_client *clp, const struct nfs4_deviceid *devid, struct list_head *result); void pnfs_layout_put_deviceid_refs(struct list_head *result); + +/* + * A CB_NOTIFY_DEVICEID DELETE naming a deviceID that live layouts + * still reference (RFC 8881 Section 18.40.4). Queued on + * nfs_client.cl_deviceid_deletes under cl_lock for the state manager + * to resolve; holds a layoutdriver reference. + */ +struct nfs4_deviceid_delete { + struct list_head list; + const struct pnfs_layoutdriver_type *ld; + struct nfs4_deviceid id; +}; + +void pnfs_deviceid_delete_mark(struct nfs_client *clp, + const struct pnfs_layoutdriver_type *ld, + const struct nfs4_deviceid *id); +struct nfs4_deviceid_delete *pnfs_deviceid_delete_dequeue( + struct nfs_client *clp); +void pnfs_deviceid_delete_queue_free(struct nfs_client *clp); int pnfs_layout_handle_reboot(struct nfs_client *clp); /* nfs4_deviceid_flags */ diff --git a/include/linux/nfs_fs_sb.h b/include/linux/nfs_fs_sb.h index 482daf1fa4b0..416c6f39f31d 100644 --- a/include/linux/nfs_fs_sb.h +++ b/include/linux/nfs_fs_sb.h @@ -103,6 +103,8 @@ struct nfs_client { /* The flags used for obtaining the clientid during EXCHANGE_ID */ u32 cl_exchange_flags; struct nfs4_session *cl_session; /* shared session */ + /* CB_NOTIFY_DEVICEID DELETE suspects, protected by cl_lock */ + struct list_head cl_deviceid_deletes; bool cl_preserve_clid; struct nfs41_server_owner *cl_serverowner; struct nfs41_server_scope *cl_serverscope; -- cgit v1.2.3