diff options
| author | Mark Brown <broonie@kernel.org> | 2026-09-30 12:48:37 +0100 |
|---|---|---|
| committer | Mark Brown <broonie@kernel.org> | 2026-09-30 12:48:37 +0100 |
| commit | c0c20ac78811bbb321b9f1b1b7ee3c00c847e95d (patch) | |
| tree | 66376e23a2e88beaa9ffcfb9420f6ed8885578a4 /include/linux | |
| parent | 19cb490f7b0c1e5808a81ac385dff32045aa98aa (diff) | |
| parent | bf234c28d9e24e3d6c42a6202e3f92fba514c2ae (diff) | |
| download | linux-next-c0c20ac78811bbb321b9f1b1b7ee3c00c847e95d.tar.gz linux-next-c0c20ac78811bbb321b9f1b1b7ee3c00c847e95d.zip | |
Merge branch 'fs-next' of linux-next
# Conflicts:
# fs/coredump.c
# fs/f2fs/f2fs.h
# fs/fuse/dax.c
# fs/xfs/libxfs/xfs_btree.c
Diffstat (limited to 'include/linux')
47 files changed, 902 insertions, 592 deletions
diff --git a/include/linux/binfmts.h b/include/linux/binfmts.h index f686a37f7a0a..2e87faf9a8c2 100644 --- a/include/linux/binfmts.h +++ b/include/linux/binfmts.h @@ -128,7 +128,8 @@ struct linux_binfmt { struct module *module; int (*load_binary)(struct linux_binprm *); #ifdef CONFIG_COREDUMP - int (*core_dump)(struct coredump_params *cprm); + /* Returns true if the whole coredump was written. */ + bool (*core_dump)(struct coredump_params *cprm); unsigned long min_coredump; /* minimal dump size */ #endif } __randomize_layout; diff --git a/include/linux/bio-integrity.h b/include/linux/bio-integrity.h index 0ea2a8bf7efb..a954c97be0b3 100644 --- a/include/linux/bio-integrity.h +++ b/include/linux/bio-integrity.h @@ -151,7 +151,6 @@ void bio_integrity_setup_default(struct bio *bio); unsigned int fs_bio_integrity_alloc(struct bio *bio); void fs_bio_integrity_free(struct bio *bio); void fs_bio_integrity_generate(struct bio *bio); -int fs_bio_integrity_verify(struct bio *bio, sector_t sector, - unsigned int size); +int fs_bio_integrity_verify(struct bio *bio, struct bvec_iter *data_iter); #endif /* _LINUX_BIO_INTEGRITY_H */ diff --git a/include/linux/bio.h b/include/linux/bio.h index bb3235497e67..17944e44b584 100644 --- a/include/linux/bio.h +++ b/include/linux/bio.h @@ -479,6 +479,7 @@ static inline void bio_init_inline(struct bio *bio, struct block_device *bdev, extern void bio_uninit(struct bio *); void bio_reset(struct bio *bio, struct block_device *bdev, blk_opf_t opf); void bio_reuse(struct bio *bio, blk_opf_t opf); +void bio_prepare_reissue(struct bio *bio, struct block_device *bdev); void bio_chain(struct bio *, struct bio *); void bio_await(struct bio *bio, void *priv, void (*submit)(struct bio *bio, void *priv)); @@ -516,16 +517,18 @@ int bdev_rw_virt(struct block_device *bdev, sector_t sector, void *data, size_t len, enum req_op op); int bio_iov_iter_get_pages(struct bio *bio, struct iov_iter *iter, - unsigned mem_align_mask, unsigned len_align_mask); + unsigned maxlen, unsigned mem_align_mask, + unsigned len_align_mask); bool bio_iov_iter_set(struct bio *bio, const struct iov_iter *iter); void __bio_release_pages(struct bio *bio, bool mark_dirty); extern void bio_set_pages_dirty(struct bio *bio); extern void bio_check_pages_dirty(struct bio *bio); -int bio_iov_iter_bounce(struct bio *bio, struct iov_iter *iter, size_t maxlen, - size_t minsize); -void bio_iov_iter_unbounce(struct bio *bio, bool is_error, bool mark_dirty); +int bio_alloc_bounce_folios(struct bio *bio, size_t total_len, size_t minsize); +void bio_free_folios(struct bio *bio); +int bio_iov_iter_bounce_write(struct bio *bio, struct iov_iter *iter, + size_t maxlen, size_t minsize); extern void bio_copy_data(struct bio *dst, struct bio *src); extern void bio_free_pages(struct bio *bio); diff --git a/include/linux/blkdev.h b/include/linux/blkdev.h index 4f7905c3412b..098a65f3e48b 100644 --- a/include/linux/blkdev.h +++ b/include/linux/blkdev.h @@ -1816,9 +1816,11 @@ static inline int bio_split_rw_at(struct bio *bio, */ static inline unsigned int max_integrity_io_size(struct queue_limits *lim) { - return min_t(unsigned int, lim->max_segment_size, - (BLK_INTEGRITY_MAX_SIZE / lim->integrity.metadata_size) << - lim->integrity.interval_exp); + u64 max_intervals; + + max_intervals = BLK_INTEGRITY_MAX_SIZE / lim->integrity.metadata_size; + return min_t(u64, lim->max_segment_size, + max_intervals << lim->integrity.interval_exp); } #define DEFINE_IO_COMP_BATCH(name) struct io_comp_batch name = { } diff --git a/include/linux/buffer_head.h b/include/linux/buffer_head.h index 4b0b7188472b..f8782b719026 100644 --- a/include/linux/buffer_head.h +++ b/include/linux/buffer_head.h @@ -59,10 +59,7 @@ struct address_space; struct buffer_head { unsigned long b_state; /* buffer state bitmap (see above) */ struct buffer_head *b_this_page;/* circular list of page's buffers */ - union { - struct page *b_page; /* the page this bh is mapped to */ - struct folio *b_folio; /* the folio this bh is mapped to */ - }; + struct folio *b_folio; /* the folio this bh is mapped to */ sector_t b_blocknr; /* start block number */ size_t b_size; /* size of mapping */ @@ -172,7 +169,36 @@ static __always_inline int buffer_uptodate(const struct buffer_head *bh) static inline unsigned long bh_offset(const struct buffer_head *bh) { - return (unsigned long)(bh)->b_data & (page_size(bh->b_page) - 1); + return (unsigned long)(bh)->b_data & (folio_size(bh->b_folio) - 1); +} + +/** + * kmap_local_bh - Map the data of a buffer. + * @bh: The buffer. + * + * Buffers usually live in the page cache, but a few are built over memory + * which is not. Those carry no folio and b_data is already a kernel address + * which is always mapped, so there is nothing to do for them. Pair with + * kunmap_local_bh(). + * + * Return: A pointer to the buffer's data. + */ +static inline void *kmap_local_bh(const struct buffer_head *bh) +{ + if (!bh->b_folio) + return bh->b_data; + return kmap_local_folio(bh->b_folio, bh_offset(bh)); +} + +/** + * kunmap_local_bh - Unmap the data of a buffer. + * @bh: The buffer. + * @addr: The address returned by kmap_local_bh(). + */ +static inline void kunmap_local_bh(const struct buffer_head *bh, void *addr) +{ + if (bh->b_folio) + kunmap_local(addr); } /* If we *know* folio->private refers to buffer_heads */ @@ -332,20 +358,58 @@ static inline void bforget(struct buffer_head *bh) __bforget(bh); } -static inline struct buffer_head * -sb_bread(struct super_block *sb, sector_t block) +/** + * sb_bread - Read a block. + * @sb: The superblock to read from. + * @block: Block number in units of block size. + * + * Read a specified block, and return the buffer head that refers + * to it. The memory is allocated from the movable area so that it can + * be migrated. The returned buffer head has its refcount increased. + * The caller should call brelse() when it has finished with the buffer. + * + * Context: May sleep waiting for I/O. + * Return: NULL if the block was unreadable. + */ +static inline +struct buffer_head *sb_bread(struct super_block *sb, sector_t block) { return __bread_gfp(sb->s_bdev, block, sb->s_blocksize, __GFP_MOVABLE); } -static inline struct buffer_head * -sb_bread_unmovable(struct super_block *sb, sector_t block) +/** + * sb_bread_unmovable - Read a block. + * @sb: The superblock to read from. + * @block: Block number in units of block size. + * + * Read a specified block, and return the buffer head that refers to it. + * The memory is allocated from the unmovable area so that pointers into + * it remain valid after compaction runs. The returned buffer head has + * its refcount increased. The caller should call brelse() when it has + * finished with the buffer. + * + * Context: May sleep waiting for I/O. + * Return: NULL if the block was unreadable. + */ +static inline +struct buffer_head *sb_bread_unmovable(struct super_block *sb, sector_t block) { return __bread_gfp(sb->s_bdev, block, sb->s_blocksize, 0); } -static inline void -sb_breadahead(struct super_block *sb, sector_t block) +/** + * sb_breadahead - Start readahead. + * @sb: Superblock identifying the block device. + * @block: The block to read. + * + * Read this block. The I/O will be flagged as being readahead rather + * than immediate read, but (unlike the page cache), surrounding blocks + * will not be read. + * + * Context: May sleep in order to allocate memory. + */ +static inline +void sb_breadahead(struct super_block *sb, sector_t block) { __breadahead(sb->s_bdev, block, sb->s_blocksize); } diff --git a/include/linux/capability.h b/include/linux/capability.h index f8532d92fcad..622137f66f09 100644 --- a/include/linux/capability.h +++ b/include/linux/capability.h @@ -186,9 +186,9 @@ static inline bool ns_capable_setid(struct user_namespace *ns, int cap) } #endif /* CONFIG_MULTIUSER */ bool privileged_wrt_inode_uidgid(struct user_namespace *ns, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, const struct inode *inode); -bool capable_wrt_inode_uidgid(struct mnt_idmap *idmap, +bool capable_wrt_inode_uidgid(const struct mnt_idmap *idmap, const struct inode *inode, int cap); extern bool file_ns_capable(const struct file *file, struct user_namespace *ns, int cap); extern bool ptracer_capable(struct task_struct *tsk, struct user_namespace *ns); @@ -215,11 +215,11 @@ static inline bool checkpoint_restore_ns_capable_noaudit(struct user_namespace * } /* audit system wants to get cap info from files as well */ -int get_vfs_caps_from_disk(struct mnt_idmap *idmap, +int get_vfs_caps_from_disk(const struct mnt_idmap *idmap, const struct dentry *dentry, struct cpu_vfs_cap_data *cpu_caps); -int cap_convert_nscap(struct mnt_idmap *idmap, struct dentry *dentry, +int cap_convert_nscap(const struct mnt_idmap *idmap, struct dentry *dentry, const void **ivalue, size_t size); #endif /* !_LINUX_CAPABILITY_H */ diff --git a/include/linux/cleanup.h b/include/linux/cleanup.h index b1b5698cbf1b..1fb8058b897d 100644 --- a/include/linux/cleanup.h +++ b/include/linux/cleanup.h @@ -261,10 +261,6 @@ const volatile void * __must_check_fn(const volatile void *val) * CLASS(name, var)(args...): * declare the variable @var as an instance of the named class * - * CLASS_INIT(name, var, init_expr): - * declare the variable @var as an instance of the named class with - * custom initialization expression. - * * Ex. * * DEFINE_CLASS(fdget, struct fd, fdput(_T), fdget(fd), int fd) @@ -302,9 +298,6 @@ static __always_inline class_##_name##_t class_##_name##ext##_constructor(_init_ class_##_name##_t var __cleanup(class_##_name##_destructor) = \ class_##_name##_constructor -#define CLASS_INIT(_name, _var, _init_expr) \ - class_##_name##_t _var __cleanup(class_##_name##_destructor) = (_init_expr) - #define __scoped_class(_name, var, _label, args...) \ for (CLASS(_name, var)(args); ; ({ goto _label; })) \ if (0) { \ diff --git a/include/linux/configfs.h b/include/linux/configfs.h index ef65c75beeaa..5bead9173ec1 100644 --- a/include/linux/configfs.h +++ b/include/linux/configfs.h @@ -66,8 +66,11 @@ struct config_item_type { struct module *ct_owner; const struct configfs_item_operations *ct_item_ops; const struct configfs_group_operations *ct_group_ops; - struct configfs_attribute **ct_attrs; - struct configfs_bin_attribute **ct_bin_attrs; + union { + struct configfs_attribute **ct_attrs; + const struct configfs_attribute *const *ct_attrs_const; + }; + const struct configfs_bin_attribute *const *ct_bin_attrs; }; /** @@ -160,41 +163,41 @@ struct configfs_bin_attribute { ssize_t (*write)(struct config_item *, const void *, size_t); }; -#define CONFIGFS_BIN_ATTR(_pfx, _name, _priv, _maxsz) \ -static struct configfs_bin_attribute _pfx##attr_##_name = { \ - .cb_attr = { \ - .ca_name = __stringify(_name), \ - .ca_mode = S_IRUGO | S_IWUSR, \ - .ca_owner = THIS_MODULE, \ - }, \ - .cb_private = _priv, \ - .cb_max_size = _maxsz, \ - .read = _pfx##_name##_read, \ - .write = _pfx##_name##_write, \ +#define CONFIGFS_BIN_ATTR(_pfx, _name, _priv, _maxsz) \ +static const struct configfs_bin_attribute _pfx##attr_##_name = { \ + .cb_attr = { \ + .ca_name = __stringify(_name), \ + .ca_mode = S_IRUGO | S_IWUSR, \ + .ca_owner = THIS_MODULE, \ + }, \ + .cb_private = _priv, \ + .cb_max_size = _maxsz, \ + .read = _pfx##_name##_read, \ + .write = _pfx##_name##_write, \ } -#define CONFIGFS_BIN_ATTR_RO(_pfx, _name, _priv, _maxsz) \ -static struct configfs_bin_attribute _pfx##attr_##_name = { \ - .cb_attr = { \ - .ca_name = __stringify(_name), \ - .ca_mode = S_IRUGO, \ - .ca_owner = THIS_MODULE, \ - }, \ - .cb_private = _priv, \ - .cb_max_size = _maxsz, \ - .read = _pfx##_name##_read, \ +#define CONFIGFS_BIN_ATTR_RO(_pfx, _name, _priv, _maxsz) \ +static const struct configfs_bin_attribute _pfx##attr_##_name = { \ + .cb_attr = { \ + .ca_name = __stringify(_name), \ + .ca_mode = S_IRUGO, \ + .ca_owner = THIS_MODULE, \ + }, \ + .cb_private = _priv, \ + .cb_max_size = _maxsz, \ + .read = _pfx##_name##_read, \ } -#define CONFIGFS_BIN_ATTR_WO(_pfx, _name, _priv, _maxsz) \ -static struct configfs_bin_attribute _pfx##attr_##_name = { \ - .cb_attr = { \ - .ca_name = __stringify(_name), \ - .ca_mode = S_IWUSR, \ - .ca_owner = THIS_MODULE, \ - }, \ - .cb_private = _priv, \ - .cb_max_size = _maxsz, \ - .write = _pfx##_name##_write, \ +#define CONFIGFS_BIN_ATTR_WO(_pfx, _name, _priv, _maxsz) \ +static const struct configfs_bin_attribute _pfx##attr_##_name = { \ + .cb_attr = { \ + .ca_name = __stringify(_name), \ + .ca_mode = S_IWUSR, \ + .ca_owner = THIS_MODULE, \ + }, \ + .cb_private = _priv, \ + .cb_max_size = _maxsz, \ + .write = _pfx##_name##_write, \ } /* @@ -220,8 +223,8 @@ struct configfs_group_operations { struct config_group *(*make_group)(struct config_group *group, const char *name); void (*disconnect_notify)(struct config_group *group, struct config_item *item); void (*drop_item)(struct config_group *group, struct config_item *item); - bool (*is_visible)(struct config_item *item, struct configfs_attribute *attr, int n); - bool (*is_bin_visible)(struct config_item *item, struct configfs_bin_attribute *attr, + bool (*is_visible)(struct config_item *item, const struct configfs_attribute *attr, int n); + bool (*is_bin_visible)(struct config_item *item, const struct configfs_bin_attribute *attr, int n); }; diff --git a/include/linux/coredump.h b/include/linux/coredump.h index 7b38ee2e7913..74af57b9406b 100644 --- a/include/linux/coredump.h +++ b/include/linux/coredump.h @@ -6,9 +6,20 @@ #include <linux/mm.h> #include <linux/fs.h> #include <linux/sched/coredump.h> +#include <uapi/linux/coredump.h> #include <asm/siginfo.h> #ifdef CONFIG_COREDUMP +/** + * enum coredump_state - what happened while the coredump was written + * @COREDUMP_STATE_STARTED: the dumper committed to writing a coredump + * @COREDUMP_STATE_TRUNCATED: the dumper stopped before it had written all of it + */ +enum coredump_state { + COREDUMP_STATE_STARTED = (1U << 0), + COREDUMP_STATE_TRUNCATED = (1U << 1), +}; + struct core_vma_metadata { unsigned long start, end; vm_flags_t flags; @@ -21,12 +32,20 @@ struct coredump_params { const kernel_siginfo_t *siginfo; struct file *file; unsigned long limit; - /* MMF_DUMP_FILTER_* bits, snapshot of mm->flags at dump start. */ - unsigned long mm_flags; + /* COREDUMP_MEMORY_* types to dump, the task's or the server's. */ + u64 memory_types; /* Snapshot of dumpable at dump start. */ enum task_dumpable dumpable; int cpu; + /* COREDUMP_* options negotiated with the coredump server. */ + u64 mask; + /* COREDUMP_STATE_* raised while the coredump is written. */ + enum coredump_state state; + /* Record header scratch, NULL unless the coredump is a record stream. */ + struct coredump_record_header *record_hdr; + /* Bytes handed to the file, record headers included. */ loff_t written; + /* Offset in the coredump, record headers excluded. */ loff_t pos; loff_t to_skip; int vma_count; @@ -41,13 +60,13 @@ extern unsigned int core_file_note_size_limit; * These are the only things you should do on a core-file: use only these * functions to write out all the necessary info. */ -extern void dump_skip_to(struct coredump_params *cprm, unsigned long to); -extern void dump_skip(struct coredump_params *cprm, size_t nr); -extern int dump_emit(struct coredump_params *cprm, const void *addr, int nr); -extern int dump_align(struct coredump_params *cprm, int align); -int dump_user_range(struct coredump_params *cprm, unsigned long start, - unsigned long len); -extern void vfs_coredump(const kernel_siginfo_t *siginfo); +void dump_skip_to(struct coredump_params *cprm, unsigned long to); +void dump_skip(struct coredump_params *cprm, size_t nr); +bool dump_emit(struct coredump_params *cprm, const void *addr, int nr); +bool dump_align(struct coredump_params *cprm, int align); +bool dump_user_range(struct coredump_params *cprm, unsigned long start, + unsigned long len); +void vfs_coredump(const kernel_siginfo_t *siginfo); /* * Logging for the coredump code, ratelimited. diff --git a/include/linux/dax.h b/include/linux/dax.h index fe6c3ded1b50..f2d47975d905 100644 --- a/include/linux/dax.h +++ b/include/linux/dax.h @@ -155,8 +155,6 @@ int dax_writeback_mapping_range(struct address_space *mapping, struct dax_device *dax_dev, struct writeback_control *wbc); int dax_folio_reset_order(struct folio *folio); -struct page *dax_layout_busy_page(struct address_space *mapping); -struct page *dax_layout_busy_page_range(struct address_space *mapping, loff_t start, loff_t end); dax_entry_t dax_lock_folio(struct folio *folio); void dax_unlock_folio(struct folio *folio, dax_entry_t cookie); dax_entry_t dax_lock_mapping_entry(struct address_space *mapping, @@ -173,16 +171,6 @@ static inline int fs_dax_get(struct dax_device *dax_dev, void *holder, { return -EOPNOTSUPP; } -static inline struct page *dax_layout_busy_page(struct address_space *mapping) -{ - return NULL; -} - -static inline struct page *dax_layout_busy_page_range(struct address_space *mapping, pgoff_t start, pgoff_t nr_pages) -{ - return NULL; -} - static inline int dax_writeback_mapping_range(struct address_space *mapping, struct dax_device *dax_dev, struct writeback_control *wbc) { diff --git a/include/linux/dcache.h b/include/linux/dcache.h index 4b1ff99608e0..adf239f8205f 100644 --- a/include/linux/dcache.h +++ b/include/linux/dcache.h @@ -116,6 +116,8 @@ struct dentry { * possible! */ + /* lockdep tracking of DCACHE_PAR_LOOKUP locks */ + struct lockdep_map lookup_map; struct list_head d_lru; /* LRU list */ struct hlist_node d_sib; /* child of parent list */ struct hlist_head d_children; /* our children */ @@ -236,7 +238,9 @@ enum dentry_flags { DCACHE_PAR_LOOKUP = BIT(24), /* being looked up (with parent locked shared) */ DCACHE_DENTRY_CURSOR = BIT(25), DCACHE_NORCU = BIT(26), /* No RCU delay for freeing */ - DCACHE_PERSISTENT = BIT(27) + DCACHE_PERSISTENT = BIT(27), +/* 28, 29, 30 free */ + DCACHE_PRIVATE = BIT(31) /* fs-specific flag */ }; #define DCACHE_MANAGED_DENTRY \ @@ -257,7 +261,9 @@ extern void d_delete(struct dentry *); extern struct dentry * d_alloc(struct dentry *, const struct qstr *); extern struct dentry * d_alloc_anon(struct super_block *); extern struct dentry * d_alloc_parallel(struct dentry *, const struct qstr *); +extern struct dentry * d_alloc_trylock(struct dentry *, struct qstr *); extern struct dentry * d_splice_alias(struct inode *, struct dentry *); +struct dentry *d_duplicate(struct dentry *dentry); /* weird procfs mess; *NOT* exported */ extern struct dentry * d_splice_alias_ops(struct inode *, struct dentry *, const struct dentry_operations *); @@ -553,6 +559,36 @@ static inline int simple_positive(const struct dentry *dentry) unsigned long vfs_pressure_ratio(unsigned long val); /** + * d_lookup_release - release ownership of DCACHE_PAR_LOOKUP lock + * @dentry: dentry that is locked + * + * If an in-lookup dentry is to be passed to another thread which + * will drop the in-lookup lock, then d_lookup_release() must be called + * to tell lockdep that this thread no lock holds the lock. The + * thread that receives the lock must call d_lookup_acquire() to + * acquire the lock. + */ +static inline void d_lookup_release(struct dentry *dentry) +{ + if (d_in_lookup(dentry)) + lock_map_release(&dentry->lookup_map); +} + +/** + * d_lookup_acquire - acquire ownership of DCACHE_PAR_LOOKUP lock + * @dentry: dentry that is locked + * + * If an in-lookup dentry was passed to this thread, the + * d_lookup_acquire() must be called to tell lockdep that this + * thread now owns the DCACHE_PAR_LOOKUP lock. + */ +static inline void d_lookup_acquire(struct dentry *dentry) +{ + if (d_in_lookup(dentry)) + lock_map_acquire_try(&dentry->lookup_map); +} + +/** * d_inode - Get the actual inode of this dentry * @dentry: The dentry to query * diff --git a/include/linux/f2fs_fs.h b/include/linux/f2fs_fs.h index bb2b6cd5d507..3081702b1ddb 100644 --- a/include/linux/f2fs_fs.h +++ b/include/linux/f2fs_fs.h @@ -14,9 +14,9 @@ #define F2FS_SUPER_OFFSET 1024 /* byte-size offset */ #define F2FS_MIN_LOG_SECTOR_SIZE 9 /* 9 bits for 512 bytes */ #define F2FS_MAX_LOG_SECTOR_SIZE PAGE_SHIFT /* Max is Block Size */ -#define F2FS_LOG_SECTORS_PER_BLOCK (PAGE_SHIFT - 9) /* log number for sector/blk */ -#define F2FS_BLKSIZE PAGE_SIZE /* support only block == page */ -#define F2FS_BLKSIZE_BITS PAGE_SHIFT /* bits for F2FS_BLKSIZE */ +#define F2FS_MIN_LOG_BLOCKSIZE 12 +#define F2FS_MIN_BLKSIZE 4096UL +#define F2FS_MAX_BLKSIZE PAGE_SIZE #define F2FS_MAX_EXTENSION 64 /* # of extension entries */ #define F2FS_EXTENSION_LEN 8 /* max size of extension */ @@ -24,19 +24,25 @@ #define NEW_ADDR ((block_t)-1) /* used as block_t addresses */ #define COMPRESS_ADDR ((block_t)-2) /* used as compressed data flag */ -#define F2FS_BLKSIZE_MASK (F2FS_BLKSIZE - 1) -#define F2FS_BYTES_TO_BLK(bytes) ((unsigned long long)(bytes) >> F2FS_BLKSIZE_BITS) -#define F2FS_BLK_TO_BYTES(blk) ((unsigned long long)(blk) << F2FS_BLKSIZE_BITS) -#define F2FS_BLK_END_BYTES(blk) (F2FS_BLK_TO_BYTES(blk + 1) - 1) -#define F2FS_BLK_ALIGN(x) (F2FS_BYTES_TO_BLK((x) + F2FS_BLKSIZE - 1)) +#define F2FS_BLKSIZE(sbi) ((sbi)->blocksize) +#define F2FS_BLKSIZE_BITS(sbi) ((sbi)->log_blocksize) +#define F2FS_BLKSIZE_MASK(sbi) (F2FS_BLKSIZE(sbi) - 1) +#define F2FS_LOG_SECTORS_PER_BLOCK(sbi) (F2FS_BLKSIZE_BITS(sbi) - 9) +#define F2FS_BLKS_PER_PAGE(sbi) (PAGE_SIZE / F2FS_BLKSIZE(sbi)) +#define F2FS_BYTES_TO_BLK(sbi, bytes) \ + ((unsigned long long)(bytes) >> F2FS_BLKSIZE_BITS(sbi)) +#define F2FS_BLK_TO_BYTES(sbi, blk) \ + ((unsigned long long)(blk) << F2FS_BLKSIZE_BITS(sbi)) +#define F2FS_BLK_END_BYTES(sbi, blk) \ + (F2FS_BLK_TO_BYTES(sbi, (blk) + 1) - 1) +#define F2FS_BLK_ALIGN(sbi, bytes) \ + F2FS_BYTES_TO_BLK(sbi, (unsigned long long)(bytes) + \ + F2FS_BLKSIZE(sbi) - 1) /* 0, 1(node nid), 2(meta nid) are reserved node id */ #define F2FS_RESERVED_NODE_NUM 3 #define F2FS_ROOT_INO(sbi) ((sbi)->root_ino_num) -#define F2FS_NODE_INO(sbi) ((sbi)->node_ino_num) -#define F2FS_META_INO(sbi) ((sbi)->meta_ino_num) -#define F2FS_COMPRESS_INO(sbi) (NM_I(sbi)->max_nid) #define F2FS_MAX_QUOTAS 3 @@ -214,20 +220,27 @@ struct f2fs_checkpoint { unsigned char sit_nat_version_bitmap[]; } __packed; -#define CP_CHKSUM_OFFSET (F2FS_BLKSIZE - sizeof(__le32)) /* default chksum offset in checkpoint */ #define CP_MIN_CHKSUM_OFFSET \ (offsetof(struct f2fs_checkpoint, sit_nat_version_bitmap)) /* * For orphan inode management + * + * The number of inode entries in an orphan block depends on the filesystem + * block size. Its exact on-disk layout is: + * + * 0 blocksize - 16 blocksize + * +--------------------------+--------------------------+ + * | ino[0] ... ino[n - 1] | struct f2fs_orphan_footer | + * +--------------------------+--------------------------+ + * + * n = (blocksize - sizeof(struct f2fs_orphan_footer)) / sizeof(__le32) */ -#define F2FS_ORPHANS_PER_BLOCK ((F2FS_BLKSIZE - 4 * sizeof(__le32)) / sizeof(__le32)) - -#define GET_ORPHAN_BLOCKS(n) (((n) + F2FS_ORPHANS_PER_BLOCK - 1) / \ - F2FS_ORPHANS_PER_BLOCK) - struct f2fs_orphan_block { - __le32 ino[F2FS_ORPHANS_PER_BLOCK]; /* inode numbers */ + DECLARE_FLEX_ARRAY(__le32, ino); +} __packed; + +struct f2fs_orphan_footer { __le32 reserved; /* reserved */ __le16 blk_addr; /* block index in current CP */ __le16 blk_count; /* Number of orphan inode blocks in CP */ @@ -260,26 +273,14 @@ struct node_footer { } __packed; /* Address Pointers in an Inode */ -#define DEF_ADDRS_PER_INODE ((F2FS_BLKSIZE - OFFSET_OF_END_OF_I_EXT \ - - SIZE_OF_I_NID \ - - sizeof(struct node_footer)) / sizeof(__le32)) -#define CUR_ADDRS_PER_INODE(inode) (DEF_ADDRS_PER_INODE - \ - get_extra_isize(inode)) +#define F2FS_DEF_ADDRS_PER_INODE(blocksize) \ + (((blocksize) - OFFSET_OF_END_OF_I_EXT - SIZE_OF_I_NID - \ + sizeof(struct node_footer)) / sizeof(__le32)) #define DEF_NIDS_PER_INODE 5 /* Node IDs in an Inode */ #define ADDRS_PER_INODE(inode) addrs_per_page(inode, true) /* Address Pointers in a Direct Block */ -#define DEF_ADDRS_PER_BLOCK ((F2FS_BLKSIZE - sizeof(struct node_footer)) / sizeof(__le32)) #define ADDRS_PER_BLOCK(inode) addrs_per_page(inode, false) -/* Node IDs in an Indirect Block */ -#define NIDS_PER_BLOCK ((F2FS_BLKSIZE - sizeof(struct node_footer)) / sizeof(__le32)) - -#define ADDRS_PER_PAGE(folio, inode) (addrs_per_page(inode, IS_INODE(folio))) - -#define NODE_DIR1_BLOCK (DEF_ADDRS_PER_INODE + 1) -#define NODE_DIR2_BLOCK (DEF_ADDRS_PER_INODE + 2) -#define NODE_IND1_BLOCK (DEF_ADDRS_PER_INODE + 3) -#define NODE_IND2_BLOCK (DEF_ADDRS_PER_INODE + 4) -#define NODE_DIND_BLOCK (DEF_ADDRS_PER_INODE + 5) +#define ADDRS_PER_PAGE(folio, inode) (addrs_per_page(inode, IS_INODE(F2FS_I_SB(inode), folio))) #define F2FS_INLINE_XATTR 0x01 /* file inline xattr flag */ #define F2FS_INLINE_DATA 0x02 /* file inline data flag */ @@ -339,18 +340,26 @@ struct f2fs_inode { */ __le32 i_extra_end[0]; /* for attribute size calculation */ } __packed; - __le32 i_addr[DEF_ADDRS_PER_INODE]; /* Pointers to data blocks */ + DECLARE_FLEX_ARRAY(__le32, i_addr); /* data block pointers */ }; - __le32 i_nid[DEF_NIDS_PER_INODE]; /* direct(2), indirect(2), - double_indirect(1) node id */ + /* + * __le32 i_nid[DEF_NIDS_PER_INODE]; + * direct(2), indirect(2), double_indirect(1) node IDs + * + * It is stored immediately before the node footer at the end of the + * filesystem block. Its offset depends on the filesystem block size, so + * locate it dynamically with F2FS_INODE_NIDS(). + */ } __packed; struct direct_node { - __le32 addr[DEF_ADDRS_PER_BLOCK]; /* array of data block address */ + /* The address count depends on the filesystem block size. */ + DECLARE_FLEX_ARRAY(__le32, addr); /* array of data block address */ } __packed; struct indirect_node { - __le32 nid[NIDS_PER_BLOCK]; /* array of data block address */ + /* The node ID count depends on the filesystem block size. */ + DECLARE_FLEX_ARRAY(__le32, nid); /* array of data block address */ } __packed; enum { @@ -369,14 +378,18 @@ struct f2fs_node { struct direct_node dn; struct indirect_node in; }; - struct node_footer footer; + /* + * struct node_footer footer; + * + * It is stored at the end of the filesystem block, after the inode or + * direct/indirect node data. Its offset depends on the filesystem block + * size, so locate it dynamically with F2FS_NODE_FOOTER(). + */ } __packed; /* * For NAT entries */ -#define NAT_ENTRY_PER_BLOCK (F2FS_BLKSIZE / sizeof(struct f2fs_nat_entry)) - struct f2fs_nat_entry { __u8 version; /* latest version of cached nat entry */ __le32 ino; /* inode number */ @@ -384,7 +397,8 @@ struct f2fs_nat_entry { } __packed; struct f2fs_nat_block { - struct f2fs_nat_entry entries[NAT_ENTRY_PER_BLOCK]; + /* The entry count depends on the filesystem block size. */ + DECLARE_FLEX_ARRAY(struct f2fs_nat_entry, entries); } __packed; /* @@ -396,8 +410,6 @@ struct f2fs_nat_block { * Not allow to change this. */ #define SIT_VBLOCK_MAP_SIZE 64 -#define SIT_ENTRY_PER_BLOCK (F2FS_BLKSIZE / sizeof(struct f2fs_sit_entry)) - /* * F2FS uses 4 bytes to represent block address. As a result, supported size of * disk is 16 TB for a 4K page size and 64 TB for a 16K page size and it equals @@ -424,8 +436,13 @@ struct f2fs_sit_entry { __le64 mtime; /* segment age for cleaning */ } __packed; +/* + * The on-disk SIT block is a filesystem-block-sized array of SIT entries. + * Its entry count depends on the filesystem block size, so it must be + * calculated by the caller rather than implied by this C structure. + */ struct f2fs_sit_block { - struct f2fs_sit_entry entries[SIT_ENTRY_PER_BLOCK]; + DECLARE_FLEX_ARRAY(struct f2fs_sit_entry, entries); } __packed; /* @@ -595,15 +612,7 @@ typedef __le32 f2fs_hash_t; * dentry, when converting inline dentry we should handle this carefully. */ -/* the number of dentry in a block */ -#define NR_DENTRY_IN_BLOCK ((BITS_PER_BYTE * F2FS_BLKSIZE) / \ - ((SIZE_OF_DIR_ENTRY + F2FS_SLOT_LEN) * BITS_PER_BYTE + 1)) #define SIZE_OF_DIR_ENTRY 11 /* by byte */ -#define SIZE_OF_DENTRY_BITMAP ((NR_DENTRY_IN_BLOCK + BITS_PER_BYTE - 1) / \ - BITS_PER_BYTE) -#define SIZE_OF_RESERVED (F2FS_BLKSIZE - ((SIZE_OF_DIR_ENTRY + \ - F2FS_SLOT_LEN) * \ - NR_DENTRY_IN_BLOCK + SIZE_OF_DENTRY_BITMAP)) #define MIN_INLINE_DENTRY_SIZE 40 /* just include '.' and '..' entries */ /* One directory entry slot representing F2FS_SLOT_LEN-sized file name */ @@ -614,14 +623,21 @@ struct f2fs_dir_entry { __u8 file_type; /* file type */ } __packed; -/* Block-sized directory entry block */ -struct f2fs_dentry_block { - /* validity bitmap for directory entries in each block */ - __u8 dentry_bitmap[SIZE_OF_DENTRY_BITMAP]; - __u8 reserved[SIZE_OF_RESERVED]; - struct f2fs_dir_entry dentry[NR_DENTRY_IN_BLOCK]; - __u8 filename[NR_DENTRY_IN_BLOCK][F2FS_SLOT_LEN]; -} __packed; +/* + * A dentry block is laid out as follows, where the number of entries and all + * offsets are determined by the filesystem block size at runtime: + * + * 0 blocksize + * +--------+----------+-------------------+-----------------------+ + * | bitmap | reserved | dir_entry[entries]| filename[entries][8] | + * +--------+----------+-------------------+-----------------------+ + * + * entries = (BITS_PER_BYTE * blocksize) / + * ((SIZE_OF_DIR_ENTRY + F2FS_SLOT_LEN) * BITS_PER_BYTE + 1) + * bitmap_size = DIV_ROUND_UP(entries, BITS_PER_BYTE) + * reserved_size = blocksize - bitmap_size - + * (SIZE_OF_DIR_ENTRY + F2FS_SLOT_LEN) * entries + */ #define F2FS_DEF_PROJID 0 /* default project ID */ diff --git a/include/linux/fdtable.h b/include/linux/fdtable.h index c45306a9f007..a46781058729 100644 --- a/include/linux/fdtable.h +++ b/include/linux/fdtable.h @@ -25,7 +25,7 @@ struct fdtable { unsigned int max_fds; - struct file __rcu **fd; /* current fd array */ + struct file __rcu **fd __counted_by_ptr(max_fds); /* current fd array */ unsigned long *close_on_exec; unsigned long *open_fds; unsigned long *full_fds_bits; @@ -101,11 +101,22 @@ struct task_struct; void put_files_struct(struct files_struct *fs); int unshare_files(void); +void switch_files_struct(struct task_struct *tsk, struct files_struct *files); +int unshare_fd(unsigned long unshare_flags, struct files_struct **new_fdp); +enum fd_range_flags { + /* Leave behind all descriptors outside of the specified range. */ + FD_RANGE_EXCEPT = (1U << 0), + + /* Only select descriptors that have close-on-exec set. */ + FD_RANGE_CLOEXEC_ONLY = (1U << 1), +}; + struct fd_range { unsigned int from, to; + enum fd_range_flags flags; }; struct files_struct *dup_fd(struct files_struct *, struct fd_range *) __latent_entropy; -void do_close_on_exec(struct files_struct *); +void close_cloexec_files(struct files_struct *); int iterate_fd(struct files_struct *, unsigned, int (*)(const void *, struct file *, unsigned), const void *); diff --git a/include/linux/file.h b/include/linux/file.h index 27484b444d31..41c3c0be1064 100644 --- a/include/linux/file.h +++ b/include/linux/file.h @@ -12,6 +12,7 @@ #include <linux/errno.h> #include <linux/cleanup.h> #include <linux/err.h> +#include <linux/vfsdebug.h> struct file; @@ -129,117 +130,84 @@ extern unsigned int sysctl_nr_open_min, sysctl_nr_open_max; /* * fd_prepare: Combined fd + file allocation cleanup class. - * @err: Error code to indicate if allocation succeeded. - * @__fd: Allocated fd (may not be accessed directly) - * @__file: Allocated struct file pointer (may not be accessed directly) + * @fd: Allocated fd + * @file: Allocated struct file pointer * * Allocates an fd and a file together. On error paths, automatically cleans * up whichever resource was successfully allocated. Allows flexible file * allocation with different functions per usage. * - * Do not use directly. + * Do not declare directly, use FD_PREPARE(). */ struct fd_prepare { - s32 err; - s32 __fd; /* do not access directly */ - struct file *__file; /* do not access directly */ + int fd; + struct file *file; }; -/* Typedef for fd_prepare cleanup guards. */ -typedef struct fd_prepare class_fd_prepare_t; - -/* - * Accessors for fd_prepare class members. - * _Generic() is used for zero-cost type safety. - */ -#define fd_prepare_fd(_fdf) \ - (_Generic((_fdf), struct fd_prepare: (_fdf).__fd)) - -#define fd_prepare_file(_fdf) \ - (_Generic((_fdf), struct fd_prepare: (_fdf).__file)) - /* Do not use directly. */ -static inline void class_fd_prepare_destructor(const struct fd_prepare *fdf) +static __always_inline void __fd_prepare_cleanup(const struct fd_prepare *fdf) { - if (unlikely(fdf->__fd >= 0)) - put_unused_fd(fdf->__fd); - if (unlikely(!IS_ERR_OR_NULL(fdf->__file))) - fput(fdf->__file); + if (unlikely(fdf->fd >= 0)) { + put_unused_fd(fdf->fd); + fput(fdf->file); + } } /* Do not use directly. */ -static inline int class_fd_prepare_lock_err(const struct fd_prepare *fdf) +static __always_inline struct fd_prepare __fd_prepare(int fd, struct file *file) { - if (unlikely(fdf->err)) - return fdf->err; - if (unlikely(fdf->__fd < 0)) - return fdf->__fd; - if (unlikely(IS_ERR(fdf->__file))) - return PTR_ERR(fdf->__file); - if (unlikely(!fdf->__file)) - return -ENOMEM; - return 0; + if (fd >= 0 && IS_ERR_OR_NULL(file)) { + int err = file ? PTR_ERR(file) : -ENOMEM; + + put_unused_fd(fd); + fd = err; + file = NULL; + } + return (struct fd_prepare){ .fd = fd, .file = file }; } /* - * __FD_PREPARE_INIT - Helper to initialize fd_prepare class. - * @_fd_flags: flags for get_unused_fd_flags() - * @_file_owned: expression that returns struct file * - * - * Returns a struct fd_prepare with fd, file, and err set. - * If fd allocation fails, fd will be negative and err will be set. If - * fd succeeds but file_init_expr fails, file will be ERR_PTR and err - * will be set. The err field is the single source of truth for error - * checking. - */ -#define __FD_PREPARE_INIT(_fd_flags, _file_owned) \ - ({ \ - struct fd_prepare fdf = { \ - .__fd = get_unused_fd_flags((_fd_flags)), \ - }; \ - if (likely(fdf.__fd >= 0)) \ - fdf.__file = (_file_owned); \ - fdf.err = ACQUIRE_ERR(fd_prepare, &fdf); \ - fdf; \ - }) - -/* - * FD_PREPARE - Macro to declare and initialize an fd_prepare variable. + * FD_PREPARE - Declare and initialize an fd_prepare instance. * - * Declares and initializes an fd_prepare variable with automatic - * cleanup. No separate scope required - cleanup happens when variable - * goes out of scope. + * This allocates a new fd and only evaluates @_file_owned if the + * allocation succeeded. Cleanup happens when the variable goes out of + * scope and the guard releases whichever of the descriptor and the file + * was allocated. If fd_publish() was called the fd and file are + * published and cleanup becomes a nop. * - * @_fdf: name of struct fd_prepare variable to define + * @_fdf: name of the const struct fd_prepare pointer to define * @_fd_flags: flags for get_unused_fd_flags() * @_file_owned: struct file to take ownership of (can be expression) */ +#define __FD_PREPARE(_guard, _fdf, _fd_flags, _file_owned) \ + struct fd_prepare _guard __cleanup(__fd_prepare_cleanup) = ({ \ + int __fd = get_unused_fd_flags(_fd_flags); \ + __fd_prepare(__fd, __fd < 0 ? NULL : (_file_owned)); \ + }); \ + const struct fd_prepare *const _fdf = &_guard + #define FD_PREPARE(_fdf, _fd_flags, _file_owned) \ - CLASS_INIT(fd_prepare, _fdf, __FD_PREPARE_INIT(_fd_flags, _file_owned)) + __FD_PREPARE(__UNIQUE_ID(fd_prepare), _fdf, _fd_flags, _file_owned) /* * fd_publish - Publish prepared fd and file to the fd table. - * @_fdf: struct fd_prepare variable + * @fdf: struct fd_prepare pointer defined by FD_PREPARE() */ -#define fd_publish(_fdf) \ - ({ \ - struct fd_prepare *fdp = &(_fdf); \ - VFS_WARN_ON_ONCE(fdp->err); \ - VFS_WARN_ON_ONCE(fdp->__fd < 0); \ - VFS_WARN_ON_ONCE(IS_ERR_OR_NULL(fdp->__file)); \ - fd_install(fdp->__fd, fdp->__file); \ - retain_and_null_ptr(fdp->__file); \ - take_fd(fdp->__fd); \ - }) +static __always_inline int fd_publish(const struct fd_prepare *fdf) +{ + /* Callers only get a const view, the guard itself is writable. */ + struct fd_prepare *guard = (struct fd_prepare *)fdf; + + VFS_WARN_ON_ONCE(guard->fd < 0); + fd_install(guard->fd, guard->file); + return take_fd(guard->fd); +} /* Do not use directly. */ -#define __FD_ADD(_fdf, _fd_flags, _file_owned) \ - ({ \ - FD_PREPARE(_fdf, _fd_flags, _file_owned); \ - s32 ret = _fdf.err; \ - if (likely(!ret)) \ - ret = fd_publish(_fdf); \ - ret; \ +#define __FD_ADD(_fdf, _fd_flags, _file_owned) \ + ({ \ + FD_PREPARE(_fdf, _fd_flags, _file_owned); \ + _fdf->fd < 0 ? _fdf->fd : fd_publish(_fdf); \ }) /* diff --git a/include/linux/fileattr.h b/include/linux/fileattr.h index 58044b598016..09e32b84e02a 100644 --- a/include/linux/fileattr.h +++ b/include/linux/fileattr.h @@ -74,7 +74,7 @@ static inline bool fileattr_has_fsx(const struct file_kattr *fa) } int vfs_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -int vfs_fileattr_set(struct mnt_idmap *idmap, struct dentry *dentry, +int vfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); int ioctl_getflags(struct file *file, unsigned int __user *argp); int ioctl_setflags(struct file *file, unsigned int __user *argp); diff --git a/include/linux/fs.h b/include/linux/fs.h index f9d1e05e8ae6..3db90996756f 100644 --- a/include/linux/fs.h +++ b/include/linux/fs.h @@ -1436,10 +1436,10 @@ static inline void i_gid_write(struct inode *inode, gid_t gid) * @idmap: idmap of the mount the inode was found from * @inode: inode to map * - * Return: whe inode's i_uid mapped down according to @idmap. + * Return: the inode's i_uid mapped down according to @idmap. * If the inode's i_uid has no mapping INVALID_VFSUID is returned. */ -static inline vfsuid_t i_uid_into_vfsuid(struct mnt_idmap *idmap, +static inline vfsuid_t i_uid_into_vfsuid(const struct mnt_idmap *idmap, const struct inode *inode) { return make_vfsuid(idmap, i_user_ns(inode), inode->i_uid); @@ -1456,7 +1456,7 @@ static inline vfsuid_t i_uid_into_vfsuid(struct mnt_idmap *idmap, * * Return: true if @inode's i_uid field needs to be updated, false if not. */ -static inline bool i_uid_needs_update(struct mnt_idmap *idmap, +static inline bool i_uid_needs_update(const struct mnt_idmap *idmap, const struct iattr *attr, const struct inode *inode) { @@ -1474,7 +1474,7 @@ static inline bool i_uid_needs_update(struct mnt_idmap *idmap, * Safely update @inode's i_uid field translating the vfsuid of any idmapped * mount into the filesystem kuid. */ -static inline void i_uid_update(struct mnt_idmap *idmap, +static inline void i_uid_update(const struct mnt_idmap *idmap, const struct iattr *attr, struct inode *inode) { @@ -1491,7 +1491,7 @@ static inline void i_uid_update(struct mnt_idmap *idmap, * Return: the inode's i_gid mapped down according to @idmap. * If the inode's i_gid has no mapping INVALID_VFSGID is returned. */ -static inline vfsgid_t i_gid_into_vfsgid(struct mnt_idmap *idmap, +static inline vfsgid_t i_gid_into_vfsgid(const struct mnt_idmap *idmap, const struct inode *inode) { return make_vfsgid(idmap, i_user_ns(inode), inode->i_gid); @@ -1508,7 +1508,7 @@ static inline vfsgid_t i_gid_into_vfsgid(struct mnt_idmap *idmap, * * Return: true if @inode's i_gid field needs to be updated, false if not. */ -static inline bool i_gid_needs_update(struct mnt_idmap *idmap, +static inline bool i_gid_needs_update(const struct mnt_idmap *idmap, const struct iattr *attr, const struct inode *inode) { @@ -1526,7 +1526,7 @@ static inline bool i_gid_needs_update(struct mnt_idmap *idmap, * Safely update @inode's i_gid field translating the vfsgid of any idmapped * mount into the filesystem kgid. */ -static inline void i_gid_update(struct mnt_idmap *idmap, +static inline void i_gid_update(const struct mnt_idmap *idmap, const struct iattr *attr, struct inode *inode) { @@ -1544,7 +1544,7 @@ static inline void i_gid_update(struct mnt_idmap *idmap, * an idmapped mount map the caller's fsuid according to @idmap. */ static inline void inode_fsuid_set(struct inode *inode, - struct mnt_idmap *idmap) + const struct mnt_idmap *idmap) { inode->i_uid = mapped_fsuid(idmap, i_user_ns(inode)); } @@ -1558,7 +1558,7 @@ static inline void inode_fsuid_set(struct inode *inode, * an idmapped mount map the caller's fsgid according to @idmap. */ static inline void inode_fsgid_set(struct inode *inode, - struct mnt_idmap *idmap) + const struct mnt_idmap *idmap) { inode->i_gid = mapped_fsgid(idmap, i_user_ns(inode)); } @@ -1575,7 +1575,7 @@ static inline void inode_fsgid_set(struct inode *inode, * Return: true if fsuid and fsgid is mapped, false if not. */ static inline bool fsuidgid_has_mapping(struct super_block *sb, - struct mnt_idmap *idmap) + const struct mnt_idmap *idmap) { struct user_namespace *fs_userns = sb->s_user_ns; kuid_t kuid; @@ -1755,25 +1755,25 @@ static inline bool file_write_not_started(const struct file *file) return sb_write_not_started(file_inode(file)->i_sb); } -bool inode_owner_or_capable(struct mnt_idmap *idmap, +bool inode_owner_or_capable(const struct mnt_idmap *idmap, const struct inode *inode); /* * VFS helper functions.. */ -int vfs_create(struct mnt_idmap *, struct dentry *, umode_t, +int vfs_create(const struct mnt_idmap *, struct dentry *, umode_t, struct delegated_inode *); -struct dentry *vfs_mkdir(struct mnt_idmap *, struct inode *, +struct dentry *vfs_mkdir(const struct mnt_idmap *, struct inode *, struct dentry *, umode_t, struct delegated_inode *); -int vfs_mknod(struct mnt_idmap *, struct inode *, struct dentry *, +int vfs_mknod(const struct mnt_idmap *, struct inode *, struct dentry *, umode_t, dev_t, struct delegated_inode *); -int vfs_symlink(struct mnt_idmap *, struct inode *, +int vfs_symlink(const struct mnt_idmap *, struct inode *, struct dentry *, const char *, struct delegated_inode *); -int vfs_link(struct dentry *, struct mnt_idmap *, struct inode *, +int vfs_link(struct dentry *, const struct mnt_idmap *, struct inode *, struct dentry *, struct delegated_inode *); -int vfs_rmdir(struct mnt_idmap *, struct inode *, struct dentry *, +int vfs_rmdir(const struct mnt_idmap *, struct inode *, struct dentry *, struct delegated_inode *); -int vfs_unlink(struct mnt_idmap *, struct inode *, struct dentry *, +int vfs_unlink(const struct mnt_idmap *, struct inode *, struct dentry *, struct delegated_inode *); /** @@ -1787,7 +1787,7 @@ int vfs_unlink(struct mnt_idmap *, struct inode *, struct dentry *, * @flags: rename flags */ struct renamedata { - struct mnt_idmap *mnt_idmap; + const struct mnt_idmap *mnt_idmap; struct dentry *old_parent; struct dentry *old_dentry; struct dentry *new_parent; @@ -1798,14 +1798,14 @@ struct renamedata { int vfs_rename(struct renamedata *); -static inline int vfs_whiteout(struct mnt_idmap *idmap, +static inline int vfs_whiteout(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry) { return vfs_mknod(idmap, dir, dentry, S_IFCHR | WHITEOUT_MODE, WHITEOUT_DEV, NULL); } -struct file *kernel_tmpfile_open(struct mnt_idmap *idmap, +struct file *kernel_tmpfile_open(const struct mnt_idmap *idmap, const struct path *parentpath, umode_t mode, int open_flag, const struct cred *cred); @@ -1830,12 +1830,12 @@ extern long compat_ptr_ioctl(struct file *file, unsigned int cmd, /* * VFS file helper functions. */ -void inode_init_owner(struct mnt_idmap *idmap, struct inode *inode, +void inode_init_owner(const struct mnt_idmap *idmap, struct inode *inode, const struct inode *dir, umode_t mode); extern bool may_open_dev(const struct path *path); -umode_t mode_strip_sgid(struct mnt_idmap *idmap, +umode_t mode_strip_sgid(const struct mnt_idmap *idmap, const struct inode *dir, umode_t mode); -bool in_group_or_capable(struct mnt_idmap *idmap, +bool in_group_or_capable(const struct mnt_idmap *idmap, const struct inode *inode, vfsgid_t vfsgid); /* @@ -1994,26 +1994,26 @@ enum fs_update_time { struct inode_operations { struct dentry * (*lookup) (struct inode *,struct dentry *, unsigned int); const char * (*get_link) (struct dentry *, struct inode *, struct delayed_call *); - int (*permission) (struct mnt_idmap *, struct inode *, int); + int (*permission) (const struct mnt_idmap *, struct inode *, int); struct posix_acl * (*get_inode_acl)(struct inode *, int, bool); int (*readlink) (struct dentry *, char __user *,int); - int (*create) (struct mnt_idmap *, struct inode *,struct dentry *, + int (*create) (const struct mnt_idmap *, struct inode *,struct dentry *, umode_t); int (*link) (struct dentry *,struct inode *,struct dentry *); int (*unlink) (struct inode *,struct dentry *); - int (*symlink) (struct mnt_idmap *, struct inode *,struct dentry *, + int (*symlink) (const struct mnt_idmap *, struct inode *,struct dentry *, const char *); - struct dentry *(*mkdir) (struct mnt_idmap *, struct inode *, + struct dentry *(*mkdir) (const struct mnt_idmap *, struct inode *, struct dentry *, umode_t); int (*rmdir) (struct inode *,struct dentry *); - int (*mknod) (struct mnt_idmap *, struct inode *,struct dentry *, + int (*mknod) (const struct mnt_idmap *, struct inode *,struct dentry *, umode_t,dev_t); - int (*rename) (struct mnt_idmap *, struct inode *, struct dentry *, + int (*rename) (const struct mnt_idmap *, struct inode *, struct dentry *, struct inode *, struct dentry *, unsigned int); - int (*setattr) (struct mnt_idmap *, struct dentry *, struct iattr *); - int (*getattr) (struct mnt_idmap *, const struct path *, + int (*setattr) (const struct mnt_idmap *, struct dentry *, struct iattr *); + int (*getattr) (const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); ssize_t (*listxattr) (struct dentry *, char *, size_t); int (*fiemap)(struct inode *, struct fiemap_extent_info *, u64 start, @@ -2024,13 +2024,13 @@ struct inode_operations { int (*atomic_open)(struct inode *, struct dentry *, struct file *, unsigned open_flag, umode_t create_mode); - int (*tmpfile) (struct mnt_idmap *, struct inode *, + int (*tmpfile) (const struct mnt_idmap *, struct inode *, struct file *, umode_t); - struct posix_acl *(*get_acl)(struct mnt_idmap *, struct dentry *, + struct posix_acl *(*get_acl)(const struct mnt_idmap *, struct dentry *, int); - int (*set_acl)(struct mnt_idmap *, struct dentry *, + int (*set_acl)(const struct mnt_idmap *, struct dentry *, struct posix_acl *, int); - int (*fileattr_set)(struct mnt_idmap *idmap, + int (*fileattr_set)(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); int (*fileattr_get)(struct dentry *dentry, struct file_kattr *fa); struct offset_ctx *(*get_offset_ctx)(struct inode *inode); @@ -2173,7 +2173,7 @@ extern loff_t vfs_dedupe_file_range_one(struct file *src_file, loff_t src_pos, (inode)->i_rdev == WHITEOUT_DEV) #define IS_ANON_FILE(inode) ((inode)->i_flags & S_ANON_INODE) -static inline bool HAS_UNMAPPED_ID(struct mnt_idmap *idmap, +static inline bool HAS_UNMAPPED_ID(const struct mnt_idmap *idmap, struct inode *inode) { return !vfsuid_valid(i_uid_into_vfsuid(idmap, inode)) || @@ -2459,7 +2459,7 @@ struct filename { static_assert(offsetof(struct filename, iname) % sizeof(long) == 0); static_assert(sizeof(struct filename) % 64 == 0); -static inline struct mnt_idmap *file_mnt_idmap(const struct file *file) +static inline const struct mnt_idmap *file_mnt_idmap(const struct file *file) { return mnt_idmap(file->f_path.mnt); } @@ -2483,7 +2483,7 @@ static inline bool is_idmapped_mnt(const struct vfsmount *mnt) } int vfs_truncate(const struct path *, loff_t); -int do_truncate(struct mnt_idmap *, struct dentry *, loff_t start, +int do_truncate(const struct mnt_idmap *, struct dentry *, loff_t start, unsigned int time_attrs, struct file *filp); extern int vfs_fallocate(struct file *file, int mode, loff_t offset, loff_t len); @@ -2707,10 +2707,10 @@ static inline int bmap(struct inode *inode, sector_t *block) } #endif -int notify_change(struct mnt_idmap *, struct dentry *, +int notify_change(const struct mnt_idmap *, struct dentry *, struct iattr *, struct delegated_inode *); -int inode_permission(struct mnt_idmap *, struct inode *, int); -int generic_permission(struct mnt_idmap *, struct inode *, int); +int inode_permission(const struct mnt_idmap *, struct inode *, int); +int generic_permission(const struct mnt_idmap *, struct inode *, int); static inline int file_permission(struct file *file, int mask) { return inode_permission(file_mnt_idmap(file), @@ -2721,12 +2721,12 @@ static inline int path_permission(const struct path *path, int mask) return inode_permission(mnt_idmap(path->mnt), d_inode(path->dentry), mask); } -int __check_sticky(struct mnt_idmap *idmap, struct inode *dir, +int __check_sticky(const struct mnt_idmap *idmap, struct inode *dir, struct inode *inode); -int may_delete_dentry(struct mnt_idmap *idmap, struct inode *dir, +int may_delete_dentry(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *victim, bool isdir); -int may_create_dentry(struct mnt_idmap *idmap, +int may_create_dentry(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *child); static inline bool execute_ok(struct inode *inode) @@ -3045,9 +3045,9 @@ static inline struct inode *new_inode_pseudo(struct super_block *sb) } extern struct inode *new_inode(struct super_block *sb); extern void free_inode_nonrcu(struct inode *inode); -extern int setattr_should_drop_suidgid(struct mnt_idmap *, struct inode *); +extern int setattr_should_drop_suidgid(const struct mnt_idmap *, struct inode *); extern int file_remove_privs(struct file *); -int setattr_should_drop_sgid(struct mnt_idmap *idmap, +int setattr_should_drop_sgid(const struct mnt_idmap *idmap, const struct inode *inode); /* @@ -3204,7 +3204,7 @@ extern int page_symlink(struct inode *inode, const char *symname, int len); extern const struct inode_operations page_symlink_inode_operations; extern void kfree_link(void *); void fill_mg_cmtime(struct kstat *stat, u32 request_mask, struct inode *inode); -void generic_fillattr(struct mnt_idmap *, u32, struct inode *, struct kstat *); +void generic_fillattr(const struct mnt_idmap *, u32, struct inode *, struct kstat *); void generic_fill_statx_attr(struct inode *inode, struct kstat *stat); void generic_fill_statx_atomic_writes(struct kstat *stat, unsigned int unit_min, @@ -3261,9 +3261,9 @@ extern int dcache_dir_open(struct inode *, struct file *); extern int dcache_dir_close(struct inode *, struct file *); extern loff_t dcache_dir_lseek(struct file *, loff_t, int); extern int dcache_readdir(struct file *, struct dir_context *); -extern int simple_setattr(struct mnt_idmap *, struct dentry *, +extern int simple_setattr(const struct mnt_idmap *, struct dentry *, struct iattr *); -extern int simple_getattr(struct mnt_idmap *, const struct path *, +extern int simple_getattr(const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); extern int simple_statfs(struct dentry *, struct kstatfs *); extern int simple_open(struct inode *inode, struct file *file); @@ -3276,7 +3276,7 @@ void simple_rename_timestamp(struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry); extern int simple_rename_exchange(struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry); -extern int simple_rename(struct mnt_idmap *, struct inode *, +extern int simple_rename(const struct mnt_idmap *, struct inode *, struct dentry *, struct inode *, struct dentry *, unsigned int); extern void simple_recursive_removal(struct dentry *, @@ -3397,11 +3397,11 @@ static inline bool generic_ci_validate_strict_name(struct inode *dir, } #endif -int may_setattr(struct mnt_idmap *idmap, struct inode *inode, +int may_setattr(const struct mnt_idmap *idmap, struct inode *inode, unsigned int ia_valid); -int setattr_prepare(struct mnt_idmap *, struct dentry *, struct iattr *); +int setattr_prepare(const struct mnt_idmap *, struct dentry *, struct iattr *); extern int inode_newsize_ok(const struct inode *, loff_t offset); -void setattr_copy(struct mnt_idmap *, struct inode *inode, +void setattr_copy(const struct mnt_idmap *, struct inode *inode, const struct iattr *attr); extern int file_update_time(struct file *file); @@ -3578,7 +3578,7 @@ static inline bool is_sxid(umode_t mode) return mode & (S_ISUID | S_ISGID); } -static inline int check_sticky(struct mnt_idmap *idmap, +static inline int check_sticky(const struct mnt_idmap *idmap, struct inode *dir, struct inode *inode) { if (!(dir->i_mode & S_ISVTX)) @@ -3653,23 +3653,6 @@ extern int vfs_fadvise(struct file *file, loff_t offset, loff_t len, extern int generic_fadvise(struct file *file, loff_t offset, loff_t len, int advice); -static inline bool vfs_empty_path(int dfd, const char __user *path) -{ - char c; - - if (dfd < 0) - return false; - - /* We now allow NULL to be used for empty path. */ - if (!path) - return true; - - if (unlikely(get_user(c, path))) - return false; - - return !c; -} - int generic_atomic_write_valid(struct kiocb *iocb, struct iov_iter *iter); static inline bool extensible_ioctl_valid(unsigned int cmd_a, diff --git a/include/linux/fs_context.h b/include/linux/fs_context.h index 0d6c8a6d7be2..c920aba5177c 100644 --- a/include/linux/fs_context.h +++ b/include/linux/fs_context.h @@ -150,6 +150,10 @@ extern int vfs_parse_fs_param_source(struct fs_context *fc, struct fs_parameter *param); extern void fc_drop_locked(struct fs_context *fc); +extern int get_tree_super(struct fs_context *fc, + int (*test)(struct super_block *, struct fs_context *), + int (*fill_super)(struct super_block *sb, + struct fs_context *fc)); extern int get_tree_nodev(struct fs_context *fc, int (*fill_super)(struct super_block *sb, struct fs_context *fc)); diff --git a/include/linux/fscache-cache.h b/include/linux/fscache-cache.h index 4c91a019972b..ee524c863fa9 100644 --- a/include/linux/fscache-cache.h +++ b/include/linux/fscache-cache.h @@ -67,7 +67,7 @@ struct fscache_cache_ops { /* Change the size of a data object */ void (*resize_cookie)(struct netfs_cache_resources *cres, - loff_t new_size); + uoff_t new_size); /* Invalidate an object */ bool (*invalidate_cookie)(struct fscache_cookie *cookie); diff --git a/include/linux/fscache.h b/include/linux/fscache.h index 58fdb9605425..f2d958bd1f48 100644 --- a/include/linux/fscache.h +++ b/include/linux/fscache.h @@ -112,7 +112,7 @@ struct fscache_cookie { struct list_head proc_link; /* Link in proc list */ struct list_head commit_link; /* Link in commit queue */ struct work_struct work; /* Commit/relinq/withdraw work */ - loff_t object_size; /* Size of the netfs object */ + uoff_t object_size; /* Size of the netfs object */ unsigned long unused_at; /* Time at which unused (jiffies) */ unsigned long flags; #define FSCACHE_COOKIE_RELINQUISHED 0 /* T if cookie has been relinquished */ @@ -147,6 +147,23 @@ struct fscache_cookie { }; }; +enum fscache_extent_type { + FSCACHE_EXTENT_DATA, + FSCACHE_EXTENT_ZERO, +} __mode(byte); + +/* + * Cache occupancy information. + */ +struct fscache_occupancy { + unsigned long long query_from; /* Point to query from */ + unsigned long long query_to; /* Point to query to */ + unsigned long long cached_from[2]; /* Point at which cache extents start */ + unsigned long long cached_to[2]; /* Point at which cache extents end */ + unsigned int granularity; /* Granularity desired */ + enum fscache_extent_type cached_type[2]; /* Type of cache extent */ +}; + /* * slow-path functions for when there is actually caching available, and the * netfs does actually have a valid token @@ -163,22 +180,22 @@ extern struct fscache_cookie *__fscache_acquire_cookie( u8, const void *, size_t, const void *, size_t, - loff_t); + uoff_t); extern void __fscache_use_cookie(struct fscache_cookie *, bool); -extern void __fscache_unuse_cookie(struct fscache_cookie *, const void *, const loff_t *); +extern void __fscache_unuse_cookie(struct fscache_cookie *, const void *, const uoff_t *); extern void __fscache_relinquish_cookie(struct fscache_cookie *, bool); -extern void __fscache_resize_cookie(struct fscache_cookie *, loff_t); -extern void __fscache_invalidate(struct fscache_cookie *, const void *, loff_t, unsigned int); +extern void __fscache_resize_cookie(struct fscache_cookie *, uoff_t); +extern void __fscache_invalidate(struct fscache_cookie *, const void *, uoff_t, unsigned int); extern int __fscache_begin_read_operation(struct netfs_cache_resources *, struct fscache_cookie *); extern int __fscache_begin_write_operation(struct netfs_cache_resources *, struct fscache_cookie *); void __fscache_write_to_cache(struct fscache_cookie *cookie, struct address_space *mapping, - loff_t start, size_t len, loff_t i_size, + uoff_t start, size_t len, uoff_t i_size, netfs_io_terminated_t term_func, void *term_func_priv, bool using_pgpriv2, bool cond); -extern void __fscache_clear_page_bits(struct address_space *, loff_t, size_t); +extern void __fscache_clear_page_bits(struct address_space *, uoff_t, size_t); /** * fscache_acquire_volume - Register a volume as desiring caching services @@ -249,7 +266,7 @@ struct fscache_cookie *fscache_acquire_cookie(struct fscache_volume *volume, size_t index_key_len, const void *aux_data, size_t aux_data_len, - loff_t object_size) + uoff_t object_size) { if (!fscache_volume_valid(volume)) return NULL; @@ -286,7 +303,7 @@ static inline void fscache_use_cookie(struct fscache_cookie *cookie, */ static inline void fscache_unuse_cookie(struct fscache_cookie *cookie, const void *aux_data, - const loff_t *object_size) + const uoff_t *object_size) { if (fscache_cookie_valid(cookie)) __fscache_unuse_cookie(cookie, aux_data, object_size); @@ -327,7 +344,7 @@ static inline void *fscache_get_aux(struct fscache_cookie *cookie) */ static inline void fscache_update_aux(struct fscache_cookie *cookie, - const void *aux_data, const loff_t *object_size) + const void *aux_data, const uoff_t *object_size) { void *p = fscache_get_aux(cookie); @@ -343,7 +360,7 @@ extern atomic_t fscache_n_updates; static inline void __fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data, - const loff_t *object_size) + const uoff_t *object_size) { #ifdef CONFIG_FSCACHE_STATS atomic_inc(&fscache_n_updates); @@ -369,7 +386,7 @@ void __fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data */ static inline void fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data, - const loff_t *object_size) + const uoff_t *object_size) { if (fscache_cookie_enabled(cookie)) __fscache_update_cookie(cookie, aux_data, object_size); @@ -386,7 +403,7 @@ void fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data, * description. */ static inline -void fscache_resize_cookie(struct fscache_cookie *cookie, loff_t new_size) +void fscache_resize_cookie(struct fscache_cookie *cookie, uoff_t new_size) { if (fscache_cookie_enabled(cookie)) __fscache_resize_cookie(cookie, new_size); @@ -413,7 +430,7 @@ void fscache_resize_cookie(struct fscache_cookie *cookie, loff_t new_size) */ static inline void fscache_invalidate(struct fscache_cookie *cookie, - const void *aux_data, loff_t size, unsigned int flags) + const void *aux_data, uoff_t size, unsigned int flags) { if (fscache_cookie_enabled(cookie)) __fscache_invalidate(cookie, aux_data, size, flags); @@ -502,7 +519,7 @@ static inline void fscache_end_operation(struct netfs_cache_resources *cres) */ static inline int fscache_read(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, enum netfs_read_from_hole read_hole, netfs_io_terminated_t term_func, @@ -561,7 +578,7 @@ int fscache_begin_write_operation(struct netfs_cache_resources *cres, */ static inline int fscache_write(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, netfs_io_terminated_t term_func, void *term_func_priv) @@ -581,7 +598,7 @@ int fscache_write(struct netfs_cache_resources *cres, * waiting. */ static inline void fscache_clear_page_bits(struct address_space *mapping, - loff_t start, size_t len, + uoff_t start, size_t len, bool caching) { if (caching) @@ -615,7 +632,7 @@ static inline void fscache_clear_page_bits(struct address_space *mapping, */ static inline void fscache_write_to_cache(struct fscache_cookie *cookie, struct address_space *mapping, - loff_t start, size_t len, loff_t i_size, + uoff_t start, size_t len, uoff_t i_size, netfs_io_terminated_t term_func, void *term_func_priv, bool using_pgpriv2, bool caching) diff --git a/include/linux/iomap.h b/include/linux/iomap.h index bc7ae6327dbf..59718f73c15a 100644 --- a/include/linux/iomap.h +++ b/include/linux/iomap.h @@ -483,13 +483,35 @@ sector_t iomap_bmap(struct address_space *mapping, sector_t bno, #define IOMAP_IOEND_BOUNDARY (1U << 2) /* is direct I/O */ #define IOMAP_IOEND_DIRECT (1U << 3) +/* generate integrity (PI) information */ +#ifdef CONFIG_BLK_DEV_INTEGRITY +#define IOMAP_IOEND_INTEGRITY (1U << 4) +#else +#define IOMAP_IOEND_INTEGRITY 0 +#endif /* CONFIG_BLK_DEV_INTEGRITY */ /* * Flags that if set on either ioend prevent the merge of two ioends. * (IOMAP_IOEND_BOUNDARY also prevents merges, but only one-way) */ #define IOMAP_IOEND_NOMERGE_FLAGS \ - (IOMAP_IOEND_SHARED | IOMAP_IOEND_UNWRITTEN | IOMAP_IOEND_DIRECT) + (IOMAP_IOEND_SHARED | IOMAP_IOEND_UNWRITTEN | IOMAP_IOEND_DIRECT | \ + IOMAP_IOEND_INTEGRITY) + +/* ioend flags directly implied by iomap flags */ +static inline u16 iomap_ioend_flags(const struct iomap *iomap) +{ + unsigned int flags = 0; + + if (iomap->type == IOMAP_UNWRITTEN) + flags |= IOMAP_IOEND_UNWRITTEN; + if (iomap->flags & IOMAP_F_SHARED) + flags |= IOMAP_IOEND_SHARED; + if (iomap->flags & IOMAP_F_INTEGRITY) + flags |= IOMAP_IOEND_INTEGRITY; + + return flags; +} /* * Structure for writeback I/O completions. @@ -500,6 +522,7 @@ sector_t iomap_bmap(struct address_space *mapping, sector_t bno, struct iomap_ioend { struct list_head io_list; /* next ioend in chain */ u16 io_flags; /* IOMAP_IOEND_* */ + u32 io_bvec_offset; /* offset into first bvec */ struct inode *io_inode; /* file being written to */ size_t io_size; /* size of the extent */ atomic_t io_remaining; /* completetion defer count */ @@ -517,6 +540,13 @@ static inline struct iomap_ioend *iomap_ioend_from_bio(struct bio *bio) return container_of(bio, struct iomap_ioend, io_bio); } +#define BVEC_ITER_IOEND(_ioend) \ +{ \ + .bi_sector = (_ioend)->io_sector, \ + .bi_size = (_ioend)->io_size, \ + .bi_offset = (_ioend)->io_bvec_offset, \ +} + struct iomap_writeback_ops { /* * Performs writeback on the passed in range @@ -565,6 +595,7 @@ void iomap_finish_ioends(struct iomap_ioend *ioend, int error); void iomap_ioend_try_merge(struct iomap_ioend *ioend, struct list_head *more_ioends); void iomap_sort_ioends(struct list_head *ioend_list); +int iomap_ioend_integrity_verify(struct iomap_ioend *ioend); ssize_t iomap_add_to_ioend(struct iomap_writepage_ctx *wpc, struct folio *folio, loff_t pos, loff_t end_pos, unsigned int dirty_len); int iomap_ioend_writeback_submit(struct iomap_writepage_ctx *wpc, int error); @@ -577,6 +608,11 @@ void iomap_finish_folio_write(struct inode *inode, struct folio *folio, int iomap_writeback_folio(struct iomap_writepage_ctx *wpc, struct folio *folio); int iomap_writepages(struct iomap_writepage_ctx *wpc); +void iomap_bounce_read(struct iomap_ioend *orig_ioend, unsigned int minsize, + void (*submit_ioend)(struct iomap_ioend *ioend)); +void iomap_bounce_read_end_io(struct iomap_ioend *ioend, struct bio *orig_bio, + int error); + struct iomap_read_folio_ctx { const struct iomap_read_ops *ops; struct folio *cur_folio; diff --git a/include/linux/lsm_hook_defs.h b/include/linux/lsm_hook_defs.h index 65c9609ec207..c9561564585e 100644 --- a/include/linux/lsm_hook_defs.h +++ b/include/linux/lsm_hook_defs.h @@ -36,6 +36,7 @@ LSM_HOOK(int, 0, binder_transfer_file, const struct cred *from, LSM_HOOK(int, 0, ptrace_access_check, struct task_struct *child, unsigned int mode) LSM_HOOK(int, 0, ptrace_traceme, struct task_struct *parent) +LSM_HOOK(int, 0, mem_foll_force, const struct cred *subject, bool opened_by_owner) LSM_HOOK(int, 0, capget, const struct task_struct *target, kernel_cap_t *effective, kernel_cap_t *inheritable, kernel_cap_t *permitted) LSM_HOOK(int, 0, capset, struct cred *new, const struct cred *old, @@ -94,7 +95,7 @@ LSM_HOOK(int, 0, path_mkdir, const struct path *dir, struct dentry *dentry, LSM_HOOK(int, 0, path_rmdir, const struct path *dir, struct dentry *dentry) LSM_HOOK(int, 0, path_mknod, const struct path *dir, struct dentry *dentry, umode_t mode, unsigned int dev) -LSM_HOOK(void, LSM_RET_VOID, path_post_mknod, struct mnt_idmap *idmap, +LSM_HOOK(void, LSM_RET_VOID, path_post_mknod, const struct mnt_idmap *idmap, struct dentry *dentry) LSM_HOOK(int, 0, path_truncate, const struct path *path) LSM_HOOK(int, 0, path_symlink, const struct path *dir, struct dentry *dentry, @@ -122,7 +123,7 @@ LSM_HOOK(int, 0, inode_init_security_anon, struct inode *inode, const struct qstr *name, const struct inode *context_inode) LSM_HOOK(int, 0, inode_create, struct inode *dir, struct dentry *dentry, umode_t mode) -LSM_HOOK(void, LSM_RET_VOID, inode_post_create_tmpfile, struct mnt_idmap *idmap, +LSM_HOOK(void, LSM_RET_VOID, inode_post_create_tmpfile, const struct mnt_idmap *idmap, struct inode *inode) LSM_HOOK(int, 0, inode_link, struct dentry *old_dentry, struct inode *dir, struct dentry *new_dentry) @@ -140,39 +141,39 @@ LSM_HOOK(int, 0, inode_readlink, struct dentry *dentry) LSM_HOOK(int, 0, inode_follow_link, struct dentry *dentry, struct inode *inode, bool rcu) LSM_HOOK(int, 0, inode_permission, struct inode *inode, int mask) -LSM_HOOK(int, 0, inode_setattr, struct mnt_idmap *idmap, struct dentry *dentry, +LSM_HOOK(int, 0, inode_setattr, const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) -LSM_HOOK(void, LSM_RET_VOID, inode_post_setattr, struct mnt_idmap *idmap, +LSM_HOOK(void, LSM_RET_VOID, inode_post_setattr, const struct mnt_idmap *idmap, struct dentry *dentry, int ia_valid) LSM_HOOK(int, 0, inode_getattr, const struct path *path) LSM_HOOK(int, 0, inode_xattr_skipcap, const char *name) -LSM_HOOK(int, 0, inode_setxattr, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_setxattr, const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, const void *value, size_t size, int flags) LSM_HOOK(void, LSM_RET_VOID, inode_post_setxattr, struct dentry *dentry, const char *name, const void *value, size_t size, int flags) LSM_HOOK(int, 0, inode_getxattr, struct dentry *dentry, const char *name) LSM_HOOK(int, 0, inode_listxattr, struct dentry *dentry) -LSM_HOOK(int, 0, inode_removexattr, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_removexattr, const struct mnt_idmap *idmap, struct dentry *dentry, const char *name) LSM_HOOK(void, LSM_RET_VOID, inode_post_removexattr, struct dentry *dentry, const char *name) LSM_HOOK(int, 0, inode_file_setattr, struct dentry *dentry, struct file_kattr *fa) LSM_HOOK(int, 0, inode_file_getattr, struct dentry *dentry, struct file_kattr *fa) -LSM_HOOK(int, 0, inode_set_acl, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_set_acl, const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl) LSM_HOOK(void, LSM_RET_VOID, inode_post_set_acl, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl) -LSM_HOOK(int, 0, inode_get_acl, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_get_acl, const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) -LSM_HOOK(int, 0, inode_remove_acl, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_remove_acl, const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) -LSM_HOOK(void, LSM_RET_VOID, inode_post_remove_acl, struct mnt_idmap *idmap, +LSM_HOOK(void, LSM_RET_VOID, inode_post_remove_acl, const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) LSM_HOOK(int, 0, inode_need_killpriv, struct dentry *dentry) -LSM_HOOK(int, 0, inode_killpriv, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_killpriv, const struct mnt_idmap *idmap, struct dentry *dentry) -LSM_HOOK(int, -EOPNOTSUPP, inode_getsecurity, struct mnt_idmap *idmap, +LSM_HOOK(int, -EOPNOTSUPP, inode_getsecurity, const struct mnt_idmap *idmap, struct inode *inode, const char *name, void **buffer, bool alloc) LSM_HOOK(int, -EOPNOTSUPP, inode_setsecurity, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/include/linux/mnt_idmapping.h b/include/linux/mnt_idmapping.h index e71a6070a8f8..78eeef4c2996 100644 --- a/include/linux/mnt_idmapping.h +++ b/include/linux/mnt_idmapping.h @@ -8,8 +8,8 @@ struct mnt_idmap; struct user_namespace; -extern struct mnt_idmap nop_mnt_idmap; -extern struct mnt_idmap invalid_mnt_idmap; +extern const struct mnt_idmap nop_mnt_idmap; +extern const struct mnt_idmap invalid_mnt_idmap; extern struct user_namespace init_user_ns; typedef struct { @@ -121,19 +121,19 @@ static inline bool vfsgid_eq_kgid(vfsgid_t vfsgid, kgid_t kgid) int vfsgid_in_group_p(vfsgid_t vfsgid); -struct mnt_idmap *mnt_idmap_get(struct mnt_idmap *idmap); -void mnt_idmap_put(struct mnt_idmap *idmap); +const struct mnt_idmap *mnt_idmap_get(const struct mnt_idmap *idmap); +void mnt_idmap_put(const struct mnt_idmap *idmap); -vfsuid_t make_vfsuid(struct mnt_idmap *idmap, +vfsuid_t make_vfsuid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, kuid_t kuid); -vfsgid_t make_vfsgid(struct mnt_idmap *idmap, +vfsgid_t make_vfsgid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, kgid_t kgid); -kuid_t from_vfsuid(struct mnt_idmap *idmap, +kuid_t from_vfsuid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, vfsuid_t vfsuid); -kgid_t from_vfsgid(struct mnt_idmap *idmap, +kgid_t from_vfsgid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, vfsgid_t vfsgid); /** @@ -148,7 +148,7 @@ kgid_t from_vfsgid(struct mnt_idmap *idmap, * * Return: true if @vfsuid has a mapping in the filesystem, false if not. */ -static inline bool vfsuid_has_fsmapping(struct mnt_idmap *idmap, +static inline bool vfsuid_has_fsmapping(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, vfsuid_t vfsuid) { @@ -186,7 +186,7 @@ static inline kuid_t vfsuid_into_kuid(vfsuid_t vfsuid) * * Return: true if @vfsgid has a mapping in the filesystem, false if not. */ -static inline bool vfsgid_has_fsmapping(struct mnt_idmap *idmap, +static inline bool vfsgid_has_fsmapping(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, vfsgid_t vfsgid) { @@ -225,7 +225,7 @@ static inline kgid_t vfsgid_into_kgid(vfsgid_t vfsgid) * * Return: the caller's current fsuid mapped up according to @idmap. */ -static inline kuid_t mapped_fsuid(struct mnt_idmap *idmap, +static inline kuid_t mapped_fsuid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns) { return from_vfsuid(idmap, fs_userns, VFSUIDT_INIT(current_fsuid())); @@ -244,7 +244,7 @@ static inline kuid_t mapped_fsuid(struct mnt_idmap *idmap, * * Return: the caller's current fsgid mapped up according to @idmap. */ -static inline kgid_t mapped_fsgid(struct mnt_idmap *idmap, +static inline kgid_t mapped_fsgid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns) { return from_vfsgid(idmap, fs_userns, VFSGIDT_INIT(current_fsgid())); diff --git a/include/linux/mount.h b/include/linux/mount.h index acfe7ef86a1b..e90ccafef281 100644 --- a/include/linux/mount.h +++ b/include/linux/mount.h @@ -59,10 +59,10 @@ struct vfsmount { struct dentry *mnt_root; /* root of the mounted tree */ struct super_block *mnt_sb; /* pointer to superblock */ int mnt_flags; - struct mnt_idmap *mnt_idmap; + const struct mnt_idmap *mnt_idmap; } __randomize_layout; -static inline struct mnt_idmap *mnt_idmap(const struct vfsmount *mnt) +static inline const struct mnt_idmap *mnt_idmap(const struct vfsmount *mnt) { /* Pairs with smp_store_release() in do_idmap_mount(). */ return READ_ONCE(mnt->mnt_idmap); diff --git a/include/linux/namei.h b/include/linux/namei.h index 86d657b24fc6..c4436e5c2ba6 100644 --- a/include/linux/namei.h +++ b/include/linux/namei.h @@ -32,8 +32,9 @@ enum { MAX_NESTED_LINKS = 8 }; #define LOOKUP_CREATE BIT(17) /* ... in object creation */ #define LOOKUP_EXCL BIT(18) /* ... in target must not exist */ #define LOOKUP_RENAME_TARGET BIT(19) /* ... in destination of rename() */ +#define LOOKUP_SHARED BIT(20) /* Parent lock is held shared */ -/* 4 spare bits for intent */ +/* 3 spare bits for intent */ /* Scoping flags for lookup. */ #define LOOKUP_NO_SYMLINKS BIT(24) /* No symlink crossing. */ @@ -70,24 +71,24 @@ extern struct dentry *try_lookup_noperm(struct qstr *, struct dentry *); extern struct dentry *lookup_noperm(struct qstr *, struct dentry *); extern struct dentry *lookup_noperm_unlocked(struct qstr *, struct dentry *); extern struct dentry *lookup_noperm_positive_unlocked(struct qstr *, struct dentry *); -struct dentry *lookup_one(struct mnt_idmap *, struct qstr *, struct dentry *); -struct dentry *lookup_one_unlocked(struct mnt_idmap *idmap, +struct dentry *lookup_one(const struct mnt_idmap *, struct qstr *, struct dentry *); +struct dentry *lookup_one_unlocked(const struct mnt_idmap *idmap, struct qstr *name, struct dentry *base); -struct dentry *lookup_one_positive_unlocked(struct mnt_idmap *idmap, +struct dentry *lookup_one_positive_unlocked(const struct mnt_idmap *idmap, struct qstr *name, struct dentry *base); -struct dentry *lookup_one_positive_killable(struct mnt_idmap *idmap, +struct dentry *lookup_one_positive_killable(const struct mnt_idmap *idmap, struct qstr *name, struct dentry *base); -struct dentry *start_creating(struct mnt_idmap *idmap, struct dentry *parent, +struct dentry *start_creating(const struct mnt_idmap *idmap, struct dentry *parent, struct qstr *name); -struct dentry *start_removing(struct mnt_idmap *idmap, struct dentry *parent, +struct dentry *start_removing(const struct mnt_idmap *idmap, struct dentry *parent, struct qstr *name); -struct dentry *start_creating_killable(struct mnt_idmap *idmap, +struct dentry *start_creating_killable(const struct mnt_idmap *idmap, struct dentry *parent, struct qstr *name); -struct dentry *start_removing_killable(struct mnt_idmap *idmap, +struct dentry *start_removing_killable(const struct mnt_idmap *idmap, struct dentry *parent, struct qstr *name); struct dentry *start_creating_noperm(struct dentry *parent, struct qstr *name); diff --git a/include/linux/netfs.h b/include/linux/netfs.h index b4dd32863dd4..67e010b6994b 100644 --- a/include/linux/netfs.h +++ b/include/linux/netfs.h @@ -22,6 +22,7 @@ enum netfs_sreq_ref_trace; typedef struct mempool mempool_t; +struct fscache_occupancy; struct folio_queue; /** @@ -62,8 +63,8 @@ struct netfs_inode { struct fscache_cookie *cache; #endif struct list_head wb_queue; /* Queue of processes wanting to do writeback */ - loff_t _remote_i_size; /* Size of the remote file */ - loff_t _zero_point; /* Size after which we assume there's no data + uoff_t _remote_i_size; /* Size of the remote file */ + uoff_t _zero_point; /* Size after which we assume there's no data * on the server */ spinlock_t lock; /* Lock covering wb_queue */ atomic_t io_count; /* Number of outstanding reqs */ @@ -125,6 +126,12 @@ static inline struct netfs_group *netfs_folio_group(struct folio *folio) return priv; } +enum netfs_cache_collect { + NETFS_CACHE_COLLECT_WRITE_GAP, /* Gap in collection, no state either way */ + NETFS_CACHE_COLLECT_WRITE_DATA, /* Currently collecting good writes */ + NETFS_CACHE_COLLECT_WRITE_CANCEL, /* Currently collecting cancelled writes */ +}; + /* * Stream of I/O subrequests going to a particular destination, such as the * server or the local cache. This is mainly intended for writing where we may @@ -142,7 +149,7 @@ struct netfs_io_stream { void (*issue_write)(struct netfs_io_subrequest *subreq); /* Collection tracking */ struct list_head subrequests; /* Contributory I/O operations */ - unsigned long long collected_to; /* Position we've collected results to */ + uoff_t collected_to; /* Position we've collected results to */ size_t transferred; /* The amount transferred from this stream */ unsigned short error; /* Aggregate error for the stream */ enum netfs_io_source source; /* Where to read from/write to */ @@ -152,6 +159,7 @@ struct netfs_io_stream { bool need_retry; /* T if this stream needs retrying */ bool failed; /* T if this stream failed */ bool transferred_valid; /* T is ->transferred is valid */ + enum netfs_cache_collect cache_collect; /* Current writeback cache collect state */ }; /* @@ -161,8 +169,11 @@ struct netfs_cache_resources { const struct netfs_cache_ops *ops; void *cache_priv; void *cache_priv2; - unsigned int debug_id; /* Cookie debug ID */ + uoff_t cache_i_size; /* Initial size of cache file */ + unsigned int cookie_id; /* Cache cookie debug ID */ + unsigned int object_id; /* Cache object debug ID */ unsigned int inval_counter; /* object->inval_counter at begin_op */ + unsigned int dio_size; /* DIO block size */ }; /* @@ -177,7 +188,7 @@ struct netfs_io_subrequest { struct work_struct work; struct list_head rreq_link; /* Link in rreq->subrequests */ struct iov_iter io_iter; /* Iterator for this subrequest */ - unsigned long long start; /* Where to start the I/O */ + uoff_t start; /* Where to start the I/O */ size_t len; /* Size of the I/O */ size_t transferred; /* Amount of data transferred */ refcount_t ref; @@ -196,6 +207,7 @@ struct netfs_io_subrequest { #define NETFS_SREQ_IN_PROGRESS 8 /* Unlocked when the subrequest completes */ #define NETFS_SREQ_NEED_RETRY 9 /* Set if the filesystem requests a retry */ #define NETFS_SREQ_FAILED 10 /* Set if the subreq failed unretryably */ +#define NETFS_SREQ_CANCELLED 11 /* Set if the subreq was cancelled by netfslib */ }; enum netfs_io_origin { @@ -208,7 +220,6 @@ enum netfs_io_origin { NETFS_DIO_READ, /* This is a direct I/O read */ NETFS_WRITEBACK, /* This write was triggered by writepages */ NETFS_WRITEBACK_SINGLE, /* This monolithic write was triggered by writepages */ - NETFS_WRITETHROUGH, /* This write was made by netfs_perform_write() */ NETFS_UNBUFFERED_WRITE, /* This is an unbuffered write */ NETFS_DIO_WRITE, /* This is a direct I/O write */ NETFS_PGPRIV2_COPY_TO_CACHE, /* [DEPRECATED] This is writing read data to the cache */ @@ -243,17 +254,18 @@ struct netfs_io_request { void *netfs_priv; /* Private data for the netfs */ void *netfs_priv2; /* Private data for the netfs */ struct bio_vec *direct_bv; /* DIO buffer list (when handling iovec-iter) */ - unsigned long long submitted; /* Amount submitted for I/O so far */ - unsigned long long len; /* Length of the request */ + uoff_t submitted; /* Amount submitted for I/O so far */ + uoff_t len; /* Length of the request */ size_t transferred; /* Amount to be indicated as transferred */ size_t progress_at; /* Report read progress when hit this much read */ long error; /* 0 or error that occurred */ - unsigned long long i_size; /* Size of the file */ - unsigned long long start; /* Start position */ + uoff_t i_size; /* Size of the file */ + uoff_t start; /* Start position */ atomic64_t issued_to; /* Write issuer folio cursor */ - unsigned long long collected_to; /* Point we've collected to */ - unsigned long long cleaned_to; /* Position we've cleaned folios to */ - unsigned long long abandon_to; /* Position to abandon folios to */ + uoff_t collected_to; /* Point we've collected to */ + uoff_t cache_coll_to; /* Point the cache has collected to */ + uoff_t cleaned_to; /* Position we've cleaned folios to */ + uoff_t abandon_to; /* Position to abandon folios to */ const struct folio *no_unlock_folio; /* Don't unlock this folio after read */ gfp_t gfp; /* GFP flags to use */ unsigned int direct_bv_count; /* Number of elements in direct_bv[] */ @@ -273,14 +285,18 @@ struct netfs_io_request { #define NETFS_RREQ_FAILED 3 /* The request failed */ #define NETFS_RREQ_RETRYING 4 /* Set if we're in the retry path */ #define NETFS_RREQ_SHORT_TRANSFER 5 /* Set if we have a short transfer */ -#define NETFS_RREQ_OFFLOAD_COLLECTION 8 /* Offload collection to workqueue */ -#define NETFS_RREQ_NO_UNLOCK_FOLIO 9 /* Don't unlock no_unlock_folio on completion */ +#define NETFS_RREQ_CACHE_STOP 8 /* Set to stop caching (ENOBUFS or error) */ +#define NETFS_RREQ_CACHE_ERROR 9 /* Set if we got an error from the cache */ #define NETFS_RREQ_CANCEL_CACHING 10 /* Set to cancel caching */ -#define NETFS_RREQ_UPLOAD_TO_SERVER 11 /* Need to write to the server */ -#define NETFS_RREQ_USE_IO_ITER 12 /* Use ->io_iter rather than ->i_pages */ +#define NETFS_RREQ_OFFLOAD_COLLECTION 12 /* Offload collection to workqueue */ +#define NETFS_RREQ_NO_UNLOCK_FOLIO 13 /* Don't unlock no_unlock_folio on completion */ +#define NETFS_RREQ_UPLOAD_TO_SERVER 14 /* Need to write to the server */ +#define NETFS_RREQ_USE_IO_ITER 15 /* Use ->io_iter rather than ->i_pages */ #define NETFS_RREQ_NEED_PUT_RA_REFS 17 /* Need to put the folio refs RA gave us */ +#ifdef CONFIG_NETFS_PGPRIV2 #define NETFS_RREQ_USE_PGPRIV2 31 /* [DEPRECATED] Use PG_private_2 to mark * write to cache on read */ +#endif const struct netfs_request_ops *netfs_ops; }; @@ -299,12 +315,12 @@ struct netfs_request_ops { int (*prepare_read)(struct netfs_io_subrequest *subreq); void (*issue_read)(struct netfs_io_subrequest *subreq); bool (*is_still_valid)(struct netfs_io_request *rreq); - int (*check_write_begin)(struct file *file, loff_t pos, unsigned len, + int (*check_write_begin)(struct file *file, uoff_t pos, unsigned len, struct folio **foliop, void **_fsdata); void (*done)(struct netfs_io_request *rreq); /* Modification handling */ - void (*update_i_size)(struct inode *inode, loff_t i_size); + void (*update_i_size)(struct inode *inode, uoff_t i_size); void (*post_modify)(struct inode *inode); /* Write request handling */ @@ -332,7 +348,7 @@ struct netfs_cache_ops { /* Read data from the cache */ int (*read)(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, enum netfs_read_from_hole read_hole, netfs_io_terminated_t term_func, @@ -340,7 +356,7 @@ struct netfs_cache_ops { /* Write data to the cache */ int (*write)(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, netfs_io_terminated_t term_func, void *term_func_priv); @@ -350,15 +366,14 @@ struct netfs_cache_ops { /* Expand readahead request */ void (*expand_readahead)(struct netfs_cache_resources *cres, - unsigned long long *_start, - unsigned long long *_len, - unsigned long long i_size); + uoff_t *_start, + uoff_t *_len, + uoff_t i_size); /* Prepare a read operation, shortening it to a cached/uncached * boundary as appropriate. */ - enum netfs_io_source (*prepare_read)(struct netfs_io_subrequest *subreq, - unsigned long long i_size); + int (*prepare_read)(struct netfs_io_subrequest *subreq); /* Prepare a write subrequest, working out if we're allowed to do it * and finding out the maximum amount of data to gather before @@ -371,15 +386,24 @@ struct netfs_cache_ops { * actually do. */ int (*prepare_write)(struct netfs_cache_resources *cres, - loff_t *_start, size_t *_len, size_t upper_len, - loff_t i_size, bool no_space_allocated_yet); + uoff_t *_start, size_t *_len, size_t upper_len, + uoff_t i_size, bool no_space_allocated_yet); /* Query the occupancy of the cache in a region, returning where the * next chunk of data starts and how long it is. */ int (*query_occupancy)(struct netfs_cache_resources *cres, - loff_t start, size_t len, size_t granularity, - loff_t *_data_start, size_t *_data_len); + struct fscache_occupancy *occ); + + /* Collect the result of buffered writeback to the cache. This + * includes copying a read to the cache. block_type is one of: + * - NETFS_CACHE_COLLECT_WRITE_DATA for a block of data + * - NETFS_CACHE_COLLECT_WRITE_GAP if a discontiguity was skipped + * - NETFS_CACHE_COLLECT_WRITE_CANCEL for a cancellation gap + */ + void (*collect_write)(struct netfs_io_request *wreq, + uoff_t start, size_t len, + enum netfs_cache_collect block_type); }; /* High-level read API. */ @@ -410,7 +434,7 @@ struct readahead_control; void netfs_readahead(struct readahead_control *); int netfs_read_folio(struct file *, struct folio *); int netfs_write_begin(struct netfs_inode *, struct file *, - struct address_space *, loff_t pos, unsigned int len, + struct address_space *, uoff_t pos, unsigned int len, struct folio **, void **fsdata); int netfs_writepages(struct address_space *mapping, struct writeback_control *wbc); @@ -488,10 +512,10 @@ static inline struct netfs_inode *netfs_inode(struct inode *inode) * cmpxchg8b without the need of the lock prefix). For SMP compiles and 64bit * archs it makes no difference if preempt is enabled or not. */ -static inline unsigned long long netfs_read_remote_i_size(const struct inode *inode) +static inline uoff_t netfs_read_remote_i_size(const struct inode *inode) { const struct netfs_inode *ictx = container_of(inode, struct netfs_inode, inode); - unsigned long long remote_i_size; + uoff_t remote_i_size; #if BITS_PER_LONG==32 && defined(CONFIG_SMP) unsigned int seq; @@ -526,7 +550,7 @@ static inline unsigned long long netfs_read_remote_i_size(const struct inode *in * spinning forever. */ static inline void netfs_write_remote_i_size(struct inode *inode, - unsigned long long remote_i_size) + uoff_t remote_i_size) { struct netfs_inode *ictx = netfs_inode(inode); @@ -563,10 +587,10 @@ static inline void netfs_write_remote_i_size(struct inode *inode, * cmpxchg8b without the need of the lock prefix). For SMP compiles and 64bit * archs it makes no difference if preempt is enabled or not. */ -static inline unsigned long long netfs_read_zero_point(const struct inode *inode) +static inline uoff_t netfs_read_zero_point(const struct inode *inode) { struct netfs_inode *ictx = container_of(inode, struct netfs_inode, inode); - unsigned long long zero_point; + uoff_t zero_point; #if BITS_PER_LONG==32 && defined(CONFIG_SMP) unsigned int seq; @@ -601,7 +625,7 @@ static inline unsigned long long netfs_read_zero_point(const struct inode *inode * forever. */ static inline void netfs_write_zero_point(struct inode *inode, - unsigned long long zero_point) + uoff_t zero_point) { struct netfs_inode *ictx = netfs_inode(inode); @@ -642,9 +666,9 @@ static inline void netfs_write_zero_point(struct inode *inode, * archs it makes no difference if preempt is enabled or not. */ static inline void netfs_read_sizes(const struct inode *inode, - unsigned long long *i_size, - unsigned long long *remote_i_size, - unsigned long long *zero_point) + uoff_t *i_size, + uoff_t *remote_i_size, + uoff_t *zero_point) { const struct netfs_inode *ictx = container_of(inode, struct netfs_inode, inode); #if BITS_PER_LONG==32 && defined(CONFIG_SMP) @@ -690,9 +714,9 @@ static inline void netfs_read_sizes(const struct inode *inode, * forever. */ static inline void netfs_write_sizes(struct inode *inode, - unsigned long long i_size, - unsigned long long remote_i_size, - unsigned long long zero_point) + uoff_t i_size, + uoff_t remote_i_size, + uoff_t zero_point) { struct netfs_inode *ictx = netfs_inode(inode); @@ -760,7 +784,7 @@ static inline void netfs_inode_init(struct netfs_inode *ctx, * Inform the netfs lib that a file got resized so that it can adjust its state. */ static inline void netfs_resize_file(struct netfs_inode *ictx, - unsigned long long new_i_size, + uoff_t new_i_size, bool changed_on_server) { #if BITS_PER_LONG==32 && defined(CONFIG_SMP) diff --git a/include/linux/nfs.h b/include/linux/nfs.h index 0906a0b40c6a..8c2818db43c5 100644 --- a/include/linux/nfs.h +++ b/include/linux/nfs.h @@ -11,59 +11,8 @@ #include <linux/cred.h> #include <linux/sunrpc/auth.h> #include <linux/sunrpc/msg_prot.h> -#include <linux/string.h> -#include <linux/crc32.h> -#include <uapi/linux/nfs.h> - -/* The LOCALIO program is entirely private to Linux and is - * NOT part of the uapi. - */ -#define NFS_LOCALIO_PROGRAM 400122 -#define LOCALIOPROC_NULL 0 -#define LOCALIOPROC_UUID_IS_LOCAL 1 - -/* - * This is the kernel NFS client file handle representation - */ -#define NFS_MAXFHSIZE 128 -struct nfs_fh { - unsigned short size; - unsigned char data[NFS_MAXFHSIZE]; -}; - -/* - * Returns a zero iff the size and data fields match. - * Checks only "size" bytes in the data field. - */ -static inline int nfs_compare_fh(const struct nfs_fh *a, const struct nfs_fh *b) -{ - return a->size != b->size || memcmp(a->data, b->data, a->size) != 0; -} - -static inline void nfs_copy_fh(struct nfs_fh *target, const struct nfs_fh *source) -{ - target->size = source->size; - memcpy(target->data, source->data, source->size); -} - -enum nfs3_stable_how { - NFS_UNSTABLE = 0, - NFS_DATA_SYNC = 1, - NFS_FILE_SYNC = 2, +#include <linux/nfs_fh.h> - /* used by direct.c to mark verf as invalid */ - NFS_INVALID_STABLE_HOW = -1 -}; +#include <uapi/linux/nfs.h> -/** - * nfs_fhandle_hash - calculate the crc32 hash for the filehandle - * @fh - pointer to filehandle - * - * returns a crc32 hash for the filehandle that is compatible with - * the one displayed by "wireshark". - */ -static inline u32 nfs_fhandle_hash(const struct nfs_fh *fh) -{ - return ~crc32_le(0xFFFFFFFF, &fh->data[0], fh->size); -} #endif /* _LINUX_NFS_H */ diff --git a/include/linux/nfs3.h b/include/linux/nfs3.h index 404b8f724fc9..b6539a75edea 100644 --- a/include/linux/nfs3.h +++ b/include/linux/nfs3.h @@ -7,6 +7,49 @@ #include <uapi/linux/nfs3.h> +/* + * NFSv3 error status values. + * See RFC 1813 Section 2.5 + */ +enum { + NFS3ERR_PERM = 1, + NFS3ERR_NOENT = 2, + NFS3ERR_IO = 5, + NFS3ERR_NXIO = 6, + NFS3ERR_ACCES = 13, + NFS3ERR_EXIST = 17, + NFS3ERR_XDEV = 18, + NFS3ERR_NODEV = 19, + NFS3ERR_NOTDIR = 20, + NFS3ERR_ISDIR = 21, + NFS3ERR_INVAL = 22, + NFS3ERR_FBIG = 27, + NFS3ERR_NOSPC = 28, + NFS3ERR_ROFS = 30, + NFS3ERR_MLINK = 31, + NFS3ERR_NAMETOOLONG = 63, + NFS3ERR_NOTEMPTY = 66, + NFS3ERR_DQUOT = 69, + NFS3ERR_STALE = 70, + NFS3ERR_REMOTE = 71, + NFS3ERR_BADHANDLE = 10001, + NFS3ERR_NOT_SYNC = 10002, + NFS3ERR_BAD_COOKIE = 10003, + NFS3ERR_NOTSUPP = 10004, + NFS3ERR_TOOSMALL = 10005, + NFS3ERR_SERVERFAULT = 10006, + NFS3ERR_BADTYPE = 10007, + NFS3ERR_JUKEBOX = 10008, +}; + +enum nfs3_stable_how { + NFS_UNSTABLE = 0, + NFS_DATA_SYNC = 1, + NFS_FILE_SYNC = 2, + + /* used to mark verf as invalid */ + NFS_INVALID_STABLE_HOW = -1 +}; /* Number of 32bit words in post_op_attr */ #define NFS3_POST_OP_ATTR_WORDS 22 diff --git a/include/linux/nfs4.h b/include/linux/nfs4.h index 1a3981c26b23..41b7cdcc674f 100644 --- a/include/linux/nfs4.h +++ b/include/linux/nfs4.h @@ -263,6 +263,12 @@ enum why_no_delegation4 { /* new to v4.1 */ WND4_IS_DIR = 8, }; +enum stable_how4 { + UNSTABLE4 = 0, + DATA_SYNC4 = 1, + FILE_SYNC4 = 2, +}; + enum lock_type4 { NFS4_UNLOCK_LT = 0, NFS4_READ_LT = 1, diff --git a/include/linux/nfs_fh.h b/include/linux/nfs_fh.h new file mode 100644 index 000000000000..49dfc5ec60fe --- /dev/null +++ b/include/linux/nfs_fh.h @@ -0,0 +1,63 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +/* + * struct nfs_fh is an NFS version-agnostic data structure that + * stores an NFS file handle. It is also commonly used in NFS + * related APIs. + */ +#ifndef _LINUX_NFS_FH_H +#define _LINUX_NFS_FH_H + +#include <linux/types.h> +#include <linux/string.h> +#include <linux/crc32.h> + +/* + * The largest file handle size today is an NFSv4 file handle, + * which can be up to 128 octets long. + */ +#define NFS_MAXFHSIZE 128 +struct nfs_fh { + unsigned short size; + unsigned char data[NFS_MAXFHSIZE]; +}; + +/** + * nfs_compare_fh - Compare two NFS file handles + * @a: An NFS file handle to be compared + * @b: An NFS file handle to be compared + * + * Checks only "size" bytes in each data field. + * + * Return: %false if the two file handles are equal, otherwise %true + */ +static inline bool nfs_compare_fh(const struct nfs_fh *a, const struct nfs_fh *b) +{ + return a->size != b->size || memcmp(a->data, b->data, a->size) != 0; +} + +/** + * nfs_copy_fh - Copy an NFS file handle + * @target: Destination file handle + * @source: Source file handle + * + * Copies source->size bytes of file handle data into target. + */ +static inline void nfs_copy_fh(struct nfs_fh *target, const struct nfs_fh *source) +{ + target->size = source->size; + memcpy(target->data, source->data, source->size); +} + +/** + * nfs_fhandle_hash - Calculate the crc32 hash for the filehandle + * @fh: An NFS file handle to hash + * + * Return: a crc32 hash for the filehandle that is compatible with + * the one displayed by "wireshark" + */ +static inline u32 nfs_fhandle_hash(const struct nfs_fh *fh) +{ + return ~crc32_le(0xFFFFFFFF, &fh->data[0], fh->size); +} + +#endif /* _LINUX_NFS_FH_H */ diff --git a/include/linux/nfs_fs.h b/include/linux/nfs_fs.h index b85a73ae7919..d2c716322c6f 100644 --- a/include/linux/nfs_fs.h +++ b/include/linux/nfs_fs.h @@ -437,11 +437,11 @@ extern int nfs_refresh_inode(struct inode *, struct nfs_fattr *); extern int nfs_post_op_update_inode(struct inode *inode, struct nfs_fattr *fattr); extern int nfs_post_op_update_inode_force_wcc(struct inode *inode, struct nfs_fattr *fattr); extern int nfs_post_op_update_inode_force_wcc_locked(struct inode *inode, struct nfs_fattr *fattr); -extern int nfs_getattr(struct mnt_idmap *, const struct path *, +extern int nfs_getattr(const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); extern void nfs_access_add_cache(struct inode *, struct nfs_access_entry *, const struct cred *); extern void nfs_access_set_mask(struct nfs_access_entry *, u32); -extern int nfs_permission(struct mnt_idmap *, struct inode *, int); +extern int nfs_permission(const struct mnt_idmap *, struct inode *, int); extern int nfs_open(struct inode *, struct file *); extern int nfs_attribute_cache_expired(struct inode *inode); extern int nfs_revalidate_inode(struct inode *inode, unsigned long flags); @@ -450,7 +450,7 @@ extern int nfs_clear_invalid_mapping(struct address_space *mapping); extern bool nfs_mapping_need_revalidate_inode(struct inode *inode); extern int nfs_revalidate_mapping(struct inode *inode, struct address_space *mapping); extern int nfs_revalidate_mapping_rcu(struct inode *inode); -extern int nfs_setattr(struct mnt_idmap *, struct dentry *, struct iattr *); +extern int nfs_setattr(const struct mnt_idmap *, struct dentry *, struct iattr *); extern void nfs_setattr_update_inode(struct inode *inode, struct iattr *attr, struct nfs_fattr *); extern void nfs_setsecurity(struct inode *inode, struct nfs_fattr *fattr); extern struct nfs_open_context *get_nfs_open_context(struct nfs_open_context *ctx); diff --git a/include/linux/nfs_fs_sb.h b/include/linux/nfs_fs_sb.h index 34d294774f8c..416c6f39f31d 100644 --- a/include/linux/nfs_fs_sb.h +++ b/include/linux/nfs_fs_sb.h @@ -74,6 +74,8 @@ struct nfs_client { u64 cl_clientid; /* constant */ nfs4_verifier cl_confirm; /* Clientid verifier */ unsigned long cl_state; + /* bumped on each CB_NOTIFY_DEVICEID CHANGE for this client */ + atomic_t cl_deviceid_change_epoch; spinlock_t cl_lock; @@ -101,6 +103,8 @@ struct nfs_client { /* The flags used for obtaining the clientid during EXCHANGE_ID */ u32 cl_exchange_flags; struct nfs4_session *cl_session; /* shared session */ + /* CB_NOTIFY_DEVICEID DELETE suspects, protected by cl_lock */ + struct list_head cl_deviceid_deletes; bool cl_preserve_clid; struct nfs41_server_owner *cl_serverowner; struct nfs41_server_scope *cl_serverscope; @@ -248,6 +252,10 @@ struct nfs_server { that are supported on this filesystem */ struct pnfs_layoutdriver_type *pnfs_curr_ld; /* Active layout driver */ + unsigned int lg_reply_sz; /* Learned LAYOUTGET reply + buffer size, when the layout + driver's default has proved + too small */ struct rpc_wait_queue roc_rpcwaitq; /* the following fields are protected by nfs_client->cl_lock */ diff --git a/include/linux/nfs_page.h b/include/linux/nfs_page.h index 4b9a35dbc062..c38e4b380be5 100644 --- a/include/linux/nfs_page.h +++ b/include/linux/nfs_page.h @@ -38,6 +38,7 @@ enum { PG_REMOVE, /* page group sync bit in write path */ PG_CONTENDED1, /* Is someone waiting for a lock? */ PG_CONTENDED2, /* Is someone waiting for a lock? */ + PG_PINNED, /* page is pinned by GUP */ }; struct nfs_inode; @@ -58,6 +59,7 @@ struct nfs_page { struct nfs_page *wb_this_page; /* list of reqs for this page */ struct nfs_page *wb_head; /* head pointer for req list */ unsigned short wb_nio; /* Number of I/O attempts */ + unsigned int wb_nr_pinned; /* Number of pinned pages */ }; struct nfs_pgio_mirror; @@ -125,15 +127,17 @@ struct nfs_pageio_descriptor { extern struct nfs_page *nfs_page_create_from_page(struct nfs_open_context *ctx, struct page *page, + bool pinned, unsigned int pgbase, loff_t offset, unsigned int count); extern struct nfs_page *nfs_page_create_from_folio(struct nfs_open_context *ctx, struct folio *folio, + bool pinned, unsigned int offset, unsigned int count); -extern void nfs_release_request(struct nfs_page *); - +void nfs_release_request(struct nfs_page *req); +void nfs_release_request_list(struct list_head *head); extern void nfs_pageio_init(struct nfs_pageio_descriptor *desc, struct inode *inode, diff --git a/include/linux/nfs_ssc.h b/include/linux/nfs_ssc.h index 22265b1ff080..c199ea23e7eb 100644 --- a/include/linux/nfs_ssc.h +++ b/include/linux/nfs_ssc.h @@ -2,80 +2,33 @@ /* * include/linux/nfs_ssc.h * + * NFSv4.2 server-to-server copy, NFS client side APIs + * * Author: Dai Ngo <dai.ngo@oracle.com> * * Copyright (c) 2020, Oracle and/or its affiliates. */ -#include <linux/nfs_fs.h> -#include <linux/sunrpc/svc.h> +#ifndef _LINUX_NFS_SSC_H +#define _LINUX_NFS_SSC_H -extern struct nfs_ssc_client_ops_tbl nfs_ssc_client_tbl; +#include <linux/nfs_fh.h> +#include <linux/nfs4.h> + +struct file; +struct vfsmount; -/* - * NFS_V4 - */ struct nfs4_ssc_client_ops { + struct module *owner; struct file *(*sco_open)(struct vfsmount *ss_mnt, struct nfs_fh *src_fh, nfs4_stateid *stateid); void (*sco_close)(struct file *filep); }; -/* - * NFS_FS - */ -struct nfs_ssc_client_ops { - void (*sco_sb_deactive)(struct super_block *sb); -}; - -struct nfs_ssc_client_ops_tbl { - const struct nfs4_ssc_client_ops *ssc_nfs4_ops; - const struct nfs_ssc_client_ops *ssc_nfs_ops; -}; - extern void nfs42_ssc_register_ops(void); extern void nfs42_ssc_unregister_ops(void); extern void nfs42_ssc_register(const struct nfs4_ssc_client_ops *ops); extern void nfs42_ssc_unregister(const struct nfs4_ssc_client_ops *ops); -#ifdef CONFIG_NFSD_V4_2_INTER_SSC -static inline struct file *nfs42_ssc_open(struct vfsmount *ss_mnt, - struct nfs_fh *src_fh, nfs4_stateid *stateid) -{ - if (nfs_ssc_client_tbl.ssc_nfs4_ops) - return (*nfs_ssc_client_tbl.ssc_nfs4_ops->sco_open)(ss_mnt, src_fh, stateid); - return ERR_PTR(-EIO); -} - -static inline void nfs42_ssc_close(struct file *filep) -{ - if (nfs_ssc_client_tbl.ssc_nfs4_ops) - (*nfs_ssc_client_tbl.ssc_nfs4_ops->sco_close)(filep); -} -#endif - -struct nfsd4_ssc_umount_item { - struct list_head nsui_list; - bool nsui_busy; - /* - * nsui_refcnt inited to 2, 1 on list and 1 for consumer. Entry - * is removed when refcnt drops to 1 and nsui_expire expires. - */ - refcount_t nsui_refcnt; - unsigned long nsui_expire; - struct vfsmount *nsui_vfsmount; - char nsui_ipaddr[RPC_MAX_ADDRBUFLEN + 1]; -}; - -/* - * NFS_FS - */ -extern void nfs_ssc_register(const struct nfs_ssc_client_ops *ops); -extern void nfs_ssc_unregister(const struct nfs_ssc_client_ops *ops); - -static inline void nfs_do_sb_deactive(struct super_block *sb) -{ - if (nfs_ssc_client_tbl.ssc_nfs_ops) - (*nfs_ssc_client_tbl.ssc_nfs_ops->sco_sb_deactive)(sb); -} +#endif /* _LINUX_NFS_SSC_H */ diff --git a/include/linux/nfs_xdr.h b/include/linux/nfs_xdr.h index 7ed8fdb930d6..c0e29b4dfa62 100644 --- a/include/linux/nfs_xdr.h +++ b/include/linux/nfs_xdr.h @@ -1693,6 +1693,7 @@ struct nfs_pgio_header { struct nfs_client *ds_clp; /* pNFS data server */ u32 ds_commit_idx; /* ds index if ds_clp is set */ u32 pgio_mirror_idx;/* mirror index in pgio layer */ + struct nfs4_deviceid_node *ds_dev; /* device node ref held across the I/O */ }; struct nfs_mds_commit_info { @@ -1731,6 +1732,7 @@ struct nfs_commit_data { struct nfs_open_context *context; struct pnfs_layout_segment *lseg; struct nfs_client *ds_clp; /* pNFS data server */ + struct nfs4_deviceid_node *ds_dev; /* device node ref held across the commit */ int ds_commit_index; loff_t lwb; const struct rpc_call_ops *mds_ops; diff --git a/include/linux/nfsd_ssc.h b/include/linux/nfsd_ssc.h new file mode 100644 index 000000000000..7001410f01c2 --- /dev/null +++ b/include/linux/nfsd_ssc.h @@ -0,0 +1,38 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +/* + * include/linux/nfsd_ssc.h + * + * NFSv4.2 server-to-server copy, NFS server side APIs + * + * Author: Dai Ngo <dai.ngo@oracle.com> + * + * Copyright (c) 2020, Oracle and/or its affiliates. + */ + +#ifndef _LINUX_NFSD_SSC_H +#define _LINUX_NFSD_SSC_H + +#include <linux/nfs_fh.h> +#include <linux/nfs4.h> + +struct file; +struct vfsmount; + +#if IS_ENABLED(CONFIG_NFS_V4_2_SSC_HELPER) +struct file *nfsd42_ssc_open(struct vfsmount *ss_mnt, struct nfs_fh *src_fh, + nfs4_stateid *stateid); +void nfsd42_ssc_close(struct file *filp); +#else +static inline struct file *nfsd42_ssc_open(struct vfsmount *ss_mnt, + struct nfs_fh *src_fh, + nfs4_stateid *stateid) +{ + return ERR_PTR(-EIO); +} + +static inline void nfsd42_ssc_close(struct file *filp) +{ +} +#endif + +#endif /* _LINUX_NFSD_SSC_H */ diff --git a/include/linux/nfslocalio.h b/include/linux/nfslocalio.h index 3d91043254e6..8ce4d978a636 100644 --- a/include/linux/nfslocalio.h +++ b/include/linux/nfslocalio.h @@ -13,9 +13,18 @@ #include <linux/uuid.h> #include <linux/sunrpc/clnt.h> #include <linux/sunrpc/svcauth.h> -#include <linux/nfs.h> +#include <linux/nfs_fh.h> + #include <net/net_namespace.h> +/* + * The LOCALIO program is entirely private to Linux and is NOT part of + * the uapi. + */ +#define NFS_LOCALIO_PROGRAM 400122 +#define LOCALIOPROC_NULL 0 +#define LOCALIOPROC_UUID_IS_LOCAL 1 + struct nfs_client; struct nfs_file_localio; diff --git a/include/linux/posix_acl.h b/include/linux/posix_acl.h index 62d497763e25..caf500bed993 100644 --- a/include/linux/posix_acl.h +++ b/include/linux/posix_acl.h @@ -74,20 +74,20 @@ extern int __posix_acl_create(struct posix_acl **, gfp_t, umode_t *); extern int __posix_acl_chmod(struct posix_acl **, gfp_t, umode_t); extern struct posix_acl *get_posix_acl(struct inode *, int); -int set_posix_acl(struct mnt_idmap *, struct dentry *, int, +int set_posix_acl(const struct mnt_idmap *, struct dentry *, int, struct posix_acl *); struct posix_acl *get_cached_acl_rcu(struct inode *inode, int type); struct posix_acl *posix_acl_clone(const struct posix_acl *acl, gfp_t flags); #ifdef CONFIG_FS_POSIX_ACL -int posix_acl_chmod(struct mnt_idmap *, struct dentry *, umode_t); +int posix_acl_chmod(const struct mnt_idmap *, struct dentry *, umode_t); extern int posix_acl_create(struct inode *, umode_t *, struct posix_acl **, struct posix_acl **); -int posix_acl_update_mode(struct mnt_idmap *, struct inode *, umode_t *, +int posix_acl_update_mode(const struct mnt_idmap *, struct inode *, umode_t *, struct posix_acl **); -int simple_set_acl(struct mnt_idmap *, struct dentry *, +int simple_set_acl(const struct mnt_idmap *, struct dentry *, struct posix_acl *, int); extern int simple_acl_create(struct inode *, struct inode *); @@ -96,7 +96,7 @@ void set_cached_acl(struct inode *inode, int type, struct posix_acl *acl); void forget_cached_acl(struct inode *inode, int type); void forget_all_cached_acls(struct inode *inode); int posix_acl_valid(struct user_namespace *, const struct posix_acl *); -int posix_acl_permission(struct mnt_idmap *, struct inode *, +int posix_acl_permission(const struct mnt_idmap *, struct inode *, const struct posix_acl *, int); static inline void cache_no_acl(struct inode *inode) @@ -105,16 +105,16 @@ static inline void cache_no_acl(struct inode *inode) inode->i_default_acl = NULL; } -int vfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int vfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl); -struct posix_acl *vfs_get_acl(struct mnt_idmap *idmap, +struct posix_acl *vfs_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name); -int vfs_remove_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int vfs_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name); int posix_acl_listxattr(struct inode *inode, char **buffer, ssize_t *remaining_size); #else -static inline int posix_acl_chmod(struct mnt_idmap *idmap, +static inline int posix_acl_chmod(const struct mnt_idmap *idmap, struct dentry *dentry, umode_t mode) { return 0; @@ -141,21 +141,21 @@ static inline void forget_all_cached_acls(struct inode *inode) { } -static inline int vfs_set_acl(struct mnt_idmap *idmap, +static inline int vfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, struct posix_acl *acl) { return -EOPNOTSUPP; } -static inline struct posix_acl *vfs_get_acl(struct mnt_idmap *idmap, +static inline struct posix_acl *vfs_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { return ERR_PTR(-EOPNOTSUPP); } -static inline int vfs_remove_acl(struct mnt_idmap *idmap, +static inline int vfs_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { return -EOPNOTSUPP; diff --git a/include/linux/quotaops.h b/include/linux/quotaops.h index f9c0f9d7c9d9..0c64ca674e77 100644 --- a/include/linux/quotaops.h +++ b/include/linux/quotaops.h @@ -20,7 +20,7 @@ static inline struct quota_info *sb_dqopt(struct super_block *sb) } /* i_rwsem must being held */ -static inline bool is_quota_modification(struct mnt_idmap *idmap, +static inline bool is_quota_modification(const struct mnt_idmap *idmap, struct inode *inode, struct iattr *ia) { return ((ia->ia_valid & ATTR_SIZE) || @@ -109,7 +109,7 @@ int dquot_set_dqblk(struct super_block *sb, struct kqid id, struct qc_dqblk *di); int __dquot_transfer(struct inode *inode, struct dquot **transfer_to); -int dquot_transfer(struct mnt_idmap *idmap, struct inode *inode, +int dquot_transfer(const struct mnt_idmap *idmap, struct inode *inode, struct iattr *iattr); static inline struct mem_dqinfo *sb_dqinfo(struct super_block *sb, int type) @@ -229,7 +229,7 @@ static inline void dquot_free_inode(struct inode *inode) { } -static inline int dquot_transfer(struct mnt_idmap *idmap, +static inline int dquot_transfer(const struct mnt_idmap *idmap, struct inode *inode, struct iattr *iattr) { return 0; diff --git a/include/linux/sched.h b/include/linux/sched.h index f45b7d43113c..87ed6705c427 100644 --- a/include/linux/sched.h +++ b/include/linux/sched.h @@ -1817,7 +1817,7 @@ extern struct pid __rcu *cad_pid; * I am cleaning dirty pages from some other bdi. */ #define PF_KTHREAD 0x00200000 /* I am a kernel thread */ #define PF_RANDOMIZE 0x00400000 /* Randomize virtual address space */ -#define PF__HOLE__00800000 0x00800000 +#define PF_NO_NOTIFY_SIGNAL 0x00800000 /* see no_notify_signal_save() */ #define PF__HOLE__01000000 0x01000000 #define PF__HOLE__02000000 0x02000000 #define PF_NO_SETAFFINITY 0x04000000 /* Userland is not allowed to meddle with cpus_mask */ diff --git a/include/linux/sched/signal.h b/include/linux/sched/signal.h index d45a5476b97d..d9419dc902f6 100644 --- a/include/linux/sched/signal.h +++ b/include/linux/sched/signal.h @@ -2,6 +2,7 @@ #ifndef _LINUX_SCHED_SIGNAL_H #define _LINUX_SCHED_SIGNAL_H +#include <linux/cleanup.h> #include <linux/rculist.h> #include <linux/signal.h> #include <linux/sched.h> @@ -79,9 +80,9 @@ struct core_thread { }; struct core_state { - atomic_t nr_threads; - struct core_thread dumper; - struct completion startup; + /* Threads the dumper still waits for. */ + atomic_t threads_remaining; + struct core_thread *tasks; }; /* @@ -384,14 +385,36 @@ static inline int task_sigpending(struct task_struct *p) return unlikely(test_tsk_thread_flag(p,TIF_SIGPENDING)); } +/* Prevent TIF_NOTIFY_SIGNAL from interrupting this task. */ +static inline unsigned int no_notify_signal_save(void) +{ + unsigned int flags = current->flags; + + current->flags |= PF_NO_NOTIFY_SIGNAL; + return flags; +} + +/* Restore the previous PF_NO_NOTIFY_SIGNAL state. */ +static inline void no_notify_signal_restore(unsigned int flags) +{ + current_restore_flags(flags, PF_NO_NOTIFY_SIGNAL); +} + +DEFINE_LOCK_GUARD_0(no_notify_signal, + _T->flags = no_notify_signal_save(), + no_notify_signal_restore(_T->flags), + unsigned int flags) + static inline int signal_pending(struct task_struct *p) { /* * TIF_NOTIFY_SIGNAL isn't really a signal, but it requires the same * behavior in terms of ensuring that we break out of wait loops - * so that notify signal callbacks can be processed. + * so that notify signal callbacks can be processed. Not for a task + * that asked not to be interrupted by it, see no_notify_signal_save(). */ - if (unlikely(test_tsk_thread_flag(p, TIF_NOTIFY_SIGNAL))) + if (unlikely(test_tsk_thread_flag(p, TIF_NOTIFY_SIGNAL)) && + likely(!(READ_ONCE(p->flags) & PF_NO_NOTIFY_SIGNAL))) return 1; return task_sigpending(p); } diff --git a/include/linux/security.h b/include/linux/security.h index 153e9043058f..f7ff72ff956b 100644 --- a/include/linux/security.h +++ b/include/linux/security.h @@ -185,11 +185,11 @@ extern int cap_capset(struct cred *new, const struct cred *old, extern int cap_bprm_creds_from_file(struct linux_binprm *bprm, const struct file *file); int cap_inode_setxattr(struct dentry *dentry, const char *name, const void *value, size_t size, int flags); -int cap_inode_removexattr(struct mnt_idmap *idmap, +int cap_inode_removexattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name); int cap_inode_need_killpriv(struct dentry *dentry); -int cap_inode_killpriv(struct mnt_idmap *idmap, struct dentry *dentry); -int cap_inode_getsecurity(struct mnt_idmap *idmap, +int cap_inode_killpriv(const struct mnt_idmap *idmap, struct dentry *dentry); +int cap_inode_getsecurity(const struct mnt_idmap *idmap, struct inode *inode, const char *name, void **buffer, bool alloc); extern int cap_mmap_addr(unsigned long addr); @@ -338,6 +338,7 @@ int security_binder_transfer_file(const struct cred *from, const struct cred *to, const struct file *file); int security_ptrace_access_check(struct task_struct *child, unsigned int mode); int security_ptrace_traceme(struct task_struct *parent); +int security_mem_foll_force(const struct cred *subject, bool opened_by_owner); int security_capget(const struct task_struct *target, kernel_cap_t *effective, kernel_cap_t *inheritable, @@ -405,7 +406,7 @@ int security_inode_init_security_anon(struct inode *inode, const struct qstr *name, const struct inode *context_inode); int security_inode_create(struct inode *dir, struct dentry *dentry, umode_t mode); -void security_inode_post_create_tmpfile(struct mnt_idmap *idmap, +void security_inode_post_create_tmpfile(const struct mnt_idmap *idmap, struct inode *inode); int security_inode_link(struct dentry *old_dentry, struct inode *dir, struct dentry *new_dentry); @@ -422,31 +423,31 @@ int security_inode_readlink(struct dentry *dentry); int security_inode_follow_link(struct dentry *dentry, struct inode *inode, bool rcu); int security_inode_permission(struct inode *inode, int mask); -int security_inode_setattr(struct mnt_idmap *idmap, +int security_inode_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); -void security_inode_post_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +void security_inode_post_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, int ia_valid); int security_inode_getattr(const struct path *path); -int security_inode_setxattr(struct mnt_idmap *idmap, +int security_inode_setxattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, const void *value, size_t size, int flags); -int security_inode_set_acl(struct mnt_idmap *idmap, +int security_inode_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl); void security_inode_post_set_acl(struct dentry *dentry, const char *acl_name, struct posix_acl *kacl); -int security_inode_get_acl(struct mnt_idmap *idmap, +int security_inode_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name); -int security_inode_remove_acl(struct mnt_idmap *idmap, +int security_inode_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name); -void security_inode_post_remove_acl(struct mnt_idmap *idmap, +void security_inode_post_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name); void security_inode_post_setxattr(struct dentry *dentry, const char *name, const void *value, size_t size, int flags); int security_inode_getxattr(struct dentry *dentry, const char *name); int security_inode_listxattr(struct dentry *dentry); -int security_inode_removexattr(struct mnt_idmap *idmap, +int security_inode_removexattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name); void security_inode_post_removexattr(struct dentry *dentry, const char *name); int security_inode_file_setattr(struct dentry *dentry, @@ -454,8 +455,8 @@ int security_inode_file_setattr(struct dentry *dentry, int security_inode_file_getattr(struct dentry *dentry, struct file_kattr *fa); int security_inode_need_killpriv(struct dentry *dentry); -int security_inode_killpriv(struct mnt_idmap *idmap, struct dentry *dentry); -int security_inode_getsecurity(struct mnt_idmap *idmap, +int security_inode_killpriv(const struct mnt_idmap *idmap, struct dentry *dentry); +int security_inode_getsecurity(const struct mnt_idmap *idmap, struct inode *inode, const char *name, void **buffer, bool alloc); int security_inode_setsecurity(struct inode *inode, const char *name, const void *value, size_t size, int flags); @@ -676,6 +677,12 @@ static inline int security_ptrace_traceme(struct task_struct *parent) return cap_ptrace_traceme(parent); } +static inline int security_mem_foll_force(const struct cred *subject, + bool opened_by_owner) +{ + return 0; +} + static inline int security_capget(const struct task_struct *target, kernel_cap_t *effective, kernel_cap_t *inheritable, @@ -910,7 +917,7 @@ static inline int security_inode_create(struct inode *dir, } static inline void -security_inode_post_create_tmpfile(struct mnt_idmap *idmap, struct inode *inode) +security_inode_post_create_tmpfile(const struct mnt_idmap *idmap, struct inode *inode) { } static inline int security_inode_link(struct dentry *old_dentry, @@ -979,7 +986,7 @@ static inline int security_inode_permission(struct inode *inode, int mask) return 0; } -static inline int security_inode_setattr(struct mnt_idmap *idmap, +static inline int security_inode_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { @@ -987,7 +994,7 @@ static inline int security_inode_setattr(struct mnt_idmap *idmap, } static inline void -security_inode_post_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +security_inode_post_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, int ia_valid) { } @@ -996,14 +1003,14 @@ static inline int security_inode_getattr(const struct path *path) return 0; } -static inline int security_inode_setxattr(struct mnt_idmap *idmap, +static inline int security_inode_setxattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, const void *value, size_t size, int flags) { return cap_inode_setxattr(dentry, name, value, size, flags); } -static inline int security_inode_set_acl(struct mnt_idmap *idmap, +static inline int security_inode_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl) @@ -1016,21 +1023,21 @@ static inline void security_inode_post_set_acl(struct dentry *dentry, struct posix_acl *kacl) { } -static inline int security_inode_get_acl(struct mnt_idmap *idmap, +static inline int security_inode_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { return 0; } -static inline int security_inode_remove_acl(struct mnt_idmap *idmap, +static inline int security_inode_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { return 0; } -static inline void security_inode_post_remove_acl(struct mnt_idmap *idmap, +static inline void security_inode_post_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { } @@ -1050,7 +1057,7 @@ static inline int security_inode_listxattr(struct dentry *dentry) return 0; } -static inline int security_inode_removexattr(struct mnt_idmap *idmap, +static inline int security_inode_removexattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name) { @@ -1078,13 +1085,13 @@ static inline int security_inode_need_killpriv(struct dentry *dentry) return cap_inode_need_killpriv(dentry); } -static inline int security_inode_killpriv(struct mnt_idmap *idmap, +static inline int security_inode_killpriv(const struct mnt_idmap *idmap, struct dentry *dentry) { return cap_inode_killpriv(idmap, dentry); } -static inline int security_inode_getsecurity(struct mnt_idmap *idmap, +static inline int security_inode_getsecurity(const struct mnt_idmap *idmap, struct inode *inode, const char *name, void **buffer, bool alloc) @@ -2085,7 +2092,7 @@ int security_path_mkdir(const struct path *dir, struct dentry *dentry, umode_t m int security_path_rmdir(const struct path *dir, struct dentry *dentry); int security_path_mknod(const struct path *dir, struct dentry *dentry, umode_t mode, unsigned int dev); -void security_path_post_mknod(struct mnt_idmap *idmap, struct dentry *dentry); +void security_path_post_mknod(const struct mnt_idmap *idmap, struct dentry *dentry); int security_path_truncate(const struct path *path); int security_path_symlink(const struct path *dir, struct dentry *dentry, const char *old_name); @@ -2120,7 +2127,7 @@ static inline int security_path_mknod(const struct path *dir, struct dentry *den return 0; } -static inline void security_path_post_mknod(struct mnt_idmap *idmap, +static inline void security_path_post_mknod(const struct mnt_idmap *idmap, struct dentry *dentry) { } diff --git a/include/linux/splice.h b/include/linux/splice.h index 9dec4861d09f..0e6c955dc6ff 100644 --- a/include/linux/splice.h +++ b/include/linux/splice.h @@ -79,8 +79,8 @@ ssize_t add_to_pipe(struct pipe_inode_info *pipe, struct pipe_buffer *buf); ssize_t vfs_splice_read(struct file *in, loff_t *ppos, struct pipe_inode_info *pipe, size_t len, unsigned int flags); -ssize_t splice_direct_to_actor(struct file *file, struct splice_desc *sd, - splice_direct_actor *actor); +ssize_t vfs_splice_to_actor(struct file *file, loff_t pos, size_t count, + splice_direct_actor *actor, void *private); ssize_t do_splice(struct file *in, loff_t *off_in, struct file *out, loff_t *off_out, size_t len, unsigned int flags); ssize_t do_splice_direct(struct file *in, loff_t *ppos, struct file *out, diff --git a/include/linux/sunrpc/svc_xprt.h b/include/linux/sunrpc/svc_xprt.h index da2a2531e110..2af222f3ea2c 100644 --- a/include/linux/sunrpc/svc_xprt.h +++ b/include/linux/sunrpc/svc_xprt.h @@ -37,6 +37,9 @@ struct svc_xprt_class { struct list_head xcl_list; u32 xcl_max_payload; int xcl_ident; + u32 xcl_flags; +/* Set only on classes whose xpo_has_wspace() reads xpt_reserved */ +#define SVC_XPRT_FLAG_WSPACE_RESERVE BIT(0) }; /* @@ -59,7 +62,7 @@ struct svc_xprt { unsigned long xpt_flags; struct svc_serv *xpt_server; /* service for transport */ - atomic_t xpt_reserved; /* space on outq that is rsvd */ + atomic_t xpt_reserved; /* outq space rsvd, UDP only */ atomic_t xpt_nr_rqsts; /* Number of requests */ struct mutex xpt_mutex; /* to serialize sending data */ spinlock_t xpt_lock; /* protects sk_deferred diff --git a/include/linux/uidgid.h b/include/linux/uidgid.h index 2dc767e08f54..02403629b49f 100644 --- a/include/linux/uidgid.h +++ b/include/linux/uidgid.h @@ -130,9 +130,9 @@ static inline bool kgid_has_mapping(struct user_namespace *ns, kgid_t gid) return from_kgid(ns, gid) != (gid_t) -1; } -u32 map_id_down(struct uid_gid_map *map, u32 id); -u32 map_id_up(struct uid_gid_map *map, u32 id); -u32 map_id_range_up(struct uid_gid_map *map, u32 id, u32 count); +u32 map_id_down(const struct uid_gid_map *map, u32 id); +u32 map_id_up(const struct uid_gid_map *map, u32 id); +u32 map_id_range_up(const struct uid_gid_map *map, u32 id, u32 count); #else @@ -182,17 +182,17 @@ static inline bool kgid_has_mapping(struct user_namespace *ns, kgid_t gid) return gid_valid(gid); } -static inline u32 map_id_down(struct uid_gid_map *map, u32 id) +static inline u32 map_id_down(const struct uid_gid_map *map, u32 id) { return id; } -static inline u32 map_id_range_up(struct uid_gid_map *map, u32 id, u32 count) +static inline u32 map_id_range_up(const struct uid_gid_map *map, u32 id, u32 count) { return id; } -static inline u32 map_id_up(struct uid_gid_map *map, u32 id) +static inline u32 map_id_up(const struct uid_gid_map *map, u32 id) { return id; } diff --git a/include/linux/user_namespace.h b/include/linux/user_namespace.h index e38d9e60569f..91232053775d 100644 --- a/include/linux/user_namespace.h +++ b/include/linux/user_namespace.h @@ -29,8 +29,8 @@ struct uid_gid_map { /* 64 bytes -- 1 cache line */ u32 nr_extents; }; struct { - struct uid_gid_extent *forward; - struct uid_gid_extent *reverse; + struct uid_gid_extent *forward __counted_by_ptr(nr_extents); + struct uid_gid_extent *reverse __counted_by_ptr(nr_extents); }; }; }; @@ -207,6 +207,13 @@ extern bool in_userns(const struct user_namespace *ancestor, const struct user_namespace *child); extern bool current_in_userns(const struct user_namespace *target_ns); struct ns_common *ns_get_owner(struct ns_common *ns); + +#if IS_ENABLED(CONFIG_KUNIT) +extern int uid_gid_map_insert_extent(struct uid_gid_map *map, + struct uid_gid_extent *extent); +extern int uid_gid_map_sort(struct uid_gid_map *map); +#endif /* CONFIG_KUNIT */ + #else static inline struct user_namespace *get_user_ns(struct user_namespace *ns) diff --git a/include/linux/wait_bit.h b/include/linux/wait_bit.h index 553d7b23e3ad..af077ed4caf6 100644 --- a/include/linux/wait_bit.h +++ b/include/linux/wait_bit.h @@ -433,6 +433,32 @@ do { \ }) /** + * wait_var_event_state - wait for a variable to be updated and notified + * @var: the address of variable being waited on + * @condition: the condition to wait for + * @state: the task state to sleep in, %TASK_UNINTERRUPTIBLE etc. + * + * Wait for a @condition to be true, only re-checking when a wake up is + * received for the given @var (an arbitrary kernel address which need + * not be directly related to the given condition, but usually is). + * + * Returns 0 if the condition became true, or %-ERESTARTSYS if a signal + * arrived which @state allows to interrupt. + * + * The condition should normally use smp_load_acquire() or a similarly + * ordered access to ensure that any changes to memory made before the + * condition became true will be visible after the wait completes. + */ +#define wait_var_event_state(var, condition, state) \ +({ \ + int __ret = 0; \ + might_sleep(); \ + if (!(condition)) \ + __ret = ___wait_var_event(var, condition, (state), 0, 0, schedule()); \ + __ret; \ +}) + +/** * wait_var_event_any_lock - wait for a variable to be updated under a lock * @var: the address of the variable being waited on * @condition: condition to wait for diff --git a/include/linux/xattr.h b/include/linux/xattr.h index 54ac3cbc133f..4cc4257de084 100644 --- a/include/linux/xattr.h +++ b/include/linux/xattr.h @@ -47,7 +47,7 @@ struct xattr_handler { struct inode *inode, const char *name, void *buffer, size_t size); int (*set)(const struct xattr_handler *, - struct mnt_idmap *idmap, struct dentry *dentry, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *buffer, size_t size, int flags); }; @@ -77,25 +77,25 @@ struct xattr { }; ssize_t __vfs_getxattr(struct dentry *, struct inode *, const char *, void *, size_t); -ssize_t vfs_getxattr(struct mnt_idmap *, struct dentry *, const char *, +ssize_t vfs_getxattr(const struct mnt_idmap *, struct dentry *, const char *, void *, size_t); ssize_t vfs_listxattr(struct dentry *d, char *list, size_t size); -int __vfs_setxattr(struct mnt_idmap *, struct dentry *, struct inode *, +int __vfs_setxattr(const struct mnt_idmap *, struct dentry *, struct inode *, const char *, const void *, size_t, int); -int __vfs_setxattr_noperm(struct mnt_idmap *, struct dentry *, +int __vfs_setxattr_noperm(const struct mnt_idmap *, struct dentry *, const char *, const void *, size_t, int); -int __vfs_setxattr_locked(struct mnt_idmap *, struct dentry *, +int __vfs_setxattr_locked(const struct mnt_idmap *, struct dentry *, const char *, const void *, size_t, int, struct delegated_inode *); -int vfs_setxattr(struct mnt_idmap *, struct dentry *, const char *, +int vfs_setxattr(const struct mnt_idmap *, struct dentry *, const char *, const void *, size_t, int); -int __vfs_removexattr(struct mnt_idmap *, struct dentry *, const char *); -int __vfs_removexattr_locked(struct mnt_idmap *, struct dentry *, +int __vfs_removexattr(const struct mnt_idmap *, struct dentry *, const char *); +int __vfs_removexattr_locked(const struct mnt_idmap *, struct dentry *, const char *, struct delegated_inode *); -int vfs_removexattr(struct mnt_idmap *, struct dentry *, const char *); +int vfs_removexattr(const struct mnt_idmap *, struct dentry *, const char *); ssize_t generic_listxattr(struct dentry *dentry, char *buffer, size_t buffer_size); -int vfs_getxattr_alloc(struct mnt_idmap *idmap, +int vfs_getxattr_alloc(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, char **xattr_value, size_t size, gfp_t flags); |
